{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nimport matplotlib\nfrom tabulate import tabulate\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nimport gc\n\nfrom sklearn.model_selection import GroupKFold, RepeatedKFold\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, PowerTransformer\nfrom sklearn.metrics import roc_auc_score\n\nfrom imblearn.over_sampling import SMOTE\n\nimport optuna\n\nimport lightgbm as lgbm\nfrom catboost import CatBoostRegressor, CatBoostClassifier\nimport xgboost as xgb\n\nfrom sklearn.linear_model import LogisticRegression","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-06T12:42:40.963679Z","iopub.execute_input":"2022-08-06T12:42:40.964203Z","iopub.status.idle":"2022-08-06T12:42:44.501743Z","shell.execute_reply.started":"2022-08-06T12:42:40.964161Z","shell.execute_reply":"2022-08-06T12:42:44.500438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.core.display import display, HTML, Javascript\n\n# ----- Notebook Theme -----\ncolor_map = ['#f4a261', '#e8f6f3', '#d0ece7', '#a2d9ce', '#73c6b6', '#45b39d', \n                        '#16a085', '#138d75', '#117a65', '#0e6655', '#e76f51']\n\nprompt = color_map[-1]\nmain_color = color_map[0]\nstrong_main_color = color_map[1]\ncustom_colors = [strong_main_color, main_color]\n\ncss_file = ''' \n\ndiv #notebook {\nbackground-color: white;\nline-height: 20px;\n}\n\n#notebook-container {\n%s\nmargin-top: 2em;\npadding-top: 2em;\nborder-top: 4px solid %s; /* light orange */\n-webkit-box-shadow: 0px 0px 8px 2px rgba(224, 212, 226, 0.5); /* pink */\n    box-shadow: 0px 0px 8px 2px rgba(224, 212, 226, 0.5); /* pink */\n}\n\ndiv .input {\nmargin-bottom: 1em;\n}\n\n.rendered_html h1, .rendered_html h2, .rendered_html h3, .rendered_html h4, .rendered_html h5, .rendered_html h6 {\ncolor: %s; /* light orange */\nfont-weight: 600;\n}\n\ndiv.input_area {\nborder: none;\n    background-color: %s; /* rgba(229, 143, 101, 0.1); light orange [exactly #E58F65] */\n    border-top: 2px solid %s; /* light orange */\n}\n\ndiv.input_prompt {\ncolor: %s; /* light blue */\n}\n\ndiv.output_prompt {\ncolor: %s; /* strong orange */\n}\n\ndiv.cell.selected:before, div.cell.selected.jupyter-soft-selected:before {\nbackground: %s; /* light orange */\n}\n\ndiv.cell.selected, div.cell.selected.jupyter-soft-selected {\n    border-color: %s; /* light orange */\n}\n\n.edit_mode div.cell.selected:before {\nbackground: %s; /* light orange */\n}\n\n.edit_mode div.cell.selected {\nborder-color: %s; /* light orange */\n\n}\n'''\ndef to_rgb(h): \n    return tuple(int(h[i:i+2], 16) for i in [0, 2, 4])\n\nmain_color_rgba = 'rgba(%s, %s, %s, 0.1)' % (to_rgb(main_color[1:]))\nopen('notebook.css', 'w').write(css_file % ('width: 95%;', main_color, main_color, main_color_rgba, main_color,  main_color, prompt, main_color, main_color, main_color, main_color))\n\ndef nb(): \n    return HTML(\"<style>\" + open(\"notebook.css\", \"r\").read() + \"</style>\")\nnb()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T07:17:13.187316Z","iopub.execute_input":"2022-08-03T07:17:13.187726Z","iopub.status.idle":"2022-08-03T07:17:13.208240Z","shell.execute_reply.started":"2022-08-03T07:17:13.187691Z","shell.execute_reply":"2022-08-03T07:17:13.206816Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:#f4a261;\">***🔥 TPS aug 22 🔥 Advanced EDA***<span>\n\n\n### <span style=\"color:#f4a261;\">*Table of content*<span>\n<a id=\"table-of-contents\"></a>\n- [1. Read data](#1)\n- [2.1 Quick view](#2.1)\n- [2.2 Basic statistics](#2.2)\n- [2.3 Target column](#2.3)\n- [3. Missing values](#3)\n- [4. Unique values](#4)\n- [5. Data Distributions](#5)\n- [6. Feature correlation](#6)","metadata":{}},{"cell_type":"markdown","source":"[back to top](#table-of-contents)\n<a id=\"1\"></a>\n### **<span style=\"color:#f4a261;\">1. Read data</span>**\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T12:21:06.220818Z","iopub.execute_input":"2022-08-01T12:21:06.221338Z","iopub.status.idle":"2022-08-01T12:21:06.230524Z","shell.execute_reply.started":"2022-08-01T12:21:06.221294Z","shell.execute_reply":"2022-08-01T12:21:06.228873Z"}}},{"cell_type":"code","source":"train_df = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ntest_df = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')\nssub = pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:42:45.097848Z","iopub.execute_input":"2022-08-06T12:42:45.098870Z","iopub.status.idle":"2022-08-06T12:42:45.529302Z","shell.execute_reply.started":"2022-08-06T12:42:45.098810Z","shell.execute_reply":"2022-08-06T12:42:45.527986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[back to top](#table-of-contents)\n<a id=\"2.1\"></a>\n### **<span style=\"color:#f4a261;\">2.1 Quick view</span>**\n#### **<span style=\"color:#f4a261;\">train df</span>**","metadata":{}},{"cell_type":"code","source":"train_df.head(5)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T07:17:13.529729Z","iopub.execute_input":"2022-08-03T07:17:13.530267Z","iopub.status.idle":"2022-08-03T07:17:13.565148Z","shell.execute_reply.started":"2022-08-03T07:17:13.530216Z","shell.execute_reply":"2022-08-03T07:17:13.563894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### **<span style=\"color:#f4a261;\">test df</span>**","metadata":{}},{"cell_type":"code","source":"test_df.head(5)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T07:17:13.566831Z","iopub.execute_input":"2022-08-03T07:17:13.567953Z","iopub.status.idle":"2022-08-03T07:17:13.600988Z","shell.execute_reply.started":"2022-08-03T07:17:13.567903Z","shell.execute_reply":"2022-08-03T07:17:13.599788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[back to top](#table-of-contents)\n<a id=\"2.2\"></a>\n### **<span style=\"color:#f4a261;\">2.2 Basic statistics</span>**\n#### **<span style=\"color:#f4a261;\">train df</span>**","metadata":{}},{"cell_type":"code","source":"train_df.describe()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T07:17:13.604221Z","iopub.execute_input":"2022-08-03T07:17:13.604866Z","iopub.status.idle":"2022-08-03T07:17:13.729289Z","shell.execute_reply.started":"2022-08-03T07:17:13.604797Z","shell.execute_reply":"2022-08-03T07:17:13.727893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### **<span style=\"color:#f4a261;\">test df</span>**","metadata":{}},{"cell_type":"code","source":"test_df.describe()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T07:17:13.730887Z","iopub.execute_input":"2022-08-03T07:17:13.731563Z","iopub.status.idle":"2022-08-03T07:17:13.838211Z","shell.execute_reply.started":"2022-08-03T07:17:13.731527Z","shell.execute_reply":"2022-08-03T07:17:13.837050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[back to top](#table-of-contents)\n<a id=\"2.3\"></a>\n### **<span style=\"color:#f4a261;\">2.3 Target column</span>**","metadata":{}},{"cell_type":"code","source":"print('Target column basic statistics:')\ntrain_df['failure'].describe()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T07:17:13.840365Z","iopub.execute_input":"2022-08-03T07:17:13.840722Z","iopub.status.idle":"2022-08-03T07:17:13.854495Z","shell.execute_reply.started":"2022-08-03T07:17:13.840691Z","shell.execute_reply":"2022-08-03T07:17:13.852955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Frequency of each target classes:')\ntrain_df['failure'].value_counts()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T07:17:13.856140Z","iopub.execute_input":"2022-08-03T07:17:13.856487Z","iopub.status.idle":"2022-08-03T07:17:13.867105Z","shell.execute_reply.started":"2022-08-03T07:17:13.856456Z","shell.execute_reply":"2022-08-03T07:17:13.865763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### **<span style=\"color:#f4a261;\">Target balance</span>**","metadata":{}},{"cell_type":"code","source":"plt.subplots(figsize=(10, 10), facecolor='#f6f5f5')\nplt.pie(train_df.failure.value_counts(), startangle=90, wedgeprops={'width':0.3}, colors=['#ff8826', '#ffd514'] )\nplt.text(0, 0, f\"{train_df.failure.value_counts()[0] / train_df.failure.count() * 100:.2f}%\", ha='center', va='center', fontweight='bold', fontsize=42, color='#ff8826')\nplt.legend(train_df.failure.value_counts().index, ncol=2, facecolor='#f6f5f5', edgecolor='#f6f5f5', loc='lower center', fontsize=16)\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T07:17:13.868397Z","iopub.execute_input":"2022-08-03T07:17:13.868762Z","iopub.status.idle":"2022-08-03T07:17:16.213807Z","shell.execute_reply.started":"2022-08-03T07:17:13.868731Z","shell.execute_reply":"2022-08-03T07:17:16.212553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[back to top](#table-of-contents)\n<a id=\"3\"></a>\n### **<span style=\"color:#f4a261;\">3. Missing values</span>**","metadata":{}},{"cell_type":"code","source":"integer_features = [col for col in train_df.columns]\n\nunique_values_train = pd.DataFrame(train_df.notna().sum())\nunique_values_train = unique_values_train.reset_index(drop=False)\nunique_values_train.columns = ['Features', 'Count']\n\nunique_values_percent_train = pd.DataFrame(train_df.notna().sum()/train_df.shape[0])\nunique_values_percent_train = unique_values_percent_train.reset_index(drop=False)\nunique_values_percent_train.columns = ['Features', 'Count']\n\nunique_values_test = pd.DataFrame(test_df.notna().sum())\nunique_values_test = unique_values_test.reset_index(drop=False)\nunique_values_test.columns = ['Features', 'Count']\n\nunique_values_percent_test = pd.DataFrame(test_df.notna().sum()/test_df.shape[0])\nunique_values_percent_test = unique_values_percent_test.reset_index(drop=False)\nunique_values_percent_test.columns = ['Features', 'Count']\n\nplt.rcParams['figure.dpi'] = 600\nfig = plt.figure(figsize=(8, 10), facecolor='#f6f5f5')\ngs = fig.add_gridspec(2, 2)\ngs.update(wspace=0.4, hspace=0.2)\n\nbackground_color = \"#f6f5f5\"\nsns.set_palette(['#ffd514']*int(len(integer_features)))\n\nax0 = fig.add_subplot(gs[0, 0])\nfor s in [\"right\", \"top\"]:\n    ax0.spines[s].set_visible(False)\nax0.set_facecolor(background_color)\nax0_sns = sns.barplot(ax=ax0, y=unique_values_train['Features'], x=unique_values_train['Count'], \n                      zorder=2, linewidth=0, orient='h', saturation=1, alpha=1)\nax0_sns.set_xlabel(\"Unique Values\",fontsize=4, weight='bold')\nax0_sns.set_ylabel(\"Features\",fontsize=4, weight='bold')\nax0_sns.tick_params(labelsize=4, width=0.5, length=1.5)\nax0_sns.grid(which='major', axis='x', zorder=0, color='#EEEEEE', linewidth=0.4)\nax0_sns.grid(which='major', axis='y', zorder=0, color='#EEEEEE', linewidth=0.4)\nax0.text(0, -1.7, 'Not Missing Values - Train Dataset', fontsize=6, ha='left', va='top', weight='bold')\nax0.text(0, -1, 'Measurement 1-17 depend on each other sequentially', fontsize=4, ha='left', va='top')\nax0.get_xaxis().set_major_formatter(matplotlib.ticker.FuncFormatter(lambda x, p: format(int(x), ',')))\n# data label\nfor p in ax0.patches:\n    value = f'{p.get_width():,.0f}'\n    x = p.get_x() + p.get_width() + 500\n    y = p.get_y() + p.get_height() / 2 \n    ax0.text(x, y, value, ha='left', va='center', fontsize=4, \n            bbox=dict(facecolor='none', edgecolor='black', boxstyle='round', linewidth=0.3))\n    \nax1 = fig.add_subplot(gs[0, 1])\nfor s in [\"right\", \"top\"]:\n    ax1.spines[s].set_visible(False)\nax1.set_facecolor(background_color)\nax1_sns = sns.barplot(ax=ax1, y=unique_values_percent_train['Features'], x=unique_values_percent_train['Count'], \n                      zorder=2, linewidth=0, orient='h', saturation=1, alpha=1)\nax1_sns.set_xlabel(\"Percentage Unique Values\",fontsize=4, weight='bold')\nax1_sns.set_ylabel(\"Features\",fontsize=4, weight='bold')\nax1_sns.tick_params(labelsize=4, width=0.5, length=1.5)\nax1_sns.grid(which='major', axis='x', zorder=0, color='#EEEEEE', linewidth=0.4)\nax1_sns.grid(which='major', axis='y', zorder=0, color='#EEEEEE', linewidth=0.4)\nax1.text(0, -1.7, 'Percentage Not Missing Values - Train Dataset', fontsize=6, ha='left', va='top', weight='bold')\nax1.text(0, -1, 'Measurement 1-17 depend on each other sequentially', fontsize=4, ha='left', va='top')\n# data label\nfor p in ax1.patches:\n    value = f'{p.get_width():.2f}'\n    x = p.get_x() + p.get_width() + 0.03\n    y = p.get_y() + p.get_height() / 2 \n    ax1.text(x, y, value, ha='left', va='center', fontsize=4, \n            bbox=dict(facecolor='none', edgecolor='black', boxstyle='round', linewidth=0.3))\n\nbackground_color = \"#f6f5f5\"\nsns.set_palette(['#ff355d']*int(len(integer_features)))\n    \nax3 = fig.add_subplot(gs[1, 0])\nfor s in [\"right\", \"top\"]:\n    ax3.spines[s].set_visible(False)\nax3.set_facecolor(background_color)\nax3_sns = sns.barplot(ax=ax3, y=unique_values_test['Features'], x=unique_values_test['Count'], \n                      zorder=2, linewidth=0, orient='h', saturation=1, alpha=1)\nax3_sns.set_xlabel(\"Unique Values\",fontsize=4, weight='bold')\nax3_sns.set_ylabel(\"Features\",fontsize=4, weight='bold')\nax3_sns.tick_params(labelsize=4, width=0.5, length=1.5)\nax3_sns.grid(which='major', axis='x', zorder=0, color='#EEEEEE', linewidth=0.4)\nax3_sns.grid(which='major', axis='y', zorder=0, color='#EEEEEE', linewidth=0.4)\nax3.text(0, -1.7, 'Not Missing Values - Test Dataset', fontsize=6, ha='left', va='top', weight='bold')\nax3.text(0, -1, 'Test dataset is quite similar with train dataset', fontsize=4, ha='left', va='top')\nax3.get_xaxis().set_major_formatter(matplotlib.ticker.FuncFormatter(lambda x, p: format(int(x), ',')))\n# data label\nfor p in ax3.patches:\n    value = f'{p.get_width():,.0f}'\n    x = p.get_x() + p.get_width() + 500\n    y = p.get_y() + p.get_height() / 2 \n    ax3.text(x, y, value, ha='left', va='center', fontsize=4, \n            bbox=dict(facecolor='none', edgecolor='black', boxstyle='round', linewidth=0.3))\n    \nax4 = fig.add_subplot(gs[1, 1])\nfor s in [\"right\", \"top\"]:\n    ax4.spines[s].set_visible(False)\nax4.set_facecolor(background_color)\nax4_sns = sns.barplot(ax=ax4, y=unique_values_percent_test['Features'], x=unique_values_percent_test['Count'], \n                      zorder=2, linewidth=0, orient='h', saturation=1, alpha=1)\nax4_sns.set_xlabel(\"Percentage Unique Values\",fontsize=4, weight='bold')\nax4_sns.set_ylabel(\"Features\",fontsize=4, weight='bold')\nax4_sns.tick_params(labelsize=4, width=0.5, length=1.5)\nax4_sns.grid(which='major', axis='x', zorder=0, color='#EEEEEE', linewidth=0.4)\nax4_sns.grid(which='major', axis='y', zorder=0, color='#EEEEEE', linewidth=0.4)\nax4.text(0, -1.7, 'Percentage Not Missing Values - Test Dataset', fontsize=6, ha='left', va='top', weight='bold')\nax4.text(0, -1, 'Test dataset is quite similar with train dataset', fontsize=4, ha='left', va='top')\n# data label\nfor p in ax4.patches:\n    value = f'{p.get_width():.2f}'\n    x = p.get_x() + p.get_width() + 0.03\n    y = p.get_y() + p.get_height() / 2 \n    ax4.text(x, y, value, ha='left', va='center', fontsize=4, \n            bbox=dict(facecolor='none', edgecolor='black', boxstyle='round', linewidth=0.3))\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T07:17:16.215865Z","iopub.execute_input":"2022-08-03T07:17:16.216751Z","iopub.status.idle":"2022-08-03T07:17:20.641782Z","shell.execute_reply.started":"2022-08-03T07:17:16.216696Z","shell.execute_reply":"2022-08-03T07:17:20.640634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[back to top](#table-of-contents)\n<a id=\"4\"></a>\n### **<span style=\"color:#f4a261;\">4. Unique values</span>**","metadata":{}},{"cell_type":"code","source":"integer_features = [ col for col in test_df._get_numeric_data().columns if col not in ['id']]\n\nunique_values_train = pd.DataFrame(train_df[integer_features].nunique())\nunique_values_train = unique_values_train.reset_index(drop=False)\nunique_values_train.columns = ['Features', 'Count']\nunique_values_train = unique_values_train.sort_values(by='Count', ascending=False)\n\nunique_values_percent_train = pd.DataFrame(train_df[integer_features].nunique()/train_df.shape[0])\nunique_values_percent_train = unique_values_percent_train.reset_index(drop=False)\nunique_values_percent_train.columns = ['Features', 'Count']\nunique_values_percent_train = unique_values_percent_train.sort_values(by='Count', ascending=False)\n\nunique_values_test = pd.DataFrame(test_df[integer_features].nunique())\nunique_values_test = unique_values_test.reset_index(drop=False)\nunique_values_test.columns = ['Features', 'Count']\nunique_values_test = unique_values_test.sort_values(by='Count', ascending=False)\n\nunique_values_percent_test = pd.DataFrame(test_df[integer_features].nunique()/test_df.shape[0])\nunique_values_percent_test = unique_values_percent_test.reset_index(drop=False)\nunique_values_percent_test.columns = ['Features', 'Count']\nunique_values_percent_test = unique_values_percent_test.sort_values(by='Count', ascending=False)\n\nplt.rcParams['figure.dpi'] = 600\nfig = plt.figure(figsize=(8, 10), facecolor='#f6f5f5')\ngs = fig.add_gridspec(2, 2)\ngs.update(wspace=0.4, hspace=0.2)\n\nbackground_color = \"#f6f5f5\"\nsns.set_palette(['#ffd514']*int(len(integer_features)))\n\nax0 = fig.add_subplot(gs[0, 0])\nfor s in [\"right\", \"top\"]:\n    ax0.spines[s].set_visible(False)\nax0.set_facecolor(background_color)\nax0_sns = sns.barplot(ax=ax0, y=unique_values_train['Features'], x=unique_values_train['Count'], \n                      zorder=2, linewidth=0, orient='h', saturation=1, alpha=1)\nax0_sns.set_xlabel(\"Unique Values\",fontsize=4, weight='bold')\nax0_sns.set_ylabel(\"Features\",fontsize=4, weight='bold')\nax0_sns.tick_params(labelsize=4, width=0.5, length=1.5)\nax0_sns.grid(which='major', axis='x', zorder=0, color='#EEEEEE', linewidth=0.4)\nax0_sns.grid(which='major', axis='y', zorder=0, color='#EEEEEE', linewidth=0.4)\nax0.text(0, -1.7, 'Unique Values - Train Dataset', fontsize=6, ha='left', va='top', weight='bold')\nax0.text(0, -1, 'attribute_2 and attribute_3 can be considered as classification features', fontsize=4, ha='left', va='top')\nax0.get_xaxis().set_major_formatter(matplotlib.ticker.FuncFormatter(lambda x, p: format(int(x), ',')))\n# data label\nfor p in ax0.patches:\n    value = f'{p.get_width():,.0f}'\n    x = p.get_x() + p.get_width() + 500\n    y = p.get_y() + p.get_height() / 2 \n    ax0.text(x, y, value, ha='left', va='center', fontsize=4, \n            bbox=dict(facecolor='none', edgecolor='black', boxstyle='round', linewidth=0.3))\n    \nax1 = fig.add_subplot(gs[0, 1])\nfor s in [\"right\", \"top\"]:\n    ax1.spines[s].set_visible(False)\nax1.set_facecolor(background_color)\nax1_sns = sns.barplot(ax=ax1, y=unique_values_percent_train['Features'], x=unique_values_percent_train['Count'], \n                      zorder=2, linewidth=0, orient='h', saturation=1, alpha=1)\nax1_sns.set_xlabel(\"Percentage Unique Values\",fontsize=4, weight='bold')\nax1_sns.set_ylabel(\"Features\",fontsize=4, weight='bold')\nax1_sns.tick_params(labelsize=4, width=0.5, length=1.5)\nax1_sns.grid(which='major', axis='x', zorder=0, color='#EEEEEE', linewidth=0.4)\nax1_sns.grid(which='major', axis='y', zorder=0, color='#EEEEEE', linewidth=0.4)\nax1.text(0, -1.7, 'Percentage Unique Values - Train Dataset', fontsize=6, ha='left', va='top', weight='bold')\nax1.text(0, -1, 'Attribute_2 and attribute_3 can be considered as classification features', fontsize=4, ha='left', va='top')\n# data label\nfor p in ax1.patches:\n    value = f'{p.get_width():.2f}'\n    x = p.get_x() + p.get_width() + 0.03\n    y = p.get_y() + p.get_height() / 2 \n    ax1.text(x, y, value, ha='left', va='center', fontsize=4, \n            bbox=dict(facecolor='none', edgecolor='black', boxstyle='round', linewidth=0.3))\n\nbackground_color = \"#f6f5f5\"\nsns.set_palette(['#ff355d']*int(len(integer_features)))\n    \nax3 = fig.add_subplot(gs[1, 0])\nfor s in [\"right\", \"top\"]:\n    ax3.spines[s].set_visible(False)\nax3.set_facecolor(background_color)\nax3_sns = sns.barplot(ax=ax3, y=unique_values_test['Features'], x=unique_values_test['Count'], \n                      zorder=2, linewidth=0, orient='h', saturation=1, alpha=1)\nax3_sns.set_xlabel(\"Unique Values\",fontsize=4, weight='bold')\nax3_sns.set_ylabel(\"Features\",fontsize=4, weight='bold')\nax3_sns.tick_params(labelsize=4, width=0.5, length=1.5)\nax3_sns.grid(which='major', axis='x', zorder=0, color='#EEEEEE', linewidth=0.4)\nax3_sns.grid(which='major', axis='y', zorder=0, color='#EEEEEE', linewidth=0.4)\nax3.text(0, -1.7, 'Unique Values - Test Dataset', fontsize=6, ha='left', va='top', weight='bold')\nax3.text(0, -1, 'Test dataset is quite similar with train dataset', fontsize=4, ha='left', va='top')\nax3.get_xaxis().set_major_formatter(matplotlib.ticker.FuncFormatter(lambda x, p: format(int(x), ',')))\n# data label\nfor p in ax3.patches:\n    value = f'{p.get_width():,.0f}'\n    x = p.get_x() + p.get_width() + 500\n    y = p.get_y() + p.get_height() / 2 \n    ax3.text(x, y, value, ha='left', va='center', fontsize=4, \n            bbox=dict(facecolor='none', edgecolor='black', boxstyle='round', linewidth=0.3))\n    \nax4 = fig.add_subplot(gs[1, 1])\nfor s in [\"right\", \"top\"]:\n    ax4.spines[s].set_visible(False)\nax4.set_facecolor(background_color)\nax4_sns = sns.barplot(ax=ax4, y=unique_values_percent_test['Features'], x=unique_values_percent_test['Count'], \n                      zorder=2, linewidth=0, orient='h', saturation=1, alpha=1)\nax4_sns.set_xlabel(\"Percentage Unique Values\",fontsize=4, weight='bold')\nax4_sns.set_ylabel(\"Features\",fontsize=4, weight='bold')\nax4_sns.tick_params(labelsize=4, width=0.5, length=1.5)\nax4_sns.grid(which='major', axis='x', zorder=0, color='#EEEEEE', linewidth=0.4)\nax4_sns.grid(which='major', axis='y', zorder=0, color='#EEEEEE', linewidth=0.4)\nax4.text(0, -1.7, 'Percentage Unique Values - Test Dataset', fontsize=6, ha='left', va='top', weight='bold')\nax4.text(0, -1, 'Test dataset is quite similar with train dataset', fontsize=4, ha='left', va='top')\n# data label\nfor p in ax4.patches:\n    value = f'{p.get_width():.2f}'\n    x = p.get_x() + p.get_width() + 0.03\n    y = p.get_y() + p.get_height() / 2 \n    ax4.text(x, y, value, ha='left', va='center', fontsize=4, \n            bbox=dict(facecolor='none', edgecolor='black', boxstyle='round', linewidth=0.3))\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T07:17:20.646216Z","iopub.execute_input":"2022-08-03T07:17:20.646698Z","iopub.status.idle":"2022-08-03T07:17:24.714308Z","shell.execute_reply.started":"2022-08-03T07:17:20.646651Z","shell.execute_reply":"2022-08-03T07:17:24.713323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[back to top](#table-of-contents)\n<a id=\"AQnF\"></a>\n## **<span style=\"color:#f4a261;\">Analys Quantitive Features</span>**","metadata":{}},{"cell_type":"markdown","source":"[back to top](#table-of-contents)\n<a id=\"5\"></a>\n### **<span style=\"color:#f4a261;\">5. Data Distributions</span>**","metadata":{}},{"cell_type":"code","source":"plt.rcParams['figure.dpi'] = 600\nfig = plt.figure(figsize=(10, 10), facecolor='#f6f5f5')\ngs = fig.add_gridspec(7, 3)\ngs.update(wspace=0.3, hspace=0.3)\nbackground_color = '#f6f5f5'\nrun_no = 0\n\ncolormap = ['#ffd514','#ff8826', '#ff355d', '#C70039']\n\nfor row in range(0, 7):\n    for col in range(0, 3):\n        locals()[\"ax\"+str(run_no)] = fig.add_subplot(gs[row, col])\n        locals()[\"ax\"+str(run_no)].set_facecolor(background_color)\n        for s in [\"top\",\"right\"]:\n            locals()[\"ax\"+str(run_no)].spines[s].set_visible(False)\n        run_no += 1  \n\n\nfeatures = [col for col in test_df._get_numeric_data().columns if col not in ['id']]\n\nrun_no = 0\nfor col in features:\n    sns.kdeplot(ax=locals()[\"ax\"+str(run_no)], x=train_df[col], zorder=2, alpha=1, linewidth=1.4, color=colormap[0], label='Train')\n    sns.kdeplot(ax=locals()[\"ax\"+str(run_no)], x=test_df[col], zorder=2, alpha=1, linewidth=1.4, color=colormap[2], label='Test')    \n    \n    locals()[\"ax\"+str(run_no)].grid(which='major', axis='x', zorder=0, color='#EEEEEE', linewidth=0.4)\n    locals()[\"ax\"+str(run_no)].grid(which='major', axis='y', zorder=0, color='#EEEEEE', linewidth=0.4)\n    locals()[\"ax\"+str(run_no)].set_ylabel('')\n    locals()[\"ax\"+str(run_no)].set_xlabel(col, fontsize=4, fontweight='bold')\n    locals()[\"ax\"+str(run_no)].tick_params(labelsize=4, width=0.5)\n    locals()[\"ax\"+str(run_no)].xaxis.offsetText.set_fontsize(4)\n    locals()[\"ax\"+str(run_no)].yaxis.offsetText.set_fontsize(4)\n    locals()[\"ax\"+str(run_no)].legend(fontsize=4, ncol=3, loc='upper right', facecolor=background_color, edgecolor=background_color)\n    #locals()[\"ax\"+str(run_no)].get_legend().remove()\n    \n    run_no += 1\n\n\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T07:17:24.715505Z","iopub.execute_input":"2022-08-03T07:17:24.716493Z","iopub.status.idle":"2022-08-03T07:17:36.634554Z","shell.execute_reply.started":"2022-08-03T07:17:24.716452Z","shell.execute_reply":"2022-08-03T07:17:36.633144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **<span style=\"color:#f4a261;\">attribute 2 & 3</span>**","metadata":{}},{"cell_type":"code","source":"plt.rcParams['figure.dpi'] = 600\nfig = plt.figure(figsize=(12, 3), facecolor='#f6f5f5')\ngs = fig.add_gridspec(1, 2)\ngs.update(wspace=0.3, hspace=0.3)\nbackground_color = '#f6f5f5'\nrun_no = 0\n\ncolormap = ['#ffd514','#ff8826', '#ff355d', '#C70039']\n\nfor row in range(0, 1):\n    for col in range(0, 2):\n        locals()[\"ax\"+str(run_no)] = fig.add_subplot(gs[row, col])\n        locals()[\"ax\"+str(run_no)].set_facecolor(background_color)\n        for s in [\"top\",\"right\"]:\n            locals()[\"ax\"+str(run_no)].spines[s].set_visible(False)\n        run_no += 1  \nfeatures = ['attribute_2', 'attribute_3']\n\nrun_no = 0\n\nfor col in features:\n    sns.kdeplot(ax=locals()[\"ax\"+str(run_no)], x=train_df[train_df['failure'] == 0][col], zorder=2, alpha=1,linewidth=1, color=colormap[0], label='Train failure - 0')\n    sns.kdeplot(ax=locals()[\"ax\"+str(run_no)], x=train_df[train_df['failure'] == 1][col], zorder=2, alpha=1,linewidth=1, color=colormap[1], label='Train failure - 1')\n    sns.kdeplot(ax=locals()[\"ax\"+str(run_no)], x=test_df[col], zorder=2, alpha=1, linewidth=1, color=colormap[2], label='Test')  \n    locals()[\"ax\"+str(run_no)].grid(which='major', axis='x', zorder=0, color='#EEEEEE', linewidth=0.4)\n    locals()[\"ax\"+str(run_no)].grid(which='major', axis='y', zorder=0, color='#EEEEEE', linewidth=0.4)\n    locals()[\"ax\"+str(run_no)].set_ylabel('')\n    locals()[\"ax\"+str(run_no)].set_xlabel(col, fontsize=4, fontweight='bold')\n    locals()[\"ax\"+str(run_no)].tick_params(labelsize=4, width=0.5)\n    locals()[\"ax\"+str(run_no)].xaxis.offsetText.set_fontsize(4)\n    locals()[\"ax\"+str(run_no)].yaxis.offsetText.set_fontsize(4)\n    locals()[\"ax\"+str(run_no)].legend(fontsize=4, ncol=3, loc='upper center', facecolor=background_color, edgecolor=background_color)\n    #locals()[\"ax\"+str(run_no)].get_legend().remove()\n    \n    run_no += 1\n\n\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T07:17:36.636048Z","iopub.execute_input":"2022-08-03T07:17:36.636488Z","iopub.status.idle":"2022-08-03T07:17:38.577723Z","shell.execute_reply.started":"2022-08-03T07:17:36.636451Z","shell.execute_reply":"2022-08-03T07:17:38.576369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[back to top](#table-of-contents)\n<a id=\"6\"></a>\n### **<span style=\"color:#f4a261;\">6. Feature correlation</span>**","metadata":{}},{"cell_type":"code","source":"features = [col for col in test_df._get_numeric_data().columns if col not in ['id']]\n\ntrain_corr = train_df[features].corr()\ntest_corr = test_df[features].corr()\ndiff_corr = abs(train_corr - test_corr)\n\nplt.rcParams['figure.dpi'] = 600\nfig = plt.figure(figsize=(10, 5), facecolor='#f6f5f5')\ngs = fig.add_gridspec(1, 3)\ngs.update(wspace=0.5, hspace=0)\n\nbackground_color = \"#f6f5f5\"\ncmap_train = sns.dark_palette('#ffd514', as_cmap=True)\ncmap_test = sns.dark_palette('#ff355d', as_cmap=True)\ncmap_diff = sns.dark_palette('#287094', as_cmap=True)\n\nmask = np.triu(np.ones_like(train_corr, dtype=bool))\n\nrun_no = 0\nfor row in range(0, 1):\n    for col in range(0, 3):\n        locals()[\"ax\"+str(run_no)] = fig.add_subplot(gs[row, col])\n        locals()[\"ax\"+str(run_no)].set_facecolor(background_color)\n        for s in [\"top\",\"right\"]:\n            locals()[\"ax\"+str(run_no)].spines[s].set_visible(False)\n        run_no += 1\n        \nsns.heatmap(train_corr, ax=ax0, cmap=cmap_train, square=True, mask=mask, linewidths=.5, linecolor='#f6f5f5', \n            cbar_kws={\"shrink\": .3})\nax0.set_xlabel(col, fontsize=5, fontweight='bold')\nax0.tick_params(labelsize=5, width=0.5, length=1.5)\nax0.set_xlabel('Train Dataset', fontsize=5, fontweight='bold')\ncax = plt.gcf().axes[-1]\ncax.tick_params(labelsize=5)\n\nsns.heatmap(test_corr, ax=ax1, cmap=cmap_test, square=True, mask=mask, linewidths=.5, linecolor='#f6f5f5', \n            cbar_kws={\"shrink\": .3})\nax1.set_xlabel(col, fontsize=5, fontweight='bold')\nax1.tick_params(labelsize=5, width=0.5, length=1.5)\nax1.set_xlabel('Test Dataset', fontsize=5, fontweight='bold')\ncax = plt.gcf().axes[-1]\ncax.tick_params(labelsize=5)\n\nsns.heatmap(diff_corr, ax=ax2, cmap=cmap_diff, square=True, mask=mask, linewidths=.5, linecolor='#f6f5f5', \n            cbar_kws={\"shrink\": .3})\nax2.set_xlabel(col, fontsize=5, fontweight='bold')\nax2.tick_params(labelsize=5, width=0.5, length=1.5)\nax2.set_xlabel('Absolute Correlation Differences', fontsize=5, fontweight='bold')\ncax = plt.gcf().axes[-1]\ncax.tick_params(labelsize=5)\n\nax0.text(0, -2.2, 'Features Correlation', fontsize=10, fontweight='bold')\nax0.text(0, -0.8, 'Correlation between features in train and test dataset with their absolute correlation differences', fontsize=7)\n\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T07:17:38.579544Z","iopub.execute_input":"2022-08-03T07:17:38.580044Z","iopub.status.idle":"2022-08-03T07:17:41.630049Z","shell.execute_reply.started":"2022-08-03T07:17:38.579985Z","shell.execute_reply":"2022-08-03T07:17:41.629100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[back to top](#table-of-contents)\n<a id=\"AQlF\"></a>\n## **<span style=\"color:#f4a261;\">Analys Qualitive Features</span>**","metadata":{}},{"cell_type":"code","source":"plt.rcParams['figure.dpi'] = 600\nfig = plt.figure(figsize=(12, 3), facecolor='#f6f5f5')\ngs = fig.add_gridspec(1, 2)\ngs.update(wspace=0.3, hspace=0.3)\nbackground_color = '#f6f5f5'\nrun_no = 0\n\ncolormap = ['#ffd514','#ff8826', '#ff355d', '#C70039']\n\nfor row in range(0, 1):\n    for col in range(0, 2):\n        locals()[\"ax\"+str(run_no)] = fig.add_subplot(gs[row, col])\n        locals()[\"ax\"+str(run_no)].set_facecolor(background_color)\n        for s in [\"top\",\"right\"]:\n            locals()[\"ax\"+str(run_no)].spines[s].set_visible(False)\n        run_no += 1  \nfeatures = ['attribute_2', 'attribute_3']\n\nrun_no = 0\n\nsns.scatterplot(ax=locals()[\"ax\"+str(0)], y=train_df['attribute_3'], x=train_df['attribute_2'], color=colormap[0], zorder=2, label='train', linewidth=0)\nsns.scatterplot(ax=locals()[\"ax\"+str(0)], y=test_df['attribute_3'], x=test_df['attribute_2'], color=colormap[2], zorder=2, label='test', linewidth=0)\nlocals()[\"ax\"+str(0)].set_ylabel('attribute_3', fontsize=6, fontweight='bold')\nlocals()[\"ax\"+str(0)].set_xlabel('attribute_2', fontsize=6, fontweight='bold')\n\nsns.scatterplot(ax=locals()[\"ax\"+str(1)], y=train_df['attribute_3'], x=train_df['product_code'], color=colormap[0], zorder=2, label='train', linewidth=0)\nsns.scatterplot(ax=locals()[\"ax\"+str(1)], y=test_df['attribute_3'], x=test_df['product_code'], color=colormap[2], zorder=2, label='test', linewidth=0)\nlocals()[\"ax\"+str(1)].set_ylabel('attribute_3', fontsize=6, fontweight='bold')\nlocals()[\"ax\"+str(1)].set_xlabel('product_code', fontsize=6, fontweight='bold')\n\nfor col in features: \n    locals()[\"ax\"+str(run_no)].grid(which='major', axis='x', zorder=0, color='#EEEEEE', linewidth=0.4)\n    locals()[\"ax\"+str(run_no)].grid(which='major', axis='y', zorder=0, color='#EEEEEE', linewidth=0.4)\n    locals()[\"ax\"+str(run_no)].tick_params(labelsize=6, width=0.5)\n    locals()[\"ax\"+str(run_no)].xaxis.offsetText.set_fontsize(6)\n    locals()[\"ax\"+str(run_no)].yaxis.offsetText.set_fontsize(6)\n    locals()[\"ax\"+str(run_no)].legend(fontsize=6, ncol=3, loc='upper center', facecolor=background_color, edgecolor=background_color)\n    #locals()[\"ax\"+str(run_no)].get_legend().remove()\n    \n    run_no += 1\n\n\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T07:17:41.631649Z","iopub.execute_input":"2022-08-03T07:17:41.632217Z","iopub.status.idle":"2022-08-03T07:17:43.917258Z","shell.execute_reply.started":"2022-08-03T07:17:41.632182Z","shell.execute_reply":"2022-08-03T07:17:43.915695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[back to top](#table-of-contents)\n<a id=\"PrPc\"></a>\n## **<span style=\"color:#f4a261;\">Pre-Proccesing</span>**","metadata":{}},{"cell_type":"markdown","source":"[back to top](#table-of-contents)\n<a id=\"7\"></a>\n### **<span style=\"color:#f4a261;\">7. Filling N/A</span>**","metadata":{}},{"cell_type":"code","source":"measurement_cols = [f'measurement_{i}' for i in range(18)]\n\ndef _scale(train_data, val_data):\n    #scaler = StandardScaler()\n    scaler = PowerTransformer()\n    \n    scaled_train = scaler.fit_transform(train_data[measurement_cols + [\"loading\"]])\n    scaled_val = scaler.transform(val_data[measurement_cols + [\"loading\"]])\n    \n    #back to dataframe\n    new_train = train_data.copy()\n    new_val = val_data.copy()\n    \n    new_train[measurement_cols + [\"loading\"]] = scaled_train\n    new_val[measurement_cols + [\"loading\"]] = scaled_val\n    \n    assert len(train_data) == len(new_train)\n    assert len(val_data) == len(val_data)\n    \n    return new_train, new_val\n\n#train_df, test_df = _scale(train_df, test_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:11:20.002889Z","iopub.execute_input":"2022-08-03T16:11:20.003372Z","iopub.status.idle":"2022-08-03T16:11:20.012230Z","shell.execute_reply.started":"2022-08-03T16:11:20.003326Z","shell.execute_reply":"2022-08-03T16:11:20.011059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['is_train'] = 1\ntest_df['is_train'] = 0\n\ndf = pd.concat([train_df, test_df])\ndf=df.reset_index(drop=True)\n\n\nfeatures_to_fill = [f'measurement_{i}' for i in range(18)] #+ ['loading']\nfor col in features_to_fill:\n    df[col] = df[col].fillna(df[col].mean())\ndf['loading'].fillna(df['loading'].median(), inplace=True)\n\n\"\"\"_FEATURES = [col for col in test_df._get_numeric_data().columns if col not in ['id', 'loading', 'failure']]\n_TARGET = 'loading'\n\nmissings_indexes = df[df[_TARGET].isna()].index\nfilled_indexes = [idx for idx in df.index if idx not in missings_indexes]\n\n#print(f'{features_to_fill}, len filled: {len(filled_indexes)}, len missing: {len(missings_indexes)}')\nmodel = CatBoostRegressor(iterations=250, loss_function='RMSE', verbose=50)\nmodel.fit(X=df.iloc[filled_indexes][_FEATURES], y=df.iloc[filled_indexes][_TARGET])\ndf.loc[missings_indexes, _TARGET] = model.predict(df.iloc[missings_indexes][_FEATURES])\"\"\"\n\ntrain_df = df[df['is_train'] == 1]\ntest_df = df[df['is_train'] == 0]\ntrain_df = train_df.drop(columns='is_train')\ntest_df = test_df.drop(columns='is_train')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:43:08.135639Z","iopub.execute_input":"2022-08-06T12:43:08.136779Z","iopub.status.idle":"2022-08-06T12:43:08.237481Z","shell.execute_reply.started":"2022-08-06T12:43:08.136725Z","shell.execute_reply":"2022-08-06T12:43:08.236258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[back to top](#table-of-contents)\n<a id=\"8\"></a>\n### **<span style=\"color:#f4a261;\">8. Scaling & labeling</span>**","metadata":{}},{"cell_type":"code","source":"train_df['attribute_1'] = train_df['attribute_1'].str.split('_', 1).str[1].astype('int')\ntrain_df['attribute_0'] = train_df['attribute_0'].str.split('_', 1).str[1].astype('int')\n\ntest_df['attribute_1'] = test_df['attribute_1'].str.split('_', 1).str[1].astype('int')\ntest_df['attribute_0'] = test_df['attribute_0'].str.split('_', 1).str[1].astype('int')\n\n\nprod_code_encoding = {'A': 1, 'B':2, 'C':3, 'D': 4, 'E': 5, 'F':6, 'G': 7, 'H':8, 'I': 9}\n\ntrain_df.product_code = [prod_code_encoding[val] for val in train_df.product_code]\ntest_df.product_code = [prod_code_encoding[val] for val in test_df.product_code]","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:43:10.691696Z","iopub.execute_input":"2022-08-06T12:43:10.692163Z","iopub.status.idle":"2022-08-06T12:43:11.105172Z","shell.execute_reply.started":"2022-08-06T12:43:10.692127Z","shell.execute_reply":"2022-08-06T12:43:11.103838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[back to top](#table-of-contents)\n<a id=\"9\"></a>\n### **<span style=\"color:#f4a261;\">9. Upsampling</span>**","metadata":{}},{"cell_type":"code","source":"features = [col for col in test_df._get_numeric_data().columns if col not in ['id', 'failure']]\n\noversample = SMOTE()\nXtrain, ytrain = oversample.fit_resample(train_df[features], train_df.failure)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:43:12.606294Z","iopub.execute_input":"2022-08-06T12:43:12.606815Z","iopub.status.idle":"2022-08-06T12:43:13.720185Z","shell.execute_reply.started":"2022-08-06T12:43:12.606776Z","shell.execute_reply":"2022-08-06T12:43:13.718736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[back to top](#table-of-contents)\n<a id=\"10\"></a>\n### **<span style=\"color:#f4a261;\">10. Modeling</span>**\n#### **<span style=\"color:#f4a261;\">CatBoost</span>**","metadata":{}},{"cell_type":"code","source":"cat_features = ['attribute_0', 'attribute_1', 'attribute_2', 'attribute_3']\n\ndef objective(trial, data=Xtrain[features], target=ytrain):\n    train_x, test_x, train_y, test_y = train_test_split(data, target, test_size=0.15, random_state=42)\n    \n    param = {\n        \"objective\": trial.suggest_categorical(\"objective\", [\"Logloss\", \"CrossEntropy\"]),\n        \"colsample_bylevel\": trial.suggest_float(\"colsample_bylevel\", 0.01, 0.1),\n        \"depth\": trial.suggest_int(\"depth\", 1, 12),\n        \"boosting_type\": trial.suggest_categorical(\"boosting_type\", [\"Ordered\", \"Plain\"]),\n        \"bootstrap_type\": trial.suggest_categorical(\n            \"bootstrap_type\", [\"Bayesian\", \"Bernoulli\", \"MVS\"]\n        ),\n        \"used_ram_limit\": \"3gb\",\n        \n    }\n    model = CatBoostClassifier(**param)  \n    \n    model.fit(train_x,train_y, cat_features=cat_features, early_stopping_rounds=200, verbose=False)\n    \n    preds = model.predict(test_x)\n    \n    metric = roc_auc_score(test_y, preds)\n    \n    return metric\n\nstudy = optuna.create_study(direction=\"maximize\")\nstudy.optimize(objective, n_trials=50, timeout=600)\n\nprint(\"Number of finished trials: {}\".format(len(study.trials)))\n\nprint(\"Best trial:\")\ntrial = study.best_trial\n\nprint(\"  Value: {}\".format(trial.value))\n\nprint(\"  Params: \")\nfor key, value in trial.params.items():\n    print(\"    {}: {}\".format(key, value))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:06:54.257159Z","iopub.execute_input":"2022-08-03T11:06:54.257618Z","iopub.status.idle":"2022-08-03T11:18:21.423014Z","shell.execute_reply.started":"2022-08-03T11:06:54.257581Z","shell.execute_reply":"2022-08-03T11:18:21.421538Z"},"_kg_hide-output":true,"_kg_hide-input":true,"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.visualization.plot_param_importances(study)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T11:18:21.426648Z","iopub.execute_input":"2022-08-03T11:18:21.427469Z","iopub.status.idle":"2022-08-03T11:18:21.922768Z","shell.execute_reply.started":"2022-08-03T11:18:21.427417Z","shell.execute_reply":"2022-08-03T11:18:21.921551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.visualization.plot_edf(study)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:46:08.378990Z","iopub.execute_input":"2022-08-03T11:46:08.379480Z","iopub.status.idle":"2022-08-03T11:46:08.405608Z","shell.execute_reply.started":"2022-08-03T11:46:08.379433Z","shell.execute_reply":"2022-08-03T11:46:08.404509Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.DataFrame()\npred_df_train = pd.DataFrame()\n\ncat_params = {'objective': 'Logloss', 'colsample_bylevel': 0.0994741421913364, 'depth': 12, 'boosting_type': 'Plain', 'bootstrap_type': 'Bayesian'}\ncat_features = ['attribute_0', 'attribute_1', 'attribute_2', 'attribute_3']\n\n\ngkf = GroupKFold(n_splits=5)\nfor fold, (idx_tr, idx_va) in enumerate(gkf.split(Xtrain, ytrain, Xtrain.product_code)):\n    \n    X_train = Xtrain.iloc[idx_tr][features]\n    X_valid = Xtrain.iloc[idx_va][features]\n    X_test = test_df[features].copy()\n    y_train = ytrain.iloc[idx_tr]\n    y_valid = ytrain.iloc[idx_va]\n    \n    cat_model = CatBoostClassifier(**cat_params).fit(X_train[features], y_train, cat_features=cat_features, verbose=0)\n    \n    val_pred = cat_model.predict_proba(X_valid[features])[:,1]\n    print(f'CatBoost fold: {fold}: ROC AUC: {roc_auc_score(y_valid, val_pred):.4f}')\n    pred_df_train[f'catb_{fold}'] =  cat_model.predict_proba(Xtrain)[:,1]\n    pred_df[f'catb_{fold}'] = cat_model.predict_proba(X_test)[:,1]\n    \n\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T09:19:45.516028Z","iopub.execute_input":"2022-08-06T09:19:45.516503Z","iopub.status.idle":"2022-08-06T09:21:32.125238Z","shell.execute_reply.started":"2022-08-06T09:19:45.516463Z","shell.execute_reply":"2022-08-06T09:21:32.123390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### **<span style=\"color:#f4a261;\">LGBM</span>**","metadata":{}},{"cell_type":"code","source":"\ndef objective(trial, data=Xtrain[features], target=ytrain):\n    train_x, test_x, train_y, test_y = train_test_split(data, target, test_size=0.15, random_state=42)\n    \n    param = {\n        #         \"device_type\": trial.suggest_categorical(\"device_type\", ['gpu']),\n        \"n_estimators\": trial.suggest_categorical(\"n_estimators\", [10000]),\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.01, 0.3),\n        \"num_leaves\": trial.suggest_int(\"num_leaves\", 20, 3000, step=20),\n        \"max_depth\": trial.suggest_int(\"max_depth\", 3, 12),\n        \"min_data_in_leaf\": trial.suggest_int(\"min_data_in_leaf\", 200, 10000, step=100),\n        \"max_bin\": trial.suggest_int(\"max_bin\", 200, 300),\n        \"lambda_l1\": trial.suggest_int(\"lambda_l1\", 0, 100, step=5),\n        \"lambda_l2\": trial.suggest_int(\"lambda_l2\", 0, 100, step=5),\n        \"min_gain_to_split\": trial.suggest_float(\"min_gain_to_split\", 0, 15),\n        \"bagging_fraction\": trial.suggest_float(\n            \"bagging_fraction\", 0.2, 0.95, step=0.1\n        ),\n        \"bagging_freq\": trial.suggest_categorical(\"bagging_freq\", [1]),\n        \"feature_fraction\": trial.suggest_float(\n            \"feature_fraction\", 0.2, 0.95, step=0.1\n        ),\n    }\n        \n    model = lgbm.LGBMClassifier(**param)  \n    \n    model.fit(train_x, train_y, verbose=False)\n    \n    preds = model.predict(test_x)\n    \n    metric = roc_auc_score(test_y, preds)\n    \n    return metric\n\nstudy = optuna.create_study(direction=\"maximize\")\nstudy.optimize(objective, n_trials=50, timeout=600)\n\nprint(\"Number of finished trials: {}\".format(len(study.trials)))\n\nprint(\"Best trial:\")\ntrial = study.best_trial\n\nprint(\"  Value: {}\".format(trial.value))\n\nprint(\"  Params: \")\nfor key, value in trial.params.items():\n    print(\"    {}: {}\".format(key, value))\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:54:55.380642Z","iopub.execute_input":"2022-08-03T11:54:55.381041Z","iopub.status.idle":"2022-08-03T12:05:02.799341Z","shell.execute_reply.started":"2022-08-03T11:54:55.380998Z","shell.execute_reply":"2022-08-03T12:05:02.798350Z"},"_kg_hide-output":true,"_kg_hide-input":true,"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.visualization.plot_param_importances(study)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:05:02.801046Z","iopub.execute_input":"2022-08-03T12:05:02.801615Z","iopub.status.idle":"2022-08-03T12:05:03.821755Z","shell.execute_reply.started":"2022-08-03T12:05:02.801581Z","shell.execute_reply":"2022-08-03T12:05:03.820148Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.visualization.plot_edf(study)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:05:03.825619Z","iopub.execute_input":"2022-08-03T12:05:03.825983Z","iopub.status.idle":"2022-08-03T12:05:03.842023Z","shell.execute_reply.started":"2022-08-03T12:05:03.825950Z","shell.execute_reply":"2022-08-03T12:05:03.840690Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_params = {'n_estimators': 10000, 'learning_rate': 0.19062056780292896, 'num_leaves': 2840, 'max_depth': 12, 'min_data_in_leaf': 700, 'max_bin': 274, 'lambda_l1': 10, 'lambda_l2': 10, 'min_gain_to_split': 8.251862584051507, 'bagging_fraction': 0.7, 'bagging_freq': 1, 'feature_fraction': 0.7}\n\ngkf = GroupKFold(n_splits=5)\nfor fold, (idx_tr, idx_va) in enumerate(gkf.split(Xtrain, ytrain, Xtrain.product_code)):\n    \n    X_train = Xtrain.iloc[idx_tr][features]\n    X_valid = Xtrain.iloc[idx_va][features]\n    X_test = test_df[features].copy()\n    y_train = ytrain.iloc[idx_tr]\n    y_valid = ytrain.iloc[idx_va]\n    \n    lgbm_model = lgbm.LGBMClassifier(**lgbm_params).fit(X_train[features], y_train, verbose=0)\n    \n    val_pred = lgbm_model.predict_proba(X_valid[features])[:,1]\n    print(f'Lightgbm fold: {fold}: ROC AUC: {roc_auc_score(y_valid, val_pred):.4f}')\n    pred_df[f'lgbm_{fold}'] = lgbm_model.predict_proba(X_test)[:,1]\n    pred_df_train[f'lgbm_{fold}'] =  lgbm_model.predict_proba(Xtrain)[:,1]\n    \n\n    gc.collect()","metadata":{"_kg_hide-input":false,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2022-08-06T08:56:07.330343Z","iopub.execute_input":"2022-08-06T08:56:07.330752Z","iopub.status.idle":"2022-08-06T08:57:31.784901Z","shell.execute_reply.started":"2022-08-06T08:56:07.330715Z","shell.execute_reply":"2022-08-06T08:57:31.783443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### **<span style=\"color:#f4a261;\">XGBoost</span>**","metadata":{}},{"cell_type":"code","source":"\ndef objective(trial, data=Xtrain[features], target=ytrain):\n    train_x, test_x, train_y, test_y = train_test_split(data, target, test_size=0.15, random_state=42)\n    \n    param = {\n       'max_depth': trial.suggest_int('max_depth', 6, 15),\n        'subsample': trial.suggest_float('subsample', 0.1, 1.0),\n        'n_estimators': trial.suggest_int('n_estimators', 2000, 12000),\n        'lambda': trial.suggest_float('lambda', 1e-3, 1e3),\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.2),\n        'min_child_weight': trial.suggest_int('min_child_weight', 300, 2000),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.10, 0.5),\n    }\n        \n    model = xgb.XGBClassifier(**param)  \n    \n    model.fit(train_x, train_y, verbose=False)\n    \n    preds = model.predict(test_x)\n    \n    metric = roc_auc_score(test_y, preds)\n    \n    return metric\n\nstudy = optuna.create_study(direction=\"maximize\")\nstudy.optimize(objective, n_trials=50, timeout=600)\n\nprint(\"Number of finished trials: {}\".format(len(study.trials)))\n\nprint(\"Best trial:\")\ntrial = study.best_trial\n\nprint(\"  Value: {}\".format(trial.value))\n\nprint(\"  Params: \")\nfor key, value in trial.params.items():\n    print(\"    {}: {}\".format(key, value))\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:35:06.541850Z","iopub.execute_input":"2022-08-03T13:35:06.542857Z","iopub.status.idle":"2022-08-03T13:45:39.430021Z","shell.execute_reply.started":"2022-08-03T13:35:06.542815Z","shell.execute_reply":"2022-08-03T13:45:39.428074Z"},"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.visualization.plot_param_importances(study)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:45:39.432466Z","iopub.execute_input":"2022-08-03T13:45:39.432920Z","iopub.status.idle":"2022-08-03T13:45:39.674108Z","shell.execute_reply.started":"2022-08-03T13:45:39.432873Z","shell.execute_reply":"2022-08-03T13:45:39.672837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.visualization.plot_edf(study)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:45:39.677997Z","iopub.execute_input":"2022-08-03T13:45:39.678419Z","iopub.status.idle":"2022-08-03T13:45:39.694696Z","shell.execute_reply.started":"2022-08-03T13:45:39.678382Z","shell.execute_reply":"2022-08-03T13:45:39.693533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_param = {'max_depth': 14, 'subsample': 0.8578106042369132, 'n_estimators': 7337, 'lambda': 374.4601814674628, 'learning_rate': 0.11371623773461367, 'min_child_weight': 371, 'colsample_bytree': 0.4301056626870441}\n\ngkf = GroupKFold(n_splits=5)\nfor fold, (idx_tr, idx_va) in enumerate(gkf.split(Xtrain, ytrain, Xtrain.product_code)):\n    \n    X_train = Xtrain.iloc[idx_tr][features]\n    X_valid = Xtrain.iloc[idx_va][features]\n    X_test = test_df[features].copy()\n    y_train = ytrain.iloc[idx_tr]\n    y_valid = ytrain.iloc[idx_va]\n    \n    xgb_model = xgb.XGBClassifier(**xgb_param).fit(X_train[features], y_train, verbose=0)\n    \n    val_pred = xgb_model.predict_proba(X_valid[features])[:,1]\n    print(f'XGBoost fold: {fold}: ROC AUC: {roc_auc_score(y_valid, val_pred):.4f}')\n    pred_df[f'xgb_{fold}'] = xgb_model.predict_proba(X_test)[:,1]\n    pred_df_train[f'xgb_{fold}'] = xgb_model.predict_proba(Xtrain)[:,1]\n\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:57:38.675916Z","iopub.execute_input":"2022-08-06T08:57:38.676404Z","iopub.status.idle":"2022-08-06T09:04:10.857809Z","shell.execute_reply.started":"2022-08-06T08:57:38.676363Z","shell.execute_reply":"2022-08-06T09:04:10.855344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### **<span style=\"color:#f4a261;\">Logit regression</span>**","metadata":{}},{"cell_type":"code","source":"\"\"\"\nfor i in range(4):\n    pred_df_train[f'attribute_{i}'] = Xtrain[f'attribute_{i}']\n    pred_df[f'attribute_{i}'] = test_df[f'attribute_{i}'].values\n\npred_df_train['failure'] = ytrain\npred_df_train['product_code'] = Xtrain['product_code']\npred_df['product_code'] = test_df['product_code'].values\n\"\"\"","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"outputs_hidden":true,"source_hidden":true},"execution":{"iopub.status.busy":"2022-08-06T12:41:06.166608Z","iopub.execute_input":"2022-08-06T12:41:06.167251Z","iopub.status.idle":"2022-08-06T12:41:06.205436Z","shell.execute_reply.started":"2022-08-06T12:41:06.167126Z","shell.execute_reply":"2022-08-06T12:41:06.204397Z"},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:21:39.516182Z","iopub.execute_input":"2022-08-03T16:21:39.517003Z","iopub.status.idle":"2022-08-03T16:21:39.540644Z","shell.execute_reply.started":"2022-08-03T16:21:39.516965Z","shell.execute_reply":"2022-08-03T16:21:39.539493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"Xtrain, ytrain = pred_df_train[[col for col in pred_df_train.columns if col not in ['failure']]], pred_df_train.failure\nXtest = pred_df[[col for col in pred_df_train.columns if col not in ['failure']]]\n\nlr_predict = []\nfeatures = [col for col in pred_df_train.columns if col not in ['failure', 'product_code']]\n\ngkf = GroupKFold(n_splits=5)\nfor fold, (idx_tr, idx_va) in enumerate(gkf.split(Xtrain, ytrain, Xtrain.product_code)):\n    \n    X_train = Xtrain.iloc[idx_tr][features]\n    X_valid = Xtrain.iloc[idx_va][features]\n    X_test = pred_df[features].copy()\n    y_train = ytrain.iloc[idx_tr]\n    y_valid = ytrain.iloc[idx_va]\n    \n    lr = LogisticRegression(penalty='elasticnet', l1_ratio=0.8, C=0.007, tol = 1e-2, solver='saga', max_iter=1000)\n    lr.fit(X_train, y_train)\n    val_pred = lr.predict(X_valid)\n    print(f'LR second stage fold: {fold}: ROC AUC: {roc_auc_score(y_valid, val_pred):.4f}')\n    lr_predict.append(lr.predict_proba(X_test)[:,1])\n    \nlr_preds = pd.DataFrame(np.array(lr_predict).T, columns=[f'lr_{i}' for i in range(5)])\n\"\"\"\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:21:46.852696Z","iopub.execute_input":"2022-08-03T16:21:46.853091Z","iopub.status.idle":"2022-08-03T16:21:47.758032Z","shell.execute_reply.started":"2022-08-03T16:21:46.853060Z","shell.execute_reply":"2022-08-03T16:21:47.756543Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#pred_df = pd.DataFrame()\ngkf = GroupKFold(n_splits=5)\nfor fold, (idx_tr, idx_va) in enumerate(gkf.split(Xtrain, ytrain, Xtrain.product_code)):\n    \n    X_train = Xtrain.iloc[idx_tr][features]\n    X_valid = Xtrain.iloc[idx_va][features]\n    X_test = test_df[features].copy()\n    y_train = ytrain.iloc[idx_tr]\n    y_valid = ytrain.iloc[idx_va]\n    \n    lr = LogisticRegression(penalty='elasticnet', l1_ratio=0.8, C=0.007, tol = 1e-2, solver='saga', max_iter=1000)\n    lr.fit(X_train, y_train)\n    val_pred = lr.predict_proba(X_valid[features])[:,1]\n    print(f'logit fold: {fold}: ROC AUC: {roc_auc_score(y_valid, val_pred):.4f}')\n    pred_df[f'logit_{fold}'] = lr.predict_proba(X_test)[:,1]\n    pred_df_train[f'lgbm_{fold}'] = lr.predict_proba(Xtrain)[:,1]\n\n    gc.collect()\n    \npred_df","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:45:08.974228Z","iopub.execute_input":"2022-08-06T12:45:08.975582Z","iopub.status.idle":"2022-08-06T12:45:21.641160Z","shell.execute_reply.started":"2022-08-06T12:45:08.975513Z","shell.execute_reply":"2022-08-06T12:45:21.639706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### **<span style=\"color:#f4a261;\">Submit</span>**","metadata":{}},{"cell_type":"code","source":"ssub['failure'] = pred_df.T.mean()\nssub.to_csv('submission.csv', index=False)\nssub","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:47:31.998685Z","iopub.execute_input":"2022-08-06T12:47:31.999694Z","iopub.status.idle":"2022-08-06T12:47:32.118992Z","shell.execute_reply.started":"2022-08-06T12:47:31.999639Z","shell.execute_reply":"2022-08-06T12:47:32.117950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ssub","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:47:35.747729Z","iopub.execute_input":"2022-08-06T12:47:35.748237Z","iopub.status.idle":"2022-08-06T12:47:35.766074Z","shell.execute_reply.started":"2022-08-06T12:47:35.748193Z","shell.execute_reply":"2022-08-06T12:47:35.764606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}