{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#### In this notebook we will do:\n\n- Load the data\n- Join tables\n- Make EDA: visualization of different features distribution\n- Train XGBoost base model\n- Make evalution\n- Create a submission table","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport polars as pl\nimport numpy  as np\n\nimport warnings as wr\nwr.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-05-22T15:37:56.014565Z","iopub.execute_input":"2024-05-22T15:37:56.015119Z","iopub.status.idle":"2024-05-22T15:37:56.808254Z","shell.execute_reply.started":"2024-05-22T15:37:56.015083Z","shell.execute_reply":"2024-05-22T15:37:56.807030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-05-22T15:37:56.810651Z","iopub.execute_input":"2024-05-22T15:37:56.811182Z","iopub.status.idle":"2024-05-22T15:37:57.585521Z","shell.execute_reply.started":"2024-05-22T15:37:56.811150Z","shell.execute_reply":"2024-05-22T15:37:57.584411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing   import LabelEncoder\nfrom sklearn.preprocessing   import OrdinalEncoder\nfrom sklearn.model_selection import train_test_split\n\nimport statsmodels.api   as sm\nimport statsmodels.tools as tools\nfrom   sklearn.linear_model import LogisticRegression\nfrom   xgboost import XGBClassifier\n\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix, roc_curve, auc","metadata":{"execution":{"iopub.status.busy":"2024-05-22T18:06:32.126092Z","iopub.execute_input":"2024-05-22T18:06:32.127220Z","iopub.status.idle":"2024-05-22T18:06:32.336897Z","shell.execute_reply.started":"2024-05-22T18:06:32.127180Z","shell.execute_reply":"2024-05-22T18:06:32.335727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Data Loading","metadata":{}},{"cell_type":"code","source":"ROOT = '/kaggle/input/home-credit-credit-risk-model-stability'\n!tree  '/kaggle/input/home-credit-credit-risk-model-stability' -d \n\ndirectory = ROOT+\"/csv_files/\"\nfolders = ['csv_files/train', 'parquet_files/train', 'csv_files/test', 'parquet_files/test']\nextensions = ['.csv',  '.parquet'] * 2\n\n# We are using only data with depth = 0 \ntrain_base_df  = pd.read_csv(directory + \"train/train_base.csv\")\ntest_base_df   = pd.read_csv(directory + \"test/test_base.csv\")\ntrain_static_cb= pd.read_csv(directory + \"train/train_static_cb_0.csv\")\ntest_static_cb = pd.read_csv(directory + \"test/test_static_cb_0.csv\")\n\ntrain_static0 = pd.read_csv(directory + \"train/train_static_0_0.csv\")\ntrain_static1 = pd.read_csv(directory + \"train/train_static_0_1.csv\")\ntrain_static  = pd.concat([train_static0, train_static1], ignore_index=True)\ndel train_static0, train_static1\n\ntest_static0 = pd.read_csv(directory + \"test/test_static_0_0.csv\")\ntest_static1 = pd.read_csv(directory + \"test/test_static_0_1.csv\")\ntest_static2 = pd.read_csv(directory + \"test/test_static_0_2.csv\")\ntest_static  = pd.concat([test_static0, test_static1, test_static2], ignore_index=True)\ndel test_static0, test_static1, test_static2\n\nfeature_definitions = pd.read_csv(ROOT + \"/feature_definitions.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-05-22T15:37:59.723832Z","iopub.execute_input":"2024-05-22T15:37:59.724774Z","iopub.status.idle":"2024-05-22T15:39:04.972894Z","shell.execute_reply.started":"2024-05-22T15:37:59.724728Z","shell.execute_reply":"2024-05-22T15:39:04.971594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_base_df","metadata":{"execution":{"iopub.status.busy":"2024-05-22T15:39:04.974838Z","iopub.execute_input":"2024-05-22T15:39:04.975291Z","iopub.status.idle":"2024-05-22T15:39:05.006308Z","shell.execute_reply.started":"2024-05-22T15:39:04.975246Z","shell.execute_reply":"2024-05-22T15:39:05.004975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_static","metadata":{"execution":{"iopub.status.busy":"2024-05-22T15:39:05.007651Z","iopub.execute_input":"2024-05-22T15:39:05.007997Z","iopub.status.idle":"2024-05-22T15:39:05.550154Z","shell.execute_reply.started":"2024-05-22T15:39:05.007958Z","shell.execute_reply":"2024-05-22T15:39:05.548977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_static_cb","metadata":{"execution":{"iopub.status.busy":"2024-05-22T15:39:05.551457Z","iopub.execute_input":"2024-05-22T15:39:05.551776Z","iopub.status.idle":"2024-05-22T15:39:06.648363Z","shell.execute_reply.started":"2024-05-22T15:39:05.551750Z","shell.execute_reply":"2024-05-22T15:39:06.647211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are lots of NaN values in the dataset ","metadata":{}},{"cell_type":"markdown","source":"# 2. Data Preparation","metadata":{}},{"cell_type":"markdown","source":"### Merge \"train\":","metadata":{}},{"cell_type":"code","source":"%%time\n# For simplicity, select only columns ending in \"A\" or \"P\" (columns of float type)\nselected_static_cols = []\nfor col in train_static.columns:\n    if col[-1] in (\"A\", \"P\"):\n        selected_static_cols.append(col)\n# print(selected_static_cols)\n\nselected_static_cb_cols = []\nfor col in train_static_cb.columns:\n    if col[-1] in (\"A\", \"P\"):\n        selected_static_cb_cols.append(col)\n# print(selected_static_cb_cols)\n\ntrain_data = pd.merge(train_base_df, train_static[[\"case_id\"]+selected_static_cols],    how=\"left\", on=\"case_id\")\ntrain_data = pd.merge(train_data, train_static_cb[[\"case_id\"]+selected_static_cb_cols], how=\"left\", on=\"case_id\")\ntrain_data","metadata":{"execution":{"iopub.status.busy":"2024-05-22T15:39:06.650051Z","iopub.execute_input":"2024-05-22T15:39:06.650492Z","iopub.status.idle":"2024-05-22T15:39:09.513650Z","shell.execute_reply.started":"2024-05-22T15:39:06.650454Z","shell.execute_reply":"2024-05-22T15:39:09.512452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Merge \"test\":","metadata":{}},{"cell_type":"code","source":"# For simplicity, select only columns ending in \"A\" or \"P\" (columns of float type)\nselected_static_cols = []\nfor col in test_static.columns:\n    if col[-1] in (\"A\", \"P\"):\n        selected_static_cols.append(col)\n# print(selected_static_cols)\n\nselected_static_cb_cols = []\nfor col in test_static_cb.columns:\n    if col[-1] in (\"A\", \"P\"):\n        selected_static_cb_cols.append(col)\n# print(selected_static_cb_cols)\n\ntest_data = pd.merge(test_base_df, test_static[[\"case_id\"]+selected_static_cols],    how=\"left\", on=\"case_id\")\ntest_data = pd.merge(test_data, test_static_cb[[\"case_id\"]+selected_static_cb_cols], how=\"left\", on=\"case_id\")","metadata":{"execution":{"iopub.status.busy":"2024-05-22T15:39:09.515264Z","iopub.execute_input":"2024-05-22T15:39:09.515724Z","iopub.status.idle":"2024-05-22T15:39:09.533471Z","shell.execute_reply.started":"2024-05-22T15:39:09.515686Z","shell.execute_reply":"2024-05-22T15:39:09.532351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. EDA","metadata":{}},{"cell_type":"code","source":"# Target Pie\nplt.figure(figsize=(4, 4))\nplt.suptitle(\"Target class distribution \\n (1 - default, 0 - not default)\", fontsize=15, fontweight='bold', y=1.0)\n\n#p=['#d3d3d3', '#ffa0a0a0']\np=['#d3d3d3', '#ff0000']\n#p=['#a3a3a3', '#ff0000']\nplt.pie(train_data.target.value_counts(), labels = train_data.target.unique(), autopct = '%1.1f%%', \n        colors = sns.color_palette(p), startangle=90, wedgeprops=dict(width=0.3))\n#plt.title('Target Pie')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-22T15:39:09.538973Z","iopub.execute_input":"2024-05-22T15:39:09.539324Z","iopub.status.idle":"2024-05-22T15:39:09.787338Z","shell.execute_reply.started":"2024-05-22T15:39:09.539296Z","shell.execute_reply":"2024-05-22T15:39:09.786261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- The target distribution is imbalanced.","metadata":{}},{"cell_type":"code","source":"def cplot(data, y, split, palette='cool'):\n    sns.catplot(data=data, y=y, hue='target', kind='count', height=(data[y].nunique()+2)/2, aspect=1.0, alpha=0.9,  col = split, palette = palette,\n               sharex=True, sharey=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-22T15:39:09.788836Z","iopub.execute_input":"2024-05-22T15:39:09.789570Z","iopub.status.idle":"2024-05-22T15:39:09.797914Z","shell.execute_reply.started":"2024-05-22T15:39:09.789532Z","shell.execute_reply":"2024-05-22T15:39:09.796666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#p='Pastel1_r'\ns=10\n#tr = train_data[0:200000]\ntr = train_data\n\ntr0 = tr[tr.target==0]\ntr1 = tr[tr.target==1]","metadata":{"execution":{"iopub.status.busy":"2024-05-22T15:39:09.799655Z","iopub.execute_input":"2024-05-22T15:39:09.801001Z","iopub.status.idle":"2024-05-22T15:39:10.181773Z","shell.execute_reply.started":"2024-05-22T15:39:09.800955Z","shell.execute_reply":"2024-05-22T15:39:10.180600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inner structure of data: Heat scatter plot in 2d axis \"Feature-Feature\", Default vs Not Default","metadata":{}},{"cell_type":"code","source":"numerical_features =['credamount_770A', 'annuity_780A', 'pmtssum_45A', 'lastrejectcredamount_222A',\n                     'disbursedcredamount_1113A','price_1097A', 'maxdpdfrom6mto36m_3546853P', 'maxdebt4_972A', 'maxannuity_159A', \n                     'mindbddpdlast24m_3658935P',  'avgdbddpdlast24m_3658932P', \n                     'lastapprcredamount_781A', 'inittransactionamount_650A']\n#excluding: posfstqpd30lastmonth_3976962P actualdpdtolerance_344P posfpd30lastmonth_3976960P posfpd10lastmonth_333P\ngs=100\nn=2 # num of columns\na=0 \n\n\n#tr = train_data[0:20000]\ntr = train_data\ntr0 = tr[tr.target==0]\ntr1 = tr[tr.target==1]\nlsize = 11\nm=1\ncolor_default    = 'red'\ncolor_notdefault = 'darkblue' # 'darkgreen' #'black'#darkblue darkgreen white\ncolorlabels = 'darkblue'\nLabel_size = 12\nfontsize=22\nTitle_size=22\nLabel_size=14\n\nplt.figure(figsize=(18, 7))\nplt.suptitle(\"Default vs  not default\", fontsize=Title_size+2, fontweight='bold', y=1.015)\n\nfor i in numerical_features[:-1]:\n    #plt.figure(figsize=(15, 4)) \n    a=a+1\n    for j in numerical_features[a:]:\n        ax1=feature_definitions[feature_definitions.Variable == i].Description.to_string(index =False)\n        ax2=feature_definitions[feature_definitions.Variable == j].Description.to_string(index =False)\n\n        x_min = pd.concat([tr0[i], tr1[i]]).min()\n        x_max = pd.concat([tr0[i], tr1[i]]).max()\n        y_min = pd.concat([tr0[j], tr1[j]]).min()\n        y_max = pd.concat([tr0[j], tr1[j]]).max()\n        \n        plt.subplot(1, n, m)\n        plt.rcParams['axes.facecolor'] = 'lightgrey'\n        sns.scatterplot(x=tr0[i], y=tr0[j], color = color_notdefault, palette=p, s = s, alpha =0.4)  #notdefault\n        sns.scatterplot(x=tr1[i], y=tr1[j], color = color_default,    palette=p, s = s, alpha =0.3)  #default  \n        \n        plt.title(f'', color='black', fontsize=Title_size)\n        plt.tick_params(axis='x', labelsize=10)\n        plt.tick_params(axis='y', labelsize=10)\n        plt.xlim([x_min, x_max])\n        plt.ylim([y_min, y_max])\n        plt.xlabel(f'{ax1} \\n \"{i}\"', fontsize=Label_size, color = colorlabels)\n        plt.ylabel(f'{ax2} \\n \"{j}\"', fontsize=Label_size, color = colorlabels) \n        plt.grid(color='white')\n        plt.rc('axes',edgecolor='black')\n        plt.legend(['Not default', 'Default'], loc='upper right', prop={'size': 15}, markerscale = 2, shadow=False, \n                   framealpha=0.8, facecolor='white', reverse=True)\n        current_values_x = plt.gca().get_xticks()\n        current_values_y = plt.gca().get_yticks()\n        if current_values_x.max() >= 1000:\n            plt.gca().set_xticklabels(['{:,.0f}'.format(z).replace(',', ' ') for z in current_values_x])\n        if current_values_y.max() >= 1000:\n            plt.gca().set_yticklabels(['{:,.0f}'.format(z).replace(',', ' ') for z in current_values_y])\n        m=-m+3\n        if m ==1:\n            plt.show()\n            plt.figure(figsize=(18, 7))\n","metadata":{"execution":{"iopub.status.busy":"2024-05-22T17:54:44.082577Z","iopub.execute_input":"2024-05-22T17:54:44.083030Z","iopub.status.idle":"2024-05-22T17:57:02.742259Z","shell.execute_reply.started":"2024-05-22T17:54:44.082974Z","shell.execute_reply":"2024-05-22T17:57:02.741100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"> ","metadata":{}},{"cell_type":"markdown","source":"## Inner structure of data: Heat scatter plot in 2d axis \"Feature-Feature\", Default vs Not default","metadata":{}},{"cell_type":"markdown","source":"- посмотрим на вид распределений\n- сравним распределение для дефолта и не дефолта difference between \"default\" and \"not default\"\n- сделаем это в одинаковых осях (масштабе)","metadata":{}},{"cell_type":"code","source":"numerical_features =['credamount_770A', 'annuity_780A', 'pmtssum_45A', 'lastrejectcredamount_222A',\n                     'disbursedcredamount_1113A','price_1097A', 'maxdpdfrom6mto36m_3546853P', 'maxdebt4_972A', 'maxannuity_159A', \n                     'mindbddpdlast24m_3658935P',  'avgdbddpdlast24m_3658932P', \n                     'lastapprcredamount_781A', 'inittransactionamount_650A']\n#excluding: posfstqpd30lastmonth_3976962P actualdpdtolerance_344P posfpd30lastmonth_3976960P posfpd10lastmonth_333P\ngs=100\nn=2 # num of columns\na=0 \ncolorlabels = 'darkblue'\n\n#tr = train_data[0:20000]\ntr = train_data\ntr0 = tr[tr.target==0]\ntr1 = tr[tr.target==1]\nLabel_size = 14 # Size font of xy labels\nTitle_size = 24 # Size font of Title\n\n#plt.figure(figsize=(18, 4))    \nfor i in numerical_features[:-1]:\n    #plt.figure(figsize=(15, 5)) \n    a=a+1\n    for j in numerical_features[a:]:\n        ax1=feature_definitions[feature_definitions.Variable == i].Description.to_string(index = False)\n        ax2=feature_definitions[feature_definitions.Variable == j].Description.to_string(index = False)\n        ax1 = ax1[:-1]\n        ax2 = ax2[:-1]\n\n        x_min = pd.concat([tr0[i], tr1[i]]).min()\n        x_max = pd.concat([tr0[i], tr1[i]]).max()\n        y_min = pd.concat([tr0[j], tr1[j]]).min()\n        y_max = pd.concat([tr0[j], tr1[j]]).max()\n        \n        plt.figure(figsize=(15, 5))\n        plt.suptitle(f'Heat Scatter in 2d-axis: \\n\"{ax1}\" vs \\n\"{ax2}\"', fontsize=Title_size, y=1.2,  fontweight='bold')        \n        plt.subplot(1, n, 1)\n        plt.hexbin(tr0[i], tr0[j],  gridsize=gs, cmap='CMRmap', bins='log', alpha = 1)\n        plt.colorbar().set_label(label='count in bin',size=10, color  = 'grey')\n        plt.title(f'Not default (Target = 0)', color='green', fontsize=Title_size)\n        plt.tick_params(axis='x', labelsize=10)\n        plt.tick_params(axis='y', labelsize=10)\n        plt.xlim([x_min, x_max])\n        plt.ylim([y_min, y_max])\n        plt.xlabel(f'{ax1} \\n ({i})', fontsize=Label_size, color = colorlabels)\n        plt.ylabel(f'{ax2} \\n ({j})', fontsize=Label_size, color = colorlabels) \n        plt.rcParams['axes.facecolor'] = 'lightgrey'\n        plt.grid(color='white')\n        plt.rc('axes',edgecolor='black')\n        current_values_x = plt.gca().get_xticks()\n        current_values_y = plt.gca().get_yticks()\n        if current_values_x.max() >= 1000:\n            plt.gca().set_xticklabels(['{:,.0f}'.format(z).replace(',', ' ') for z in current_values_x])\n        if current_values_y.max() >= 1000:\n            plt.gca().set_yticklabels(['{:,.0f}'.format(z).replace(',', ' ') for z in current_values_y])\n                \n        plt.subplot(1, n, 2)\n        plt.hexbin(tr1[i], tr1[j], gridsize=gs-20, cmap='CMRmap', bins='log', alpha = 1)\n        plt.colorbar().set_label(label='count in bin',size=8, color  = 'grey')\n        plt.title(f'Default (Target = 1)', color='r', fontsize=Title_size)\n        plt.tick_params(axis='x', labelsize=9) \n        plt.tick_params(axis='y', labelsize=9)\n        plt.xlim([x_min, x_max])\n        plt.ylim([y_min, y_max])\n        plt.xlabel(f'{ax1} \\n \"{i}\"', fontsize=Label_size, color = colorlabels)\n        plt.ylabel(f'{ax2} \\n \"{j}\"', fontsize=Label_size, color = colorlabels)   \n        plt.rcParams['axes.facecolor'] = 'lightgrey'\n        plt.grid(color='white')\n        plt.rc('axes',edgecolor='black')\n        current_values_x = plt.gca().get_xticks()\n        current_values_y = plt.gca().get_yticks()\n        if current_values_x.max() >= 1000:\n            plt.gca().set_xticklabels(['{:,.0f}'.format(z).replace(',', ' ') for z in current_values_x])\n        if current_values_y.max() >= 1000:\n            plt.gca().set_yticklabels(['{:,.0f}'.format(z).replace(',', ' ') for z in current_values_y])\n        \n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-22T18:14:43.350088Z","iopub.execute_input":"2024-05-22T18:14:43.351040Z","iopub.status.idle":"2024-05-22T18:16:44.784392Z","shell.execute_reply.started":"2024-05-22T18:14:43.350996Z","shell.execute_reply":"2024-05-22T18:16:44.783022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr = train_data\n\n#tr0= non default\n#tr1= default\n","metadata":{"execution":{"iopub.status.busy":"2024-05-22T15:45:34.705573Z","iopub.status.idle":"2024-05-22T15:45:34.706122Z","shell.execute_reply.started":"2024-05-22T15:45:34.705843Z","shell.execute_reply":"2024-05-22T15:45:34.705863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sns.catplot( data=train, x=\"target\", y=\"pmtssum_45A\", height=4, aspect=.6, alpha = 0.7)","metadata":{"execution":{"iopub.status.busy":"2024-05-22T15:45:34.707805Z","iopub.status.idle":"2024-05-22T15:45:34.708227Z","shell.execute_reply.started":"2024-05-22T15:45:34.708031Z","shell.execute_reply":"2024-05-22T15:45:34.708048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sns.catplot(    data=train, x=\"target\", y=\"annuity_780A\", kind=\"violin\", bw_adjust=.5, cut=0, split=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-22T15:45:34.709827Z","iopub.status.idle":"2024-05-22T15:45:34.710251Z","shell.execute_reply.started":"2024-05-22T15:45:34.710056Z","shell.execute_reply":"2024-05-22T15:45:34.710073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sns.catplot(    data=train, x=\"target\", y=\"annuity_780A\", height=4, aspect=.6)","metadata":{"execution":{"iopub.status.busy":"2024-05-22T15:45:34.712022Z","iopub.status.idle":"2024-05-22T15:45:34.712548Z","shell.execute_reply.started":"2024-05-22T15:45:34.712269Z","shell.execute_reply":"2024-05-22T15:45:34.712290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sns.catplot(   data=train, x=\"target\", y=\"lastrejectcredamount_222A\", height=4, aspect=.6)\n#sns.swarmplot(data=train, x=\"target\", y=\"lastrejectcredamount_222A\", size=3) # оч долго\n#sns.catplot(data=train, x=\"target\", y=\"lastrejectcredamount_222A\", kind=\"violin\", bw_adjust=.5, cut=0, split=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-22T15:45:34.714143Z","iopub.status.idle":"2024-05-22T15:45:34.714551Z","shell.execute_reply.started":"2024-05-22T15:45:34.714350Z","shell.execute_reply":"2024-05-22T15:45:34.714368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Modelling","metadata":{}},{"cell_type":"markdown","source":"## Feature Important ","metadata":{}},{"cell_type":"code","source":"%%time\ntrain = train_data.drop(['case_id', 'date_decision'], axis = 1)[0:10000]\n#train_data.columns\nimport eli5\nfrom eli5.sklearn import PermutationImportance\n\nX = train.drop(['target'], axis = 1)\ny = train['target']\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.2, random_state = 24)\n\npermut = PermutationImportance(XGBClassifier(random_state=0).fit(X, y),random_state=1).fit(X, y)\neli5.show_weights(permut, feature_names = X.columns.tolist())","metadata":{"execution":{"iopub.status.busy":"2024-05-22T15:45:34.718321Z","iopub.status.idle":"2024-05-22T15:45:34.718725Z","shell.execute_reply.started":"2024-05-22T15:45:34.718525Z","shell.execute_reply":"2024-05-22T15:45:34.718542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## XGBoost Modelling","metadata":{}},{"cell_type":"code","source":"%%time\ntrain_data2 = train_data.drop(columns = [\"case_id\", \"MONTH\", \"WEEK_NUM\", \"date_decision\"], inplace = False)\ntest_data2  = test_data.drop(columns  = [\"case_id\", \"MONTH\", \"WEEK_NUM\", \"date_decision\"], inplace = False)\nX = train_data2.drop(['target'], axis=1)\ny = train_data2['target']\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-05-22T18:06:17.265376Z","iopub.execute_input":"2024-05-22T18:06:17.265873Z","iopub.status.idle":"2024-05-22T18:06:18.923138Z","shell.execute_reply.started":"2024-05-22T18:06:17.265836Z","shell.execute_reply":"2024-05-22T18:06:18.921974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_classifier = XGBClassifier()\nxgb_classifier.fit(X_train, y_train)\n\ny_train_pred = xgb_classifier.predict(X_train)\ny_test_pred  = xgb_classifier.predict(X_test )","metadata":{"execution":{"iopub.status.busy":"2024-05-22T18:06:42.317992Z","iopub.execute_input":"2024-05-22T18:06:42.318419Z","iopub.status.idle":"2024-05-22T18:07:06.394769Z","shell.execute_reply.started":"2024-05-22T18:06:42.318385Z","shell.execute_reply":"2024-05-22T18:07:06.393455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluation","metadata":{}},{"cell_type":"code","source":"n=60\ndef text(text):\n    bold = '\\033[1m'\n    bold_end = '\\033[0m'\n    \n    red = '\\033[91m'\n    RED_end = '\\033[910m'\n    \n    green = '\\033[92m'\n    green_end = '\\033[920m'\n    \n    BLUE = '\\033[94m'\n    BLUE_end = '\\033[940m'\n    return bold + BLUE + text + BLUE_end + bold_end","metadata":{"execution":{"iopub.status.busy":"2024-05-22T18:07:33.297781Z","iopub.execute_input":"2024-05-22T18:07:33.298222Z","iopub.status.idle":"2024-05-22T18:07:33.305546Z","shell.execute_reply.started":"2024-05-22T18:07:33.298188Z","shell.execute_reply":"2024-05-22T18:07:33.304484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_train = list(map(round, xgb_classifier.predict(X_train)))\nprediction_test  = list(map(round, xgb_classifier.predict(X_test )))\n\n# accuracy_train = accuracy_score(y_train, prediction_train).round(4)\n# accuracy_test  = accuracy_score(y_test,  prediction_test).round(4)\n\n# print(text(\"ACCURACY\"))\n# print(\"Accuracy on Train set: {:.1f}%\".format(accuracy_train * 100))\n# print(\"Accuracy on Test set:  {:.1f}%\".format(accuracy_test  * 100))\n# print(\"-\"*n)\n# print()\n\nprint(text(\"CONFUSION MATRIX\"))\nprint(\"Confusion Matrix on Train set:\\n\", confusion_matrix(y_train, prediction_train))\nprint()\nprint(\"Confusion Matrix on Test  set:\\n\", confusion_matrix(y_test,  prediction_test))\nprint(\"-\"*n) \nprint()\n\nprint(text(\"CLASSIFICATION REPORT\"))\nprint(\"Classification Report on Train set:\\n\\n\", classification_report(y_train, prediction_train))\nprint()\nprint(\"Classification Report on Test set:\\n\\n\",  classification_report(y_test,  prediction_test))\n","metadata":{"execution":{"iopub.status.busy":"2024-05-22T18:08:03.609811Z","iopub.execute_input":"2024-05-22T18:08:03.610279Z","iopub.status.idle":"2024-05-22T18:08:13.164222Z","shell.execute_reply.started":"2024-05-22T18:08:03.610243Z","shell.execute_reply":"2024-05-22T18:08:13.162907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(11, 5))\n#fig = plt.figure()\nplt.suptitle(\"ROC Curve (Receiver Operating Characteristic) \", fontsize=15, fontweight='bold', y=1.05)\n\nplt.subplot(1, 2, 1)\nfact=y_train\npred=xgb_classifier.predict(X_train)\nfpr, tpr, thresholds = roc_curve(fact, pred)\nroc_auc = auc(fpr, tpr)\ngini = 2*roc_auc-1\nplt.plot(fpr, tpr, color='darkorange', lw=2, label=f'ROC Curve (AUC = {roc_auc:.4f})')\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--', label='Random')\nplt.title('Train set \\n  Gini= {:.3f}'.format(gini), fontsize=13)\n#plt.title('Train set ', fontsize=13)\nplt.xlabel('FPR \\n(False Positive Rate)', fontsize=10)\nplt.ylabel('TPR \\n(True Positive Rate)' , fontsize=10)\nplt.legend(loc=\"lower right\", fontsize=10)\n\nplt.subplot(1, 2, 2)\nfact=y_test\npred=xgb_classifier.predict(X_test)\nfpr, tpr, thresholds = roc_curve(fact, pred)\nroc_auc = auc(fpr, tpr)\ngini = 2*roc_auc-1\nplt.plot(fpr, tpr, color='darkorange', lw=2, label=f'ROC Curve (AUC = {roc_auc:.4f})')\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--', label='Random')\nplt.title('Train set \\n  Gini= {:.3f}'.format(gini), fontsize=13)\nplt.xlabel('FPR \\n(False Positive Rate)', fontsize=10)\nplt.ylabel('TPR \\n(True Positive Rate)',  fontsize=10)\nplt.legend(loc=\"lower right\", fontsize=10)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-22T18:12:54.396896Z","iopub.execute_input":"2024-05-22T18:12:54.397383Z","iopub.status.idle":"2024-05-22T18:12:58.746687Z","shell.execute_reply.started":"2024-05-22T18:12:54.397329Z","shell.execute_reply":"2024-05-22T18:12:58.745561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Submit","metadata":{}},{"cell_type":"code","source":"predictions = xgb_classifier.predict_proba(test_data)[:, 1]\ntest_Id = test_base_df[\"case_id\"]\nsubmission = pd.DataFrame({\n    'case_id': test_Id,\n    'score': predictions\n})\n\nsubmission.to_csv('submission.csv', index=False)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-22T15:45:34.731773Z","iopub.status.idle":"2024-05-22T15:45:34.732188Z","shell.execute_reply.started":"2024-05-22T15:45:34.731991Z","shell.execute_reply":"2024-05-22T15:45:34.732008Z"},"trusted":true},"execution_count":null,"outputs":[]}]}