{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":7624767,"sourceType":"datasetVersion","datasetId":4441829},{"sourceId":9537572,"sourceType":"datasetVersion","datasetId":5809399}],"dockerImageVersionId":30648,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"colab":{"provenance":[]}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import Libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.linear_model import LogisticRegression\nfrom tqdm import tqdm\nfrom imblearn.over_sampling import SMOTE\nfrom sklearn.utils import resample\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import GridSearchCV\n\nfrom sklearn.linear_model import LogisticRegression\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\n\nimport xgboost as xgb\nimport lightgbm as lgb\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.svm import LinearSVC\nfrom sklearn.calibration import CalibratedClassifierCV\nfrom sklearn.ensemble import StackingClassifier\nfrom scipy.stats import mode\n\nimport torch\nimport numpy as np\nfrom sklearn.preprocessing import MinMaxScaler, LabelEncoder, StandardScaler\nfrom sklearn.metrics import log_loss\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.metrics import roc_curve\nimport sklearn.metrics as metrics\n\nimport gc\nfrom sklearn.model_selection import train_test_split, RandomizedSearchCV\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import roc_auc_score\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"700af4b2-ca93-4a2f-b7de-0fef3c212eb6","_cell_guid":"34ea337a-c686-4618-bfe8-ed89858725b8","jupyter":{"outputs_hidden":false},"id":"_oaSvR_0C8N5","outputId":"7af948ec-9846-42f6-f32b-a159f5dd4854","collapsed":false,"execution":{"iopub.status.busy":"2024-10-24T23:55:40.473496Z","iopub.execute_input":"2024-10-24T23:55:40.474077Z","iopub.status.idle":"2024-10-24T23:55:40.487675Z","shell.execute_reply.started":"2024-10-24T23:55:40.474040Z","shell.execute_reply":"2024-10-24T23:55:40.486091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import Dataset","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/dstcdata/Bảng ĐH_Vòng 2_DSTC 2024_Dataset/01_dataset.csv')\n","metadata":{"_uuid":"45ea90fc-a81b-4a0f-a334-148d814c683f","_cell_guid":"9ddfadf3-666d-43b5-9be1-9d10a05c95c4","jupyter":{"outputs_hidden":false},"id":"HJc0_F2eC8N6","outputId":"878890ce-ec33-4313-faac-9213c3fc0bc6","collapsed":false,"execution":{"iopub.status.busy":"2024-10-24T23:55:40.489959Z","iopub.execute_input":"2024-10-24T23:55:40.490347Z","iopub.status.idle":"2024-10-24T23:55:40.816684Z","shell.execute_reply.started":"2024-10-24T23:55:40.490315Z","shell.execute_reply":"2024-10-24T23:55:40.815445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"_uuid":"eedbd5f1-ab35-40b7-adbe-a41b8a17f449","_cell_guid":"0d6c387c-ef0d-4e41-b0ee-ac2c2c2a6ffd","jupyter":{"outputs_hidden":false},"id":"DrewREjTC8N7","collapsed":false,"execution":{"iopub.status.busy":"2024-10-24T23:55:40.820520Z","iopub.execute_input":"2024-10-24T23:55:40.820911Z","iopub.status.idle":"2024-10-24T23:55:40.848259Z","shell.execute_reply.started":"2024-10-24T23:55:40.820880Z","shell.execute_reply":"2024-10-24T23:55:40.847256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:40.849492Z","iopub.execute_input":"2024-10-24T23:55:40.849879Z","iopub.status.idle":"2024-10-24T23:55:40.861357Z","shell.execute_reply.started":"2024-10-24T23:55:40.849844Z","shell.execute_reply":"2024-10-24T23:55:40.860219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filtered_df = df[(df[\"NUMBER_OF_RELATIONSHIP\"] >= 7) &\n                 (df[\"NUMBER_OF_LOANS\"] >= 4) &\n                 (df[\"NUM_NEW_LOAN_TAKEN_3M\"] >= 4)]\nlen(filtered_df[filtered_df[\"label\"]==1])/len(filtered_df)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:40.864095Z","iopub.execute_input":"2024-10-24T23:55:40.864456Z","iopub.status.idle":"2024-10-24T23:55:40.882350Z","shell.execute_reply.started":"2024-10-24T23:55:40.864424Z","shell.execute_reply":"2024-10-24T23:55:40.881214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filtered_df = df[(df[\"NUMBER_OF_RELATIONSHIP\"] < 7) &\n                 (df[\"NUMBER_OF_LOANS\"] < 4) &\n                 (df[\"NUM_NEW_LOAN_TAKEN_3M\"] < 4)]\nlen(filtered_df[filtered_df[\"label\"]==1])/len(filtered_df)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:40.883710Z","iopub.execute_input":"2024-10-24T23:55:40.884105Z","iopub.status.idle":"2024-10-24T23:55:40.897281Z","shell.execute_reply.started":"2024-10-24T23:55:40.884071Z","shell.execute_reply":"2024-10-24T23:55:40.896036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Initial Processing","metadata":{"_uuid":"d1af0853-06e6-4242-ae8a-2041a3cdb0a5","_cell_guid":"9e64cb4f-d3cf-4a32-b29e-dd968b6d2040","id":"fF905yMIC8N7","trusted":true}},{"cell_type":"markdown","source":"### Missing Values","metadata":{"_uuid":"3572b4c1-f150-4e5f-bfbc-ed65f038e19d","_cell_guid":"437f7759-ca8d-4907-b4fb-7731ce373ebe","id":"H7_RFr7ZC8N9","trusted":true}},{"cell_type":"code","source":"df.isnull().sum(axis = 0)","metadata":{"_uuid":"6fea26ac-727f-4957-9b5f-76b0c2f1e834","_cell_guid":"a72f3c65-72d8-4301-983e-6a48182d68b6","jupyter":{"outputs_hidden":false},"id":"hJASX1AsC8N-","collapsed":false,"execution":{"iopub.status.busy":"2024-10-24T23:55:40.898946Z","iopub.execute_input":"2024-10-24T23:55:40.899666Z","iopub.status.idle":"2024-10-24T23:55:40.915600Z","shell.execute_reply.started":"2024-10-24T23:55:40.899622Z","shell.execute_reply":"2024-10-24T23:55:40.914735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Label Distribution","metadata":{"_uuid":"682e36fb-c15d-46e7-9b78-178594cea178","_cell_guid":"4f8904f0-2954-41f9-8980-7317c968f426","id":"YX12ogv1C8N-","trusted":true}},{"cell_type":"code","source":"fig = plt.figure(figsize=(6,5))\nplt.bar(df.label.value_counts().index, df.label.value_counts().values)\nplt.xlabel('label')\nplt.ylabel('counts')\nplt.xticks([0,1])\nplt.show()","metadata":{"_uuid":"f224d6b9-a9ad-44c6-bcd1-8aa5df4b7775","_cell_guid":"d342da90-ac37-41e3-9294-d672f00cd87c","jupyter":{"outputs_hidden":false},"id":"mhrSIte5C8N_","collapsed":false,"execution":{"iopub.status.busy":"2024-10-24T23:55:40.916875Z","iopub.execute_input":"2024-10-24T23:55:40.917202Z","iopub.status.idle":"2024-10-24T23:55:41.074655Z","shell.execute_reply.started":"2024-10-24T23:55:40.917175Z","shell.execute_reply":"2024-10-24T23:55:41.073551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fill missing values with mean","metadata":{"id":"P_FriXavC8OA"}},{"cell_type":"code","source":"numerical_cols = df.select_dtypes(include='number').columns\ncategorical_cols = df.select_dtypes(include='object').columns\n\n\nfor col in df.select_dtypes('object').columns:\n    df[col].fillna(df[col].mode()[0], inplace=True)\nfor col in df.select_dtypes('number').columns:\n    df[col].fillna(df[col].mean(), inplace=True)\n","metadata":{"id":"ke8W8L7OC8OA","execution":{"iopub.status.busy":"2024-10-24T23:55:41.076183Z","iopub.execute_input":"2024-10-24T23:55:41.076611Z","iopub.status.idle":"2024-10-24T23:55:41.155517Z","shell.execute_reply.started":"2024-10-24T23:55:41.076572Z","shell.execute_reply":"2024-10-24T23:55:41.153119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"id":"tG1JbMcfpcdR","execution":{"iopub.status.busy":"2024-10-24T23:55:41.157058Z","iopub.execute_input":"2024-10-24T23:55:41.157516Z","iopub.status.idle":"2024-10-24T23:55:41.164474Z","shell.execute_reply.started":"2024-10-24T23:55:41.157462Z","shell.execute_reply":"2024-10-24T23:55:41.163406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ROC curve","metadata":{"id":"RS2AD3eLC8OB"}},{"cell_type":"code","source":"results = []\nfrom sklearn.metrics import roc_curve, accuracy_score, classification_report\nfrom sklearn.metrics import roc_auc_score, precision_score, recall_score, f1_score\n\ndef evaluate(y_test,y_pred,y_pred_proba, model):\n    # Calculate accuracy\n    accuracy = accuracy_score(y_test, y_pred)\n    # Calculate ROC curve\n    fpr, tpr, _ = roc_curve(y_test, y_pred_proba)\n    # Plot the ROC curve\n    plt.figure(figsize=(10, 7))\n    plt.plot(fpr, tpr, color='blue', label='ROC Curve')\n    plt.fill_between(fpr, 0, tpr, color='skyblue', alpha=0.2)\n    plt.plot([0, 1], [0, 1], color='red', linestyle='--', label='Random Chance')\n    # Add labels and title\n    plt.ylabel('True Positive Rate')\n    plt.xlabel('False Positive Rate')\n    plt.title('ROC Curve for {}'.format(model))\n\n    # Generate classification report\n    report = classification_report(y_test, y_pred)\n    classification_report_str = classification_report(y_test, y_pred, output_dict=True)\n    f1_score = classification_report_str['weighted avg']['f1-score']\n    auc_score = roc_auc_score(y_test, y_pred_proba)\n    gini = 2 * auc_score - 1\n\n    \n\n    # Print accuracy and classification report\n    print(\"\\nClassification Report:\\n\", report)\n    print(\"\\nGini Index: {:.2f}\", gini)\n\n    # Add result\n    results.append({\"Model\": model,\"GINI\": gini, \"ROC AUC Score\": auc_score})\n    # Show the plot\n    plt.show()","metadata":{"id":"wD4zehBmC8OC","execution":{"iopub.status.busy":"2024-10-24T23:55:41.165795Z","iopub.execute_input":"2024-10-24T23:55:41.166119Z","iopub.status.idle":"2024-10-24T23:55:41.178370Z","shell.execute_reply.started":"2024-10-24T23:55:41.166092Z","shell.execute_reply":"2024-10-24T23:55:41.176946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Baseline Models Performances","metadata":{}},{"cell_type":"markdown","source":"## Baseline Model - Logistic Regression","metadata":{"_uuid":"b7c3705c-0459-49df-97b0-db93ccdae2c0","_cell_guid":"dce875e1-bbf5-4081-9458-7346d36e507a","id":"1oRek7VwC8OC","trusted":true}},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import train_test_split\n\n# Assuming your data is stored in X and y\nX_train, X_val_test, y_train, y_val_test = train_test_split(df.drop(\"label\", axis=1), df[\"label\"], test_size=0.4, random_state=42)\n# Split the validation and test sets into equal proportions\nX_val, X_test, y_val, y_test = train_test_split(X_val_test, y_val_test, test_size=0.5, random_state=42)\nX_train.drop('customer_id', axis=1, inplace=True)\nX_val.drop('customer_id', axis=1, inplace=True)\nprint(\"Train set size:\", len(X_train))\nprint(\"Validation set size:\", len(X_val))\nprint(\"Test set size:\", len(X_test))","metadata":{"_uuid":"317ed048-bd30-4fa0-b938-bf68d66963a8","_cell_guid":"100f348c-dc24-4ac3-8277-5b89379dde11","jupyter":{"outputs_hidden":false},"id":"kswvZsqWC8OD","collapsed":false,"execution":{"iopub.status.busy":"2024-10-24T23:55:41.179928Z","iopub.execute_input":"2024-10-24T23:55:41.180275Z","iopub.status.idle":"2024-10-24T23:55:41.221663Z","shell.execute_reply.started":"2024-10-24T23:55:41.180245Z","shell.execute_reply":"2024-10-24T23:55:41.220549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\n# Initialize logistic regression model with 'sag' solver\nmodel = LogisticRegression()\n\n# Train the model\nmodel.fit(X_train, y_train)\n\n# Predictions\ny_pred = model.predict(X_val)\ny_pred_proba = model.predict_proba(X_val)[:, 1]\n\n# Evaluate the model\naccuracy = accuracy_score(y_val, y_pred)\nauc_score = roc_auc_score(y_val, y_pred_proba)\n\nprint(f\"Accuracy: {accuracy * 100.0}%\")\nprint(f\"AUC Score: {auc_score}\")\nevaluate(y_val,y_pred,y_pred_proba, \"Baseline Logistic Regression\")","metadata":{"_uuid":"2401988b-9f6b-4cd9-a4eb-42146eeaccc0","_cell_guid":"2a7b16c3-2e0f-459b-9fad-d02d148a3d6e","jupyter":{"outputs_hidden":false},"id":"okwiYOTtC8OD","collapsed":false,"execution":{"iopub.status.busy":"2024-10-24T23:55:41.222882Z","iopub.execute_input":"2024-10-24T23:55:41.223176Z","iopub.status.idle":"2024-10-24T23:55:41.654846Z","shell.execute_reply.started":"2024-10-24T23:55:41.223151Z","shell.execute_reply":"2024-10-24T23:55:41.653662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Baseline Model - LightGBM","metadata":{"id":"OadRmFJqC8OE"}},{"cell_type":"code","source":"lgb_model = lgb.LGBMClassifier()\n\n# Perform cross-validation\ncv_scores = cross_val_score(lgb_model, X_train, y_train, cv=5, scoring='roc_auc')\n\n# Train the model on the entire training set\nlgb_model.fit(X_train, y_train)\n\n# Predictions on the test set\nlgb_y_pred = lgb_model.predict(X_val)\ny_pred_proba = lgb_model.predict_proba(X_val)[:, 1]\n# Calculate accuracy and AUC Score on the test set\nlgb_accuracy = accuracy_score(y_val, lgb_y_pred)\nlgb_auc_score = roc_auc_score(y_val, y_pred_proba)\n\nprint(f\"Cross-Validation ROC-AUC Scores: {cv_scores}\")\nprint(f\"Average Cross-Validation ROC-AUC Score: {np.mean(cv_scores)}\")\nprint(f\"Test Set Accuracy: {lgb_accuracy * 100.0}%\")\nprint(f\"Test Set AUC Score: {lgb_auc_score}\")\n\n\naccuracy = accuracy_score(y_val, y_pred)\nauc_score = roc_auc_score(y_val, y_pred_proba)\n\nprint(f\"Accuracy: {accuracy * 100.0}%\")\nprint(f\"AUC Score: {auc_score}\")\nevaluate(y_val,lgb_y_pred,y_pred_proba, \"Light GBM Baseline\")","metadata":{"id":"mjEctgyoC8OE","execution":{"iopub.status.busy":"2024-10-24T23:55:41.661691Z","iopub.execute_input":"2024-10-24T23:55:41.662089Z","iopub.status.idle":"2024-10-24T23:55:53.356694Z","shell.execute_reply.started":"2024-10-24T23:55:41.662057Z","shell.execute_reply":"2024-10-24T23:55:53.355641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb.plot_importance(lgb_model, importance_type=\"gain\", max_num_features=10, figsize=(10,5), title=\"LightGBM Feature Importance (Gain)\", grid=False, color = \"#009dd9\")\nplt.show()\n","metadata":{"id":"xn8Rl2c6KWNf","execution":{"iopub.status.busy":"2024-10-24T23:55:53.358021Z","iopub.execute_input":"2024-10-24T23:55:53.358334Z","iopub.status.idle":"2024-10-24T23:55:53.642009Z","shell.execute_reply.started":"2024-10-24T23:55:53.358306Z","shell.execute_reply":"2024-10-24T23:55:53.640671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{"_uuid":"ea51b032-4611-4b22-8c3c-c412f33356b9","_cell_guid":"21014213-b229-4259-83cb-e1fea8fb2ad4","id":"MNeSejyUC8OF","trusted":true}},{"cell_type":"markdown","source":"### Drop Features that has too many duplicate values","metadata":{"_uuid":"1fa97815-10b1-4c04-89d6-1b00544dcd26","_cell_guid":"f8e68432-d21c-4e41-ba11-aa05f84142e2","id":"u2rnVKR0C8OF","trusted":true}},{"cell_type":"code","source":"df.drop([\"CREDIT_CARD_MONTH_SINCE_10DPD\",\"CREDIT_CARD_MONTH_SINCE_30DPD\",\"CREDIT_CARD_MONTH_SINCE_60DPD\",\"CREDIT_CARD_MONTH_SINCE_90DPD\"], inplace = True, axis = 1)","metadata":{"id":"PC1BHwCdC8OG","execution":{"iopub.status.busy":"2024-10-24T23:55:53.643439Z","iopub.execute_input":"2024-10-24T23:55:53.643878Z","iopub.status.idle":"2024-10-24T23:55:53.655650Z","shell.execute_reply.started":"2024-10-24T23:55:53.643843Z","shell.execute_reply":"2024-10-24T23:55:53.654437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature 1: Binary Features (Short Term, Mid Term, Long Term)","metadata":{"id":"Vd4EHGJ1hzEB"}},{"cell_type":"code","source":"df_t = df[['SHORT_TERM_COUNT', 'MID_TERM_COUNT', 'LONG_TERM_COUNT', 'label']].copy(deep=True)\ndf_t['MANY_SHORT'] = np.where(df_t['SHORT_TERM_COUNT'] >= 4, 1, 0)\ndf_t['MANY_MID'] = np.where(df_t['MID_TERM_COUNT'] >= 4, 1, 0)\ndf_t['MANY_LONG'] = np.where(df_t['LONG_TERM_COUNT'] >= 4, 1, 0)\ndf_t['Theory 1'] = (df_t['MANY_SHORT'] * 10 + df_t['MANY_MID'] * 5 + df_t['MANY_LONG']) / 15\ndf['Diff_in_term_count'] = df_t['Theory 1']","metadata":{"id":"3HLHfiHug1Sa","execution":{"iopub.status.busy":"2024-10-24T23:55:53.657320Z","iopub.execute_input":"2024-10-24T23:55:53.657706Z","iopub.status.idle":"2024-10-24T23:55:53.674053Z","shell.execute_reply.started":"2024-10-24T23:55:53.657674Z","shell.execute_reply":"2024-10-24T23:55:53.672806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature 2: Binary Features (Number of Loans, Number of Loans Bank, Number of Loans Non Bank) ","metadata":{"id":"uqWokysWh2Ly"}},{"cell_type":"code","source":"df_t = df[['NUMBER_OF_LOANS', 'NUMBER_OF_LOANS_BANK', 'NUMBER_OF_LOANS_NON_BANK', 'label']].copy(deep=True)\ndf_t['MANY_LOANS'] = np.where(df_t['NUMBER_OF_LOANS'] >= 4, 1, 0)\ndf_t['MANY_LOANS_BANK'] = np.where(df_t['NUMBER_OF_LOANS_BANK'] >= 4, 1, 0)\ndf_t['MANY_LOANS_NON_BANK'] = np.where(df_t['NUMBER_OF_LOANS_NON_BANK'] >= 4, 1, 0)\ndf_t['Theory 1'] = df_t.iloc[:, -3:].sum(axis=1)\ndf['Diff_in_num_loan'] = df_t['Theory 1']","metadata":{"id":"MUSh9Dszh7Ow","execution":{"iopub.status.busy":"2024-10-24T23:55:53.676070Z","iopub.execute_input":"2024-10-24T23:55:53.676499Z","iopub.status.idle":"2024-10-24T23:55:53.695389Z","shell.execute_reply.started":"2024-10-24T23:55:53.676460Z","shell.execute_reply":"2024-10-24T23:55:53.694049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature 3: Differece in 3 features realted to Number of Relationships","metadata":{"id":"_2RUiAWhh33-"}},{"cell_type":"code","source":"df_t = df[['NUMBER_OF_RELATIONSHIP', 'NUMBER_OF_RELATIONSHIP_BANK', 'NUMBER_OF_RELATIONSHIP_NON_BANK', 'label']].copy(deep=True)\n# df_t.median()\ndf_t['MANY_RELATIONSHIP'] = np.where(df_t['NUMBER_OF_RELATIONSHIP'] >= 4, 1, 0)\ndf_t['MANY_RELATIONSHIP_BANK'] = np.where(df_t['NUMBER_OF_RELATIONSHIP_BANK'] >= 4, 1, 0)\ndf_t['MANY_RELATIONSHIP_NON_BANK'] = np.where(df_t['NUMBER_OF_RELATIONSHIP_NON_BANK'] >= 4, 1, 0)\n\ndf_t['Theory 1'] = (df_t['MANY_RELATIONSHIP'] * 5 + df_t['MANY_RELATIONSHIP_BANK'] + df_t['MANY_RELATIONSHIP_NON_BANK']) / 7\ndf['Diff_in_num_relationship'] = df_t['Theory 1']","metadata":{"id":"J03atiSwiESp","execution":{"iopub.status.busy":"2024-10-24T23:55:53.697026Z","iopub.execute_input":"2024-10-24T23:55:53.697491Z","iopub.status.idle":"2024-10-24T23:55:53.711890Z","shell.execute_reply.started":"2024-10-24T23:55:53.697432Z","shell.execute_reply":"2024-10-24T23:55:53.710615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature 4","metadata":{"id":"DnPpDhhZh5gm"}},{"cell_type":"code","source":"cols = ['NUM_NEW_LOAN_TAKEN_3M', 'NUM_NEW_LOAN_TAKEN_6M',\n       'NUM_NEW_LOAN_TAKEN_9M', 'NUM_NEW_LOAN_TAKEN_12M',\n       'NUM_NEW_LOAN_TAKEN_BANK_3M', 'NUM_NEW_LOAN_TAKEN_BANK_6M',\n       'NUM_NEW_LOAN_TAKEN_BANK_9M', 'NUM_NEW_LOAN_TAKEN_BANK_12M',\n       'NUM_NEW_LOAN_TAKEN_NON_BANK_3M', 'NUM_NEW_LOAN_TAKEN_NON_BANK_6M',\n       'NUM_NEW_LOAN_TAKEN_NON_BANK_9M', 'NUM_NEW_LOAN_TAKEN_NON_BANK_12M', 'label']\n# cols = ['NUM_NEW_LOAN_TAKEN_3M', 'NUM_NEW_LOAN_TAKEN_6M',\n#        'NUM_NEW_LOAN_TAKEN_9M', 'NUM_NEW_LOAN_TAKEN_12M', 'label']\ndf_t = df[cols].copy(deep=True)\ndf_t['MANY_NEW_LOAN_3M'] = np.where(df_t['NUM_NEW_LOAN_TAKEN_3M'] >= 1, 1, 0)\ndf_t['MANY_NEW_LOAN_6M'] = np.where(df_t['NUM_NEW_LOAN_TAKEN_6M'] >= 4, 1, 0)\ndf_t['MANY_NEW_LOAN_9M'] = np.where(df_t['NUM_NEW_LOAN_TAKEN_9M'] >= 4, 1, 0)\ndf_t['MANY_NEW_LOAN_12M'] = np.where(df_t['NUM_NEW_LOAN_TAKEN_12M'] >= 4, 1, 0)\n\ndf_t['MANY_NEW_LOAN_BANK_3M'] = np.where(df_t['NUM_NEW_LOAN_TAKEN_BANK_3M'] >= 1, 1, 0)\ndf_t['MANY_NEW_LOAN_BANK_6M'] = np.where(df_t['NUM_NEW_LOAN_TAKEN_BANK_6M'] >= 4, 1, 0)\ndf_t['MANY_NEW_LOAN_BANK_9M'] = np.where(df_t['NUM_NEW_LOAN_TAKEN_BANK_9M'] >= 4, 1, 0)\ndf_t['MANY_NEW_LOAN_BANK_12M'] = np.where(df_t['NUM_NEW_LOAN_TAKEN_BANK_12M'] >= 4, 1, 0)\n\ndf_t['MANY_NEW_LOAN_NON_BANK_3M'] = np.where(df_t['NUM_NEW_LOAN_TAKEN_NON_BANK_3M'] >= 1, 1, 0)\ndf_t['MANY_NEW_LOAN_NON_BANK_6M'] = np.where(df_t['NUM_NEW_LOAN_TAKEN_NON_BANK_6M'] >= 4, 1, 0)\ndf_t['MANY_NEW_LOAN_NON_BANK_9M'] = np.where(df_t['NUM_NEW_LOAN_TAKEN_NON_BANK_9M'] >= 4, 1, 0)\ndf_t['MANY_NEW_LOAN_NON_BANK_12M'] = np.where(df_t['NUM_NEW_LOAN_TAKEN_NON_BANK_12M'] >= 4, 1, 0)\n\ndf_t['Theory 1'] = df_t.iloc[:, -12:].sum(axis=1)\ndf['Diff_in_num_new_loan'] = df_t['Theory 1']","metadata":{"id":"A0yjn4GvhoQS","execution":{"iopub.status.busy":"2024-10-24T23:55:53.713230Z","iopub.execute_input":"2024-10-24T23:55:53.713589Z","iopub.status.idle":"2024-10-24T23:55:53.742534Z","shell.execute_reply.started":"2024-10-24T23:55:53.713555Z","shell.execute_reply":"2024-10-24T23:55:53.741611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature 5: Ratio with number of loans","metadata":{}},{"cell_type":"code","source":"cols = ['SHORT_TERM_COUNT', 'MID_TERM_COUNT', 'LONG_TERM_COUNT', \"SHORT_TERM_COUNT_BANK\",\"MID_TERM_COUNT_BANK\",\"LONG_TERM_COUNT_BANK\",\"SHORT_TERM_COUNT_NON_BANK\",\n        \"MID_TERM_COUNT_NON_BANK\",\"LONG_TERM_COUNT_NON_BANK\",\"NUMBER_OF_LOANS_BANK\",\"NUMBER_OF_LOANS_NON_BANK\", \"NUMBER_OF_RELATIONSHIP\",\"NUMBER_OF_RELATIONSHIP_BANK\",\"NUMBER_OF_RELATIONSHIP_NON_BANK\",\n        'NUMBER_OF_LOANS', 'label']\ndf_t = df[cols].copy(deep=True)\n# df_t['NUMBER_OF_LOANS'].fillna(df_t['NUMBER_OF_LOANS'].mean(), inplace=True)\ndf[['RATIO_SHORT', 'RATIO_MID', 'RATIO_LONG']] = \\\ndf_t[['SHORT_TERM_COUNT', 'MID_TERM_COUNT', 'LONG_TERM_COUNT']].div(df_t['NUMBER_OF_LOANS'], axis=0)\ndf[['RATIO_SHORT_BANK', 'RATIO_MID_BANK', 'RATIO_LONG_BANK']] = \\\ndf_t[['SHORT_TERM_COUNT_BANK', 'MID_TERM_COUNT_BANK', 'LONG_TERM_COUNT_BANK']].div(df_t['NUMBER_OF_LOANS_BANK'], axis=0)\ndf[['RATIO_SHORT_NON_BANK', 'RATIO_MID_NON_BANK', 'RATIO_LONG_NON_BANK']] = \\\ndf_t[['SHORT_TERM_COUNT_NON_BANK', 'MID_TERM_COUNT_NON_BANK', 'LONG_TERM_COUNT_NON_BANK']].div(df_t['NUMBER_OF_LOANS_NON_BANK'], axis=0)\ndf[['RATIO_RELATIONSHIP']] = df_t[['NUMBER_OF_RELATIONSHIP']].div(df_t['NUMBER_OF_LOANS'], axis=0)\ndf[['RATIO_RELATIONSHIP_BANK']] = df_t[['NUMBER_OF_RELATIONSHIP_BANK']].div(df_t['NUMBER_OF_LOANS_BANK'], axis=0)\ndf[['RATIO_RELATIONSHIP_NON_BANK']] = df_t[['NUMBER_OF_RELATIONSHIP_NON_BANK']].div(df_t['NUMBER_OF_LOANS_NON_BANK'], axis=0)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:53.743536Z","iopub.execute_input":"2024-10-24T23:55:53.743889Z","iopub.status.idle":"2024-10-24T23:55:53.767528Z","shell.execute_reply.started":"2024-10-24T23:55:53.743862Z","shell.execute_reply":"2024-10-24T23:55:53.766405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Transform OUTSTANDING BALANCE into 10 bins","metadata":{"id":"83dO5LJkrOcu"}},{"cell_type":"code","source":"df_temp = df.copy(deep=True)\ncorr_matrix = df_temp.corr()\nfiltered_corr = df_temp.corr().applymap(lambda x: x if abs(x) > 0.5 and x != 1 else np.nan)\n\nupper_tri = np.triu(np.ones(corr_matrix.shape), k=1).astype(bool)\nhigh_corr_pairs = filtered_corr.where(upper_tri).stack()\n\n# Step 3: Drop one column for each pair of highly correlated features\nto_drop = set()\nfor col1, col2 in high_corr_pairs.index:\n    if col1 not in to_drop and col2 not in to_drop:\n        to_drop.add(col2)  # Keep col1, drop col2\n\n# Step 4: Drop the identified columns\ndf = df_temp.drop(columns=to_drop)\n\nprint(f\"Columns to drop: {to_drop}\")\nprint(f\"Reduced dataframe shape: {df.shape}\")","metadata":{"id":"PI-mEXCVrOAg","execution":{"iopub.status.busy":"2024-10-24T23:55:53.768953Z","iopub.execute_input":"2024-10-24T23:55:53.769290Z","iopub.status.idle":"2024-10-24T23:55:55.844929Z","shell.execute_reply.started":"2024-10-24T23:55:53.769263Z","shell.execute_reply":"2024-10-24T23:55:55.843629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:55.847083Z","iopub.execute_input":"2024-10-24T23:55:55.847493Z","iopub.status.idle":"2024-10-24T23:55:55.880039Z","shell.execute_reply.started":"2024-10-24T23:55:55.847460Z","shell.execute_reply":"2024-10-24T23:55:55.878879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## WOE - IV","metadata":{}},{"cell_type":"code","source":"import traceback\nimport re","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:55.881797Z","iopub.execute_input":"2024-10-24T23:55:55.882173Z","iopub.status.idle":"2024-10-24T23:55:55.886666Z","shell.execute_reply.started":"2024-10-24T23:55:55.882143Z","shell.execute_reply":"2024-10-24T23:55:55.885532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\nMAX_VAL = 1900000\nMIN_VAL = -10\ndef _bin_table(df, colname, n_bins = 10, qcut = None):\n  X = df[[colname, 'label']]\n  X = X.sort_values(colname)\n  coltype = X[colname].dtype\n\n  if coltype in ['float', 'int']:\n    if qcut is None:\n      try:\n        bins, thres = pd.qcut(X[colname], q = n_bins, retbins=True)\n        # Thay thế threshold đầu và cuối của thres\n        thres[0] = MIN_VAL\n        thres[-1] = MAX_VAL\n        bins, thres = pd.cut(X[colname], bins=thres, retbins=True)\n        X['bins'] = bins\n      except:\n        print('n_bins must be lower to bin interval is valid!')\n    else:\n      bins, thres = pd.cut(X[colname], bins=qcut, retbins=True)\n      X['bins'] = bins\n  elif coltype == 'object':\n    X['bins'] = X[colname]\n\n  df_GB = pd.pivot_table(X, \n                index = ['bins'],\n                values = ['label'],\n                columns = ['label'],\n                aggfunc = {\n                    'label':np.size\n                })\n\n  df_Count = pd.pivot_table(X, \n                index = ['bins'],\n                values = ['label'],\n                aggfunc = {\n                    'label': np.size\n                })\n  \n  if coltype in ['float', 'int']:\n    df_Thres = pd.DataFrame({'Thres':thres[1:]}, index=df_GB.index)\n  elif coltype == 'object':\n    df_Thres = pd.DataFrame(index=df_GB.index)\n    thres = None\n  df_Count.columns = ['No_Obs']\n  df_GB.columns = ['#GOOD', '#BAD']\n  df_summary = df_Thres.join(df_Count).join(df_GB)\n  return df_summary, thres","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:55.887933Z","iopub.execute_input":"2024-10-24T23:55:55.888332Z","iopub.status.idle":"2024-10-24T23:55:55.904585Z","shell.execute_reply.started":"2024-10-24T23:55:55.888294Z","shell.execute_reply":"2024-10-24T23:55:55.903426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len((df.columns))-2","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:55.905959Z","iopub.execute_input":"2024-10-24T23:55:55.906747Z","iopub.status.idle":"2024-10-24T23:55:55.919955Z","shell.execute_reply.started":"2024-10-24T23:55:55.906684Z","shell.execute_reply":"2024-10-24T23:55:55.918892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:55.921281Z","iopub.execute_input":"2024-10-24T23:55:55.921691Z","iopub.status.idle":"2024-10-24T23:55:56.530460Z","shell.execute_reply.started":"2024-10-24T23:55:55.921653Z","shell.execute_reply":"2024-10-24T23:55:56.529266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['NUMBER_OF_LOANS_BANK'].max()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:56.532029Z","iopub.execute_input":"2024-10-24T23:55:56.532459Z","iopub.status.idle":"2024-10-24T23:55:56.540398Z","shell.execute_reply.started":"2024-10-24T23:55:56.532421Z","shell.execute_reply":"2024-10-24T23:55:56.539324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for i in df.columns.drop(['label', 'customer_id','SHORT_TERM_COUNT', 'MID_TERM_COUNT', 'LONG_TERM_COUNT', 'SHORT_TERM_COUNT_BANK']):\ndf_summary, thres = _bin_table(df, 'NUMBER_OF_LOANS_BANK', qcut = [df['NUMBER_OF_LOANS_BANK'].min(), 3.84, 4, df['NUMBER_OF_LOANS_BANK'].max()])\ndf_summary['Share'] = df_summary['No_Obs'] / df_summary['No_Obs'].sum()\ndf_summary['Bad Rate'] = df_summary['#BAD'] / df_summary['No_Obs']\ndf_summary['Distribution Good'] = (df_summary['No_Obs'] - df_summary['#BAD']) / (df_summary['No_Obs'].sum() - df_summary['#BAD'].sum())\ndf_summary['Distribution Bad'] = df_summary['#BAD'] / df_summary['#BAD'].sum()\ndf_summary['WoE'] = np.log(df_summary['Distribution Bad'] / df_summary['Distribution Good'])","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:56.541926Z","iopub.execute_input":"2024-10-24T23:55:56.542597Z","iopub.status.idle":"2024-10-24T23:55:56.577604Z","shell.execute_reply.started":"2024-10-24T23:55:56.542547Z","shell.execute_reply":"2024-10-24T23:55:56.576581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_summary","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:56.579127Z","iopub.execute_input":"2024-10-24T23:55:56.579954Z","iopub.status.idle":"2024-10-24T23:55:56.595895Z","shell.execute_reply.started":"2024-10-24T23:55:56.579914Z","shell.execute_reply":"2024-10-24T23:55:56.594605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_summary = df_summary.reset_index()\n\ndf_summary['bins']","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:56.597287Z","iopub.execute_input":"2024-10-24T23:55:56.597616Z","iopub.status.idle":"2024-10-24T23:55:56.609679Z","shell.execute_reply.started":"2024-10-24T23:55:56.597590Z","shell.execute_reply":"2024-10-24T23:55:56.608544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pandas as pd\nimport numpy as np\n\n# Assuming df_summary is your dataframe\nfig, ax1 = plt.subplots(figsize=(8, 6))\n\n# Bar plot for distribution of good and bad\nbars_good = ax1.bar(df_summary.index.astype(str), df_summary['Distribution Good'], label='#GOOD', color=\"#009dd9\", width=0.5)\nbars_bad = ax1.bar(df_summary.index.astype(str), df_summary['Distribution Bad'], bottom=df_summary['Distribution Good'], label='#BAD', color='salmon', width=0.5)\n\nax1.set_ylabel('Bin count distribution')\nax1.set_xlabel('Bins')\nax1.legend()\n\n# Add the secondary axis for bad probability\nax2 = ax1.twinx()\nax2.plot(df_summary.index.astype(str), df_summary['Bad Rate'], color='blue', marker='o', label='Bad Rate')\nax2.set_ylabel('Bad probability', color='blue')\nax2.tick_params(axis='y', labelcolor='blue')\n\n# Annotate the bars with No_Obs\nfor i, bar in enumerate(bars_good):\n    ax1.text(bar.get_x() + bar.get_width() / 2., \n             bar.get_height() + bars_bad[i].get_height(), \n             f\"{df_summary['No_Obs'].iloc[i]:,}\", ha='center', va='bottom')\n\n# Uncomment this section if you want to annotate bad rate on the line\n# for i, (x, y) in enumerate(zip(df_summary.index.astype(str), df_summary['Bad Rate'])):\n#     ax2.annotate(f\"{y*100:.1f}%\", (x, y), textcoords=\"offset points\", xytext=(0, -30), ha='center', color='black')\n\nplt.title('Phân phối của các bin theo WOE của biến NUMBER_OF_LOANS_BANK')\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:56.612499Z","iopub.execute_input":"2024-10-24T23:55:56.612972Z","iopub.status.idle":"2024-10-24T23:55:57.009221Z","shell.execute_reply.started":"2024-10-24T23:55:56.612932Z","shell.execute_reply":"2024-10-24T23:55:57.008104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _WOE(df, colname, n_bins = None, min_obs = 100, qcut = None):\n  # Thống kê bins và lấy ra thres hold ban đầu\n  df_summary, thres = _bin_table(df, colname, n_bins = n_bins, qcut = qcut)\n  # Thay thế giá trị 0 của #BAD trong df_summary bằng 1 để không bị lỗi chia cho 0\n  df_summary['#BAD'] = df_summary['#BAD'].replace({0:1})\n  \n  if qcut is not None:\n    # Lọc bỏ threshold để tạo thành threshold mới mà thỏa mãn số lượng quan sát >= min_obs\n    exclude_ind = np.where(df_summary['No_Obs'] <= min_obs)[0]\n    if exclude_ind.shape[0] > 0:\n      new_thres = np.delete(thres, exclude_ind)\n      print('Auto combine {} bins into {} bins'.format(n_bins, new_thres.shape[0]-1))\n      # Tính toán lại bảng summary\n      df_summary, thres = _bin_table(df, colname, qcut=new_thres)\n  \n  new_thres = thres\n  df_summary['GOOD/BAD'] = df_summary['#GOOD']/df_summary['#BAD']\n  df_summary['%BAD'] = df_summary['#BAD']/df_summary['#BAD'].sum()\n  df_summary['%GOOD'] = df_summary['#GOOD']/df_summary['#GOOD'].sum()\n  df_summary['WOE'] = np.log(df_summary['%GOOD']/df_summary['%BAD'])\n  df_summary['IV'] = (df_summary['%GOOD']-df_summary['%BAD'])*df_summary['WOE']\n  df_summary['COLUMN'] = colname\n  IV = df_summary['IV'].sum()\n  print('Information Value of {} column: {}'.format(colname, IV))\n  return df_summary, IV, new_thres\n\ndf_summary, IV, thres = _WOE(df, 'NUMBER_OF_LOANS_BANK', n_bins = 3, min_obs= 100)\ndf_summary","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:57.010697Z","iopub.execute_input":"2024-10-24T23:55:57.011129Z","iopub.status.idle":"2024-10-24T23:55:57.064246Z","shell.execute_reply.started":"2024-10-24T23:55:57.011089Z","shell.execute_reply":"2024-10-24T23:55:57.063059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _plot(df_summary):\n  colname = list(df_summary['COLUMN'].unique())[0]\n  df_summary['WOE'].plot(linestyle='-', marker='o')\n  plt.title('WOE of {} field'.format(colname))\n  plt.axhline(y=0, color = 'red')\n  plt.xticks(rotation=45)\n  plt.ylabel('WOE')\n  plt.xlabel('Bin group')\n\n_plot(df_summary)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:57.065880Z","iopub.execute_input":"2024-10-24T23:55:57.066288Z","iopub.status.idle":"2024-10-24T23:55:57.334031Z","shell.execute_reply.started":"2024-10-24T23:55:57.066256Z","shell.execute_reply":"2024-10-24T23:55:57.332921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns.drop('customer_id')","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:57.335334Z","iopub.execute_input":"2024-10-24T23:55:57.335682Z","iopub.status.idle":"2024-10-24T23:55:57.343455Z","shell.execute_reply.started":"2024-10-24T23:55:57.335652Z","shell.execute_reply":"2024-10-24T23:55:57.342253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nbins = {\n    'OUTSTANDING_BAL_LOAN_CURRENT': 10,\n\n    'OUTSTANDING_BAL_LOAN_3M_6M': 10,\n   \n    'OUTSTANDING_BAL_CC_3M_6M': 3,\n    \n    'OUTSTANDING_BAL_ALL_9M_12M': 3,\n  \n    'SHORT_TERM_COUNT': 4,\n    'MID_TERM_COUNT': 2,\n    'LONG_TERM_COUNT': 1,\n\n    'LONG_TERM_COUNT_NON_BANK': 1,\n    'NUMBER_OF_LOANS_BANK': 3,\n    'NUMBER_OF_LOANS_NON_BANK': 3,\n    'NUMBER_OF_CREDIT_CARDS': 3,\n    'NUMBER_OF_CREDIT_CARDS_NON_BANK': 1,\n  \n    'NUM_NEW_LOAN_TAKEN_3M': 2,\n\n\n    'INCREASING_BAL_3M_LOAN': 1,\n    'INCREASING_BAL_6M_LOAN': 1,\n\n    'CREDIT_CARD_NUMBER_OF_LATE_PAYMENT': 1,\n    'ENQUIRIES_3M': 2,\n\n    'ENQUIRIES_FROM_NON_BANK_FOR_LOAN_3M': 2,\n    'ENQUIRIES_FROM_NON_BANK_FOR_CC_3M': 1,\n \n    'ENQUIRIES_3M_6M': 1,\n    'ENQUIRIES_6M_9M': 1,\n    'ENQUIRIES_9M_12M': 1,\n \n    'ENQUIRIES_FROM_BANK_6M_9M': 2,\n  \n    'Diff_in_term_count': 3,\n   \n    'RATIO_SHORT': 4,\n    \n    'RATIO_SHORT_BANK': 2,\n    'RATIO_SHORT_NON_BANK': 3,\n\n    'RATIO_MID_NON_BANK': 4,\n    'RATIO_RELATIONSHIP': 6,\n    'RATIO_RELATIONSHIP_BANK': 4,\n    'RATIO_RELATIONSHIP_NON_BANK': 3,\n}\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:57.345169Z","iopub.execute_input":"2024-10-24T23:55:57.345535Z","iopub.status.idle":"2024-10-24T23:55:57.355927Z","shell.execute_reply.started":"2024-10-24T23:55:57.345504Z","shell.execute_reply":"2024-10-24T23:55:57.354653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"WOE_dict=dict()\nfor (col, bins) in nbins.items():\n  df_summary, IV, thres = _WOE(df, colname=col, n_bins=bins)\n  WOE_dict[col] = {'table':df_summary, 'IV':IV}","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:57.357303Z","iopub.execute_input":"2024-10-24T23:55:57.357791Z","iopub.status.idle":"2024-10-24T23:55:58.157462Z","shell.execute_reply.started":"2024-10-24T23:55:57.357750Z","shell.execute_reply":"2024-10-24T23:55:58.156119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = []\nIVs = []\n# cols = df.columns[df.columns.str.startswith('OUTS') | df.columns.str.startswith('INCR')]\n# Loop through all columns except 'label', 'customer_id', and those in 'cols'\nfor col in df.columns.drop(['label', 'customer_id']):\n    columns.append(col)\n    IVs.append(WOE_dict[col]['IV'])\ndf_WOE = pd.DataFrame({'column': columns, 'IV': IVs})\n\ndef _rank_IV(iv):\n  if iv <= 0.02:\n    return 'Useless'\n  elif iv <= 0.1:\n    return 'Weak'\n  elif iv <= 0.3:\n    return 'Medium'\n  elif iv <= 0.5:\n    return 'Strong'\n  else:\n    return 'suspicious'\n\ndf_WOE['rank']=df_WOE['IV'].apply(lambda x: _rank_IV(x))\ndf_WOE.sort_values('IV', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:58.158965Z","iopub.execute_input":"2024-10-24T23:55:58.159283Z","iopub.status.idle":"2024-10-24T23:55:58.180919Z","shell.execute_reply.started":"2024-10-24T23:55:58.159254Z","shell.execute_reply":"2024-10-24T23:55:58.179879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fix 1: Assuming 'column' refers to column names in df_WOE\ndrop_cols = df_WOE[df_WOE['rank'] == 'Useless']['column'].to_list()\n\n# Fix 2: Adding column names to the list properly\n# drop_cols += [\"Diff_in_num_loan\", 'RATIO_LONG_NON_BANK', 'RATIO_MID', 'RATIO_MID_BANK']\n\n# Drop the columns from df_WOE\n# df_WOE.drop(columns=drop_cols, inplace=True)\nfor col in drop_cols:\n    WOE_dict.pop(col, None)  ","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:58.182505Z","iopub.execute_input":"2024-10-24T23:55:58.183001Z","iopub.status.idle":"2024-10-24T23:55:58.192516Z","shell.execute_reply.started":"2024-10-24T23:55:58.182961Z","shell.execute_reply":"2024-10-24T23:55:58.191189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(WOE_dict)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T00:44:07.653462Z","iopub.execute_input":"2024-10-25T00:44:07.654633Z","iopub.status.idle":"2024-10-25T00:44:07.665106Z","shell.execute_reply.started":"2024-10-25T00:44:07.654591Z","shell.execute_reply":"2024-10-25T00:44:07.663772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"WOE_dict.keys()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:58.194210Z","iopub.execute_input":"2024-10-24T23:55:58.194675Z","iopub.status.idle":"2024-10-24T23:55:58.209257Z","shell.execute_reply.started":"2024-10-24T23:55:58.194625Z","shell.execute_reply":"2024-10-24T23:55:58.208032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in WOE_dict.keys():\n  try:\n    key = list(WOE_dict[col]['table']['WOE'].index)\n    woe = list(WOE_dict[col]['table']['WOE'])\n    d = dict(zip(key, woe))\n    col_woe = col+'_WOE'\n    df[col_woe] = df[col].map(d)\n  except:\n    print(col)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:58.216492Z","iopub.execute_input":"2024-10-24T23:55:58.216910Z","iopub.status.idle":"2024-10-24T23:55:58.893333Z","shell.execute_reply.started":"2024-10-24T23:55:58.216877Z","shell.execute_reply":"2024-10-24T23:55:58.892246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We will drop some low (< 0.02) IV columns**","metadata":{}},{"cell_type":"markdown","source":"## Train-Validation-Test Split for 6 Models after Feature Engineering","metadata":{"_uuid":"02706aaa-df7d-41f6-9829-80f81724f17c","_cell_guid":"62838d0f-5374-4565-821e-066aa3fc257d","id":"dHdCP4CVC8OO","trusted":true}},{"cell_type":"code","source":"# Assuming your data is stored in X and y\nX = df.filter(like='_WOE', axis = 1)\nX_train, X_val_test, y_train, y_val_test = train_test_split(X, df[\"label\"], test_size=0.4, random_state=42)\n# Split the validation and test sets into equal proportions\nX_val, X_test, y_val, y_test = train_test_split(X_val_test, y_val_test, test_size=0.5, random_state=42)\nprint(\"Train set size:\", len(X_train))\nprint(\"Validation set size:\", len(X_val))\nprint(\"Test set size:\", len(X_test))","metadata":{"id":"GeIJEpJsvM-T","execution":{"iopub.status.busy":"2024-10-24T23:55:58.894627Z","iopub.execute_input":"2024-10-24T23:55:58.894999Z","iopub.status.idle":"2024-10-24T23:55:58.919924Z","shell.execute_reply.started":"2024-10-24T23:55:58.894969Z","shell.execute_reply":"2024-10-24T23:55:58.918700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train set size:\", y_train.value_counts() / len(y_train))\nprint(\"Validation set size:\", y_val.value_counts() / len(y_val))\nprint(\"Test set size:\", y_test.value_counts() / len(y_test))","metadata":{"id":"lEn7OfBgvfnU","execution":{"iopub.status.busy":"2024-10-24T23:55:58.921208Z","iopub.execute_input":"2024-10-24T23:55:58.921559Z","iopub.status.idle":"2024-10-24T23:55:58.932503Z","shell.execute_reply.started":"2024-10-24T23:55:58.921533Z","shell.execute_reply":"2024-10-24T23:55:58.931089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Standard scaler","metadata":{"id":"fIQFjaTVC8OQ"}},{"cell_type":"code","source":"scaler = StandardScaler()\nX_train = scaler.fit_transform(X_train)\nX_val = scaler.transform(X_val)\nX_test = scaler.transform(X_test)","metadata":{"id":"EPGo7bE1C8OQ","execution":{"iopub.status.busy":"2024-10-24T23:55:58.934200Z","iopub.execute_input":"2024-10-24T23:55:58.935443Z","iopub.status.idle":"2024-10-24T23:55:58.955889Z","shell.execute_reply.started":"2024-10-24T23:55:58.935397Z","shell.execute_reply":"2024-10-24T23:55:58.954883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Selection","metadata":{}},{"cell_type":"markdown","source":"## Model 1: Logistic Regression","metadata":{"id":"55C8rSKDy0XQ"}},{"cell_type":"code","source":"param_grid = {\n    'solver': ['lbfgs'],  # Different solvers to try\n    'C': [ 0.01],  # Regularization strength\n    'penalty': ['l2'],  # Penalty type (l1 can only be used with some solvers)\n    'max_iter': [5000, 4000]  # Iterations for convergence\n    \n}\n\n# Instantiate the logistic regression model\nlr = LogisticRegression(fit_intercept=True)\n\n# Set up the GridSearchCV\ngrid_search = GridSearchCV(estimator=lr,\n                           param_grid=param_grid,\n                           cv=5,  # 5-fold cross-validation\n                           scoring='roc_auc',  # AUC score for evaluation\n                           verbose=1,\n                           n_jobs=-1)\n\n# Fit the model on the training data\ngrid_search.fit(X_train, y_train)\n\n# Get the best model and its parameters\nbest_lr_model = grid_search.best_estimator_\nprint(\"Best parameters:\", grid_search.best_params_)\n\n# Predictions using the best model\ny_pred = best_lr_model.predict(X_val)\ny_pred_proba = best_lr_model.predict_proba(X_val)[:, 1]\n\n# Evaluate the model\naccuracy = accuracy_score(y_val, y_pred)\nauc_score = roc_auc_score(y_val, y_pred_proba)\n\nprint(f\"Accuracy: {accuracy * 100.0}%\")\nprint(f\"AUC Score: {auc_score}\")\nevaluate(y_val,y_pred,y_pred_proba, \"Logistic Regression\")","metadata":{"id":"hvkiZqc_y5Q_","execution":{"iopub.status.busy":"2024-10-24T23:55:58.957350Z","iopub.execute_input":"2024-10-24T23:55:58.957848Z","iopub.status.idle":"2024-10-24T23:55:59.592973Z","shell.execute_reply.started":"2024-10-24T23:55:58.957805Z","shell.execute_reply":"2024-10-24T23:55:59.591890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\ndef _CreditScore(beta, alpha, woe, n = 134, odds = 1/4, pdo = 50, thres_score = 600):\n  factor = pdo/np.log(2)\n  offset = thres_score - factor*np.log(odds)\n  score = (beta*woe+alpha/n)*factor+offset/n\n  return score\n\n_CreditScore(beta = 0.5, alpha = -1, woe = 0.15, n = 134)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:59.594501Z","iopub.execute_input":"2024-10-24T23:55:59.594948Z","iopub.status.idle":"2024-10-24T23:55:59.604971Z","shell.execute_reply.started":"2024-10-24T23:55:59.594909Z","shell.execute_reply":"2024-10-24T23:55:59.603798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a list of columns to drop\ncolumns_to_drop = ['label', 'customer_id'] + drop_cols\n\n# Drop the unwanted columns and create the dictionary\nbetas_dict = dict(zip(df.columns.drop(columns_to_drop), best_lr_model.coef_[0]))\nalpha = best_lr_model.intercept_[0]\ndf.columns.drop(columns_to_drop)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:59.606687Z","iopub.execute_input":"2024-10-24T23:55:59.607135Z","iopub.status.idle":"2024-10-24T23:55:59.622281Z","shell.execute_reply.started":"2024-10-24T23:55:59.607093Z","shell.execute_reply":"2024-10-24T23:55:59.620969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns_to_drop.remove('label')\ncolumns_to_drop.remove('customer_id')","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:59.623655Z","iopub.execute_input":"2024-10-24T23:55:59.624175Z","iopub.status.idle":"2024-10-24T23:55:59.631376Z","shell.execute_reply.started":"2024-10-24T23:55:59.624133Z","shell.execute_reply":"2024-10-24T23:55:59.630276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in columns_to_drop:\n    columns.remove(i)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:59.632688Z","iopub.execute_input":"2024-10-24T23:55:59.633064Z","iopub.status.idle":"2024-10-24T23:55:59.642975Z","shell.execute_reply.started":"2024-10-24T23:55:59.633032Z","shell.execute_reply":"2024-10-24T23:55:59.641813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = []\nfeatures = []\nwoes = []\nbetas = []\nscores = []\n\nfor col in columns:\n    for feature, woe in WOE_dict[col]['table']['WOE'].to_frame().iterrows():\n        cols.append(col)\n        # Add feature\n        feature = str(feature)\n        features.append(feature)\n        # Add woe\n        woe = woe.values[0]\n        woes.append(woe)\n        # Add beta (adjusted to match the keys in betas_dict)\n        beta = betas_dict[col]\n        betas.append(beta)\n        # Add score\n        score = _CreditScore(beta=beta, alpha=alpha, woe=woe, n=134)\n        scores.append(score)\n\n# Create DataFrame with the collected data\ndf_WOE = pd.DataFrame({'Columns': cols, 'Features': features, 'WOE': woes, 'Betas': betas, 'Scores': scores})\ndf_WOE.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:59.646182Z","iopub.execute_input":"2024-10-24T23:55:59.646914Z","iopub.status.idle":"2024-10-24T23:55:59.678311Z","shell.execute_reply.started":"2024-10-24T23:55:59.646870Z","shell.execute_reply":"2024-10-24T23:55:59.677109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_WOE.to_csv('WOE.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T00:29:11.923640Z","iopub.execute_input":"2024-10-25T00:29:11.924095Z","iopub.status.idle":"2024-10-25T00:29:11.932437Z","shell.execute_reply.started":"2024-10-25T00:29:11.924062Z","shell.execute_reply":"2024-10-25T00:29:11.931313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Giả sử một hồ sơ ngẫu nhiên có các thông số như sau\ntest_obs = df[columns].iloc[0:2, :]\ntest_obs","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:55:59.680008Z","iopub.execute_input":"2024-10-24T23:55:59.680456Z","iopub.status.idle":"2024-10-24T23:55:59.707660Z","shell.execute_reply.started":"2024-10-24T23:55:59.680415Z","shell.execute_reply":"2024-10-24T23:55:59.706466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _search_score(obs, col):\n  feature = [str(inter) for inter in list(WOE_dict[col]['table'].index) if obs[col].values[0] in inter][0]\n  score = df_WOE[(df_WOE['Columns'] == col) & (df_WOE['Features'] == feature)]['Scores'].values[0]\n  return score\n\n# Tính điểm cho trường 'LOAN' của bộ hồ sơ test\nscore = _search_score(test_obs, 'NUMBER_OF_LOANS_BANK')\nscore","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:57:10.631009Z","iopub.execute_input":"2024-10-24T23:57:10.631468Z","iopub.status.idle":"2024-10-24T23:57:10.643231Z","shell.execute_reply.started":"2024-10-24T23:57:10.631436Z","shell.execute_reply":"2024-10-24T23:57:10.642072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _total_score(obs, columns = columns):\n  scores = dict()\n  for col in columns:\n    scores[col] = _search_score(obs, col)\n  total_score = sum(scores.values())\n  return scores, total_score\n\nscores, total_score = _total_score(test_obs)\nprint('score for each fields: \\n', scores)\nprint('final total score: ', total_score)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:57:13.419780Z","iopub.execute_input":"2024-10-24T23:57:13.420865Z","iopub.status.idle":"2024-10-24T23:57:13.455346Z","shell.execute_reply.started":"2024-10-24T23:57:13.420825Z","shell.execute_reply":"2024-10-24T23:57:13.453960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_scores = []\nfor i in np.arange(df[columns].shape[0]):\n  obs = df[columns].iloc[i:(i+1), :]\n  _, score = _total_score(obs)\n  total_scores.append(score)\ndf['Score'] = total_scores","metadata":{"execution":{"iopub.status.busy":"2024-10-24T23:57:16.281898Z","iopub.execute_input":"2024-10-24T23:57:16.282288Z","iopub.status.idle":"2024-10-25T00:03:44.477710Z","shell.execute_reply.started":"2024-10-24T23:57:16.282258Z","shell.execute_reply":"2024-10-25T00:03:44.476554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df[\"Score\"].max())\nprint(df[\"Score\"].min())\nprint(df[\"Score\"].mean())\nprint(df[\"Score\"].std())\nprint(df.Score.quantile([0.25, 0.5, 0.75, 0.9, 0.95, 0.99, 1]))\ndf.Score.to_csv(\"Score.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-10-25T00:04:52.167366Z","iopub.execute_input":"2024-10-25T00:04:52.167861Z","iopub.status.idle":"2024-10-25T00:04:52.223948Z","shell.execute_reply.started":"2024-10-25T00:04:52.167823Z","shell.execute_reply":"2024-10-25T00:04:52.222653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 3: XGBoost","metadata":{"_uuid":"382b8437-8f09-4f67-b981-cc7751fd4d02","_cell_guid":"751a1b0f-abf8-4f83-b041-83da3e477ce9","id":"HKFj1EqcC8OR","trusted":true}},{"cell_type":"code","source":"params = {\n#     'subsample': 0.7222222222222222, 'reg_lambda': 0.6666666666666666, 'reg_alpha': 0.5555555555555556, 'n_estimators': 200, 'min_child_weight': 7, 'max_depth': 5, 'learning_rate': 0.019999999999999997, 'gamma': 0.05555555555555555, 'colsample_bytree': 0.5\n          }\n\n# Initialize XGBoost Classifier with selected parameters\nxgb_model = xgb.XGBClassifier(**params)\n\n# Perform cross-validation\ncv_scores = cross_val_score(xgb_model, X_train, y_train, cv=5, scoring='roc_auc')\n\n# Train the model on the entire training set\nxgb_model.fit(X_train, y_train)\n\n# Predictions on the test set\nxgb_y_pred = xgb_model.predict(X_val)\ny_pred_proba = xgb_model.predict_proba(X_val)[:, 1]\n\n# Calculate accuracy and AUC Score on the test set\nxgb_accuracy = accuracy_score(y_val, xgb_y_pred)\nxgb_auc_score = roc_auc_score(y_val, y_pred_proba)\n\nprint(f\"Cross-Validation ROC-AUC Scores: {cv_scores}\")\nprint(f\"Average Cross-Validation ROC-AUC Score: {np.mean(cv_scores)}\")\nprint(f\"Test Set Accuracy: {xgb_accuracy * 100.0}%\")\nprint(f\"Test Set AUC Score: {xgb_auc_score}\")\n\n# # Display feature importance\n# feature_importance = xgb_model.feature_importances_\n# sorted_idx = np.argsort(feature_importance)[::-1]\n\n# print(\"\\nTop 10 Feature Importance:\")\n# for i in range(10):\n#     print(f\"{train_balanced[sorted_idx[i]]}: {feature_importance[sorted_idx[i]]}\")\nevaluate(y_val,xgb_y_pred,y_pred_proba, \"XGBoost\")","metadata":{"_uuid":"94177d6f-c9dc-474f-8ded-8210624f8434","_cell_guid":"a295632d-9338-48bc-86cc-d8ec9bee3484","jupyter":{"outputs_hidden":false},"id":"tCp4R4F9C8OR","collapsed":false,"execution":{"iopub.status.busy":"2024-10-25T00:11:58.830678Z","iopub.execute_input":"2024-10-25T00:11:58.831117Z","iopub.status.idle":"2024-10-25T00:12:01.021589Z","shell.execute_reply.started":"2024-10-25T00:11:58.831083Z","shell.execute_reply":"2024-10-25T00:12:01.020432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model - LightGBM","metadata":{"_uuid":"986082d9-8475-424c-8964-1923b9fa22c8","_cell_guid":"5035b192-dce5-411a-b0ab-7b30271839a5","id":"7bAyIYL1C8OS","trusted":true}},{"cell_type":"code","source":"params = {\n#     'bagging_fraction': 0.7559247410755839, 'bagging_freq': 2, 'feature_fraction': 0.6785986360743963, 'lambda_l1': 0.003084514923441306, 'learning_rate': 0.041501214401114275, 'max_depth': 15, 'min_data_in_leaf': 22, 'n_estimators': 136, 'num_leaves': 31\n          }\n\n# Initialize LightGBM Classifier with selected parameters\nlgb_model = lgb.LGBMClassifier(**params)\n\n# Perform cross-validation\ncv_scores = cross_val_score(lgb_model,X_train, y_train, cv=5, scoring='roc_auc')\n\n# Train the model on the entire training set\nlgb_model.fit(X_train, y_train)\n\n# Predictions on the test set\nlgb_y_pred = lgb_model.predict(X_val)\ny_pred_proba = lgb_model.predict_proba(X_val)[:, 1]\n\n# Calculate accuracy and AUC Score on the test set\nlgb_accuracy = accuracy_score(y_val, lgb_y_pred)\nlgb_auc_score = roc_auc_score(y_val, y_pred_proba)\n\nprint(f\"Cross-Validation ROC-AUC Scores: {cv_scores}\")\nprint(f\"Average Cross-Validation ROC-AUC Score: {np.mean(cv_scores)}\")\nprint(f\"Test Set Accuracy: {lgb_accuracy * 100.0}%\")\nprint(f\"Test Set AUC Score: {lgb_auc_score}\")\nevaluate(y_val,lgb_y_pred,y_pred_proba, \"Light GBM\")","metadata":{"_uuid":"25afb764-b956-45d3-97f4-9e1229686608","_cell_guid":"669043b3-889f-42d0-8fae-409e9908bf47","jupyter":{"outputs_hidden":false},"id":"ONEDXYLQC8OS","collapsed":false,"execution":{"iopub.status.busy":"2024-10-25T00:12:01.023715Z","iopub.execute_input":"2024-10-25T00:12:01.024107Z","iopub.status.idle":"2024-10-25T00:12:03.203204Z","shell.execute_reply.started":"2024-10-25T00:12:01.024077Z","shell.execute_reply":"2024-10-25T00:12:03.201859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lgb.plot_importance(lgb_model, importance_type=\"gain\", max_num_features=15 ,figsize=(20,15), title=\"LightGBM Feature Importance (Gain)\", grid=False, color = \"#009dd9\")\n# plt.show()\n","metadata":{"_uuid":"27b03d74-a54c-45e9-a0c5-b0e9a2231a52","_cell_guid":"55d17844-b063-48b6-926c-7e67bb8908dd","jupyter":{"outputs_hidden":false},"id":"2zijfftIC8OT","collapsed":false,"execution":{"iopub.status.busy":"2024-10-25T00:12:03.204659Z","iopub.execute_input":"2024-10-25T00:12:03.205070Z","iopub.status.idle":"2024-10-25T00:12:03.209970Z","shell.execute_reply.started":"2024-10-25T00:12:03.205036Z","shell.execute_reply":"2024-10-25T00:12:03.208801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 6: Random Forest","metadata":{"_uuid":"50f59850-c645-4ea8-a736-2feda46cb80c","_cell_guid":"7edc1ec3-5c38-4f24-a96e-36b7734502dc","id":"dQMYhPz5C8OX","trusted":true}},{"cell_type":"code","source":"# Initialize Random Forest Classifier\nrf_model = RandomForestClassifier(\n#     n_estimators=300,\n#     n_jobs=-1,\n#     max_depth=16,\n#     min_samples_split=10,\n#     min_samples_leaf=7,\n#     max_features='sqrt',\n#     bootstrap=True,\n#     criterion='gini',\n#     random_state=42,\n    \n)\n\n# Perform cross-validation\ncv_scores = cross_val_score(rf_model, X_train, y_train, cv=5, scoring='roc_auc')\n\n# Train the model on the entire training set\nrf_model.fit(X_train, y_train)\n\n# Predictions on the test set\nrf_y_pred = rf_model.predict(X_val)\ny_pred_proba = rf_model.predict_proba(X_val)[:, 1]\n\n# Calculate accuracy and AUC Score on the test set\nrf_accuracy = accuracy_score(y_val, rf_y_pred)\nrf_auc_score = roc_auc_score(y_val, y_pred_proba)\n\nprint(f\"Cross-Validation ROC-AUC Scores: {cv_scores}\")\nprint(f\"Average Cross-Validation ROC-AUC Score: {np.mean(cv_scores)}\")\nprint(f\"Test Set Accuracy: {rf_accuracy * 100.0}%\")\nprint(f\"Test Set AUC Score: {rf_auc_score}\")\nevaluate(y_val,rf_y_pred,y_pred_proba, \"Random Forest\")\n","metadata":{"_uuid":"94cc51dc-b29c-48ce-8b1f-6d2f6afb4464","_cell_guid":"a3777945-2215-4c76-ba88-c78d392d901f","jupyter":{"outputs_hidden":false},"id":"VngpXgQjC8OX","collapsed":false,"execution":{"iopub.status.busy":"2024-10-25T00:12:03.214636Z","iopub.execute_input":"2024-10-25T00:12:03.215085Z","iopub.status.idle":"2024-10-25T00:12:11.013298Z","shell.execute_reply.started":"2024-10-25T00:12:03.215052Z","shell.execute_reply":"2024-10-25T00:12:11.012157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.metrics import accuracy_score, roc_auc_score\n\n# Define the Decision Tree Classifier with similar hyperparameters (adjust as needed)\ndt_model = DecisionTreeClassifier(\n    max_depth=16,\n    min_samples_split=10,\n    min_samples_leaf=7,\n    random_state=42\n)\n\n# Perform cross-validation with ROC-AUC as the scoring metric\ncv_scores = cross_val_score(dt_model, X_train, y_train, cv=5, scoring='roc_auc')\n\n# Train the decision tree model on the entire training set\ndt_model.fit(X_train, y_train)\n\n# Make predictions on the test set\ndt_y_pred = dt_model.predict(X_val)\ndt_pred_proba = dt_model.predict_proba(X_val)[:, 1]\n\n# Calculate accuracy and AUC Score on the test set\ndt_accuracy = accuracy_score(y_val, dt_y_pred)\ndt_auc_score = roc_auc_score(y_val, dt_pred_proba)\n\nprint(f\"Cross-Validation ROC-AUC Scores: {cv_scores}\")\nprint(f\"Average Cross-Validation ROC-AUC Score: {np.mean(cv_scores)}\")\nprint(f\"Test Set Accuracy: {dt_accuracy * 100.0}%\")\nprint(f\"Test Set AUC Score: {dt_auc_score}\")\n\nevaluate(y_val,dt_y_pred,dt_pred_proba, \"Random Forest\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T00:12:11.014803Z","iopub.execute_input":"2024-10-25T00:12:11.015132Z","iopub.status.idle":"2024-10-25T00:12:11.546982Z","shell.execute_reply.started":"2024-10-25T00:12:11.015105Z","shell.execute_reply":"2024-10-25T00:12:11.545790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Final Results","metadata":{"id":"tNh5_C57C8OY"}},{"cell_type":"code","source":"final_results = pd.DataFrame(results)\nfinal_results","metadata":{"id":"-bNUxtVXC8OY","execution":{"iopub.status.busy":"2024-10-25T00:12:11.548556Z","iopub.execute_input":"2024-10-25T00:12:11.549009Z","iopub.status.idle":"2024-10-25T00:12:11.565092Z","shell.execute_reply.started":"2024-10-25T00:12:11.548977Z","shell.execute_reply":"2024-10-25T00:12:11.563615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fit Best Model on Test Set: Gradient Boosting","metadata":{}},{"cell_type":"code","source":"\n# Predictions on the test set\nlr_y_pred = best_lr_model.predict(X_test)\ny_pred_proba = best_lr_model.predict_proba(X_test)[:, 1]\n\n# Calculate accuracy and AUC Score on the test set\nlr_accuracy = accuracy_score(y_test, lr_y_pred)\nlr_auc_score = roc_auc_score(y_test, y_pred_proba)\n\nprint(f\"Cross-Validation ROC-AUC Scores: {cv_scores}\")\nprint(f\"Average Cross-Validation ROC-AUC Score: {np.mean(cv_scores)}\")\nprint(f\"Test Set Accuracy: {lr_accuracy * 100.0}%\")\nprint(f\"Test Set AUC Score: {lr_auc_score}\")\nevaluate(y_test,lr_y_pred,y_pred_proba, \"Logistic Regression on Test Set\")","metadata":{"_uuid":"a421381e-be68-4a6a-bf59-6adce1aecbd1","_cell_guid":"0f7265e8-3773-433f-86bf-f2cb9ccf0cf0","jupyter":{"outputs_hidden":false},"id":"TjHfl_CkC8OY","collapsed":false,"execution":{"iopub.status.busy":"2024-10-25T00:09:20.972609Z","iopub.execute_input":"2024-10-25T00:09:20.973066Z","iopub.status.idle":"2024-10-25T00:09:21.315406Z","shell.execute_reply.started":"2024-10-25T00:09:20.973027Z","shell.execute_reply":"2024-10-25T00:09:21.314162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results.clear()","metadata":{"execution":{"iopub.status.busy":"2024-10-25T00:11:51.280867Z","iopub.execute_input":"2024-10-25T00:11:51.282319Z","iopub.status.idle":"2024-10-25T00:11:51.287854Z","shell.execute_reply.started":"2024-10-25T00:11:51.282276Z","shell.execute_reply":"2024-10-25T00:11:51.286164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}