{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7602123,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-07T04:38:35.348414Z","iopub.execute_input":"2024-02-07T04:38:35.349047Z","iopub.status.idle":"2024-02-07T04:38:35.384568Z","shell.execute_reply.started":"2024-02-07T04:38:35.349016Z","shell.execute_reply":"2024-02-07T04:38:35.383628Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport xgboost as xgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:38:35.389040Z","iopub.execute_input":"2024-02-07T04:38:35.389807Z","iopub.status.idle":"2024-02-07T04:38:35.395024Z","shell.execute_reply.started":"2024-02-07T04:38:35.389778Z","shell.execute_reply":"2024-02-07T04:38:35.394000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Loading the data","metadata":{}},{"cell_type":"code","source":"train_base = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_base.csv\")\ntrain_static_cb_0 = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_static_cb_0.csv\")\ntrain_applprev_1_0 = pd.read_csv(\"/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_applprev_1_0.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:41:00.892167Z","iopub.execute_input":"2024-02-07T04:41:00.892779Z","iopub.status.idle":"2024-02-07T04:41:29.686078Z","shell.execute_reply.started":"2024-02-07T04:41:00.892748Z","shell.execute_reply":"2024-02-07T04:41:29.684687Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Merging the tables on case_id","metadata":{}},{"cell_type":"code","source":"train_merged = pd.merge(train_base, train_static_cb_0, on=\"case_id\", how=\"left\")\ntrain_merged = pd.merge(train_merged, train_applprev_1_0, on=\"case_id\", how=\"left\")","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:42:01.968149Z","iopub.execute_input":"2024-02-07T04:42:01.968535Z","iopub.status.idle":"2024-02-07T04:42:11.293585Z","shell.execute_reply.started":"2024-02-07T04:42:01.968488Z","shell.execute_reply":"2024-02-07T04:42:11.292561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Droping unnecessary columns like case_id, date_decision, etc.","metadata":{}},{"cell_type":"code","source":"train_merged.drop([\"case_id\", \"date_decision\"], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:42:22.828334Z","iopub.execute_input":"2024-02-07T04:42:22.828801Z","iopub.status.idle":"2024-02-07T04:42:25.668933Z","shell.execute_reply.started":"2024-02-07T04:42:22.828770Z","shell.execute_reply":"2024-02-07T04:42:25.668292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Handling missing values if any","metadata":{}},{"cell_type":"code","source":"train_merged.fillna(-999, inplace=True)  # Filling missing values with a placeholder","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:42:53.680644Z","iopub.execute_input":"2024-02-07T04:42:53.680990Z","iopub.status.idle":"2024-02-07T04:43:09.657308Z","shell.execute_reply.started":"2024-02-07T04:42:53.680963Z","shell.execute_reply":"2024-02-07T04:43:09.655746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Split data into features and target","metadata":{}},{"cell_type":"code","source":"X = train_merged.drop(\"target\", axis=1)\ny = train_merged[\"target\"]","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:43:15.455584Z","iopub.execute_input":"2024-02-07T04:43:15.455949Z","iopub.status.idle":"2024-02-07T04:43:17.546661Z","shell.execute_reply.started":"2024-02-07T04:43:15.455919Z","shell.execute_reply":"2024-02-07T04:43:17.545695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Spliting data into train and validation sets\n","metadata":{}},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:43:35.692177Z","iopub.execute_input":"2024-02-07T04:43:35.692551Z","iopub.status.idle":"2024-02-07T04:44:03.497116Z","shell.execute_reply.started":"2024-02-07T04:43:35.692523Z","shell.execute_reply":"2024-02-07T04:44:03.496160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Defining XGBoost parameters","metadata":{}},{"cell_type":"code","source":"params = {\n    \"objective\": \"binary:logistic\",\n    \"eval_metric\": \"auc\",\n    \"eta\": 0.1,\n    \"max_depth\": 6,\n    \"subsample\": 0.8,\n    \"colsample_bytree\": 0.8,\n    \"seed\": 42\n}","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:45:02.321605Z","iopub.execute_input":"2024-02-07T04:45:02.321938Z","iopub.status.idle":"2024-02-07T04:45:02.328472Z","shell.execute_reply.started":"2024-02-07T04:45:02.321913Z","shell.execute_reply":"2024-02-07T04:45:02.326966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Converting data to DMatrix format","metadata":{}},{"cell_type":"code","source":"print(\"X_train shape:\", X_train.shape)\nprint(\"X_valid shape:\", X_valid.shape)\nprint(\"y_train shape:\", y_train.shape)\nprint(\"y_valid shape:\", y_valid.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:47:33.138674Z","iopub.execute_input":"2024-02-07T04:47:33.139031Z","iopub.status.idle":"2024-02-07T04:47:33.144051Z","shell.execute_reply.started":"2024-02-07T04:47:33.139004Z","shell.execute_reply":"2024-02-07T04:47:33.143109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"X_train data types:\\n\", X_train.dtypes)\nprint(\"X_valid data types:\\n\", X_valid.dtypes)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:47:45.491055Z","iopub.execute_input":"2024-02-07T04:47:45.491473Z","iopub.status.idle":"2024-02-07T04:47:45.498990Z","shell.execute_reply.started":"2024-02-07T04:47:45.491447Z","shell.execute_reply":"2024-02-07T04:47:45.498145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Missing values in X_train:\\n\", X_train.isnull().sum())\nprint(\"Missing values in X_valid:\\n\", X_valid.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:47:56.982554Z","iopub.execute_input":"2024-02-07T04:47:56.983396Z","iopub.status.idle":"2024-02-07T04:48:04.440776Z","shell.execute_reply.started":"2024-02-07T04:47:56.983360Z","shell.execute_reply":"2024-02-07T04:48:04.439370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# Initialize LabelEncoder\nlabel_encoder = LabelEncoder()\n\n# Loop through each column in X_train and X_valid\nfor col in X_train.columns:\n    if X_train[col].dtype == \"object\":\n        # Convert column values to string type\n        X_train[col] = X_train[col].astype(str)\n        X_valid[col] = X_valid[col].astype(str)\n        # Fit LabelEncoder on both X_train and X_valid\n        label_encoder.fit(pd.concat([X_train[col], X_valid[col]], axis=0))\n        # Transform both X_train and X_valid\n        X_train[col] = label_encoder.transform(X_train[col])\n        X_valid[col] = label_encoder.transform(X_valid[col])","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:51:51.211999Z","iopub.execute_input":"2024-02-07T04:51:51.212369Z","iopub.status.idle":"2024-02-07T04:52:51.358030Z","shell.execute_reply.started":"2024-02-07T04:51:51.212342Z","shell.execute_reply":"2024-02-07T04:52:51.357257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert data to DMatrix format after encoding\ndtrain = xgb.DMatrix(X_train, label=y_train)\ndvalid = xgb.DMatrix(X_valid, label=y_valid)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:53:05.785657Z","iopub.execute_input":"2024-02-07T04:53:05.786005Z","iopub.status.idle":"2024-02-07T04:53:13.553129Z","shell.execute_reply.started":"2024-02-07T04:53:05.785978Z","shell.execute_reply":"2024-02-07T04:53:13.552369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Training","metadata":{}},{"cell_type":"code","source":"model = xgb.train(params, dtrain, num_boost_round=1000, evals=[(dvalid, \"Validation\")], early_stopping_rounds=50, verbose_eval=100)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:53:22.062631Z","iopub.execute_input":"2024-02-07T04:53:22.063935Z","iopub.status.idle":"2024-02-07T05:09:56.786359Z","shell.execute_reply.started":"2024-02-07T04:53:22.063896Z","shell.execute_reply":"2024-02-07T05:09:56.785545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prediction on validation set","metadata":{}},{"cell_type":"code","source":"y_pred = model.predict(dvalid)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T05:11:03.202790Z","iopub.execute_input":"2024-02-07T05:11:03.203398Z","iopub.status.idle":"2024-02-07T05:11:11.881247Z","shell.execute_reply.started":"2024-02-07T05:11:03.203369Z","shell.execute_reply":"2024-02-07T05:11:11.880552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Calculating AUC score","metadata":{}},{"cell_type":"code","source":"auc_score = roc_auc_score(y_valid, y_pred)\nprint(\"Validation AUC:\", auc_score)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T05:11:19.127740Z","iopub.execute_input":"2024-02-07T05:11:19.128464Z","iopub.status.idle":"2024-02-07T05:11:19.410384Z","shell.execute_reply.started":"2024-02-07T05:11:19.128440Z","shell.execute_reply":"2024-02-07T05:11:19.408727Z"},"trusted":true},"execution_count":null,"outputs":[]}]}