{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score\nimport lightgbm as lgb\nimport glob\nimport os\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-30T15:05:47.785778Z","iopub.execute_input":"2024-06-30T15:05:47.786630Z","iopub.status.idle":"2024-06-30T15:05:47.797449Z","shell.execute_reply.started":"2024-06-30T15:05:47.786600Z","shell.execute_reply":"2024-06-30T15:05:47.796527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Base path for the data","metadata":{}},{"cell_type":"code","source":"base_path = '/kaggle/input/home-credit-credit-risk-model-stability/csv_files'","metadata":{"execution":{"iopub.status.busy":"2024-06-30T15:05:57.609960Z","iopub.execute_input":"2024-06-30T15:05:57.610349Z","iopub.status.idle":"2024-06-30T15:05:57.615432Z","shell.execute_reply.started":"2024-06-30T15:05:57.610315Z","shell.execute_reply":"2024-06-30T15:05:57.614464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Function to load and concatenate data files","metadata":{}},{"cell_type":"code","source":"def load_data(base_path, file_pattern):\n    files = glob.glob(os.path.join(base_path, file_pattern))\n    data_frames = [pd.read_csv(file) for file in files]\n    return pd.concat(data_frames, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-30T15:05:58.442627Z","iopub.execute_input":"2024-06-30T15:05:58.443665Z","iopub.status.idle":"2024-06-30T15:05:58.448519Z","shell.execute_reply.started":"2024-06-30T15:05:58.443626Z","shell.execute_reply":"2024-06-30T15:05:58.447551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load base data\ntrain_base = pd.read_csv(os.path.join(base_path, '/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_base.csv'))\ntest_base = pd.read_csv(os.path.join(base_path, '/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_base.csv'))","metadata":{"execution":{"iopub.status.busy":"2024-06-30T15:05:59.140603Z","iopub.execute_input":"2024-06-30T15:05:59.140980Z","iopub.status.idle":"2024-06-30T15:06:00.199088Z","shell.execute_reply.started":"2024-06-30T15:05:59.140951Z","shell.execute_reply":"2024-06-30T15:06:00.198005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load static data\ntrain_static_0 = load_data(base_path, '/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_static_0_0.csv')\ntest_static_0 = load_data(base_path, '/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_static_0_0.csv')\ntrain_static_cb_0 = pd.read_csv(os.path.join(base_path, '/kaggle/input/home-credit-credit-risk-model-stability/csv_files/train/train_static_cb_0.csv'))\ntest_static_cb_0 = pd.read_csv(os.path.join(base_path, '/kaggle/input/home-credit-credit-risk-model-stability/csv_files/test/test_static_cb_0.csv'))","metadata":{"execution":{"iopub.status.busy":"2024-06-30T15:06:00.454426Z","iopub.execute_input":"2024-06-30T15:06:00.455379Z","iopub.status.idle":"2024-06-30T15:06:39.481148Z","shell.execute_reply.started":"2024-06-30T15:06:00.455340Z","shell.execute_reply":"2024-06-30T15:06:39.479806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Function to merge data","metadata":{}},{"cell_type":"code","source":"def merge_data(base, static_0, static_cb_0):\n    base = base.merge(static_0, on='case_id', how='left')\n    base = base.merge(static_cb_0, on='case_id', how='left')\n    return base","metadata":{"execution":{"iopub.status.busy":"2024-06-30T15:06:39.483090Z","iopub.execute_input":"2024-06-30T15:06:39.483430Z","iopub.status.idle":"2024-06-30T15:06:39.488573Z","shell.execute_reply.started":"2024-06-30T15:06:39.483400Z","shell.execute_reply":"2024-06-30T15:06:39.487518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Merge data","metadata":{}},{"cell_type":"code","source":"train = merge_data(train_base, train_static_0, train_static_cb_0)\ntest = merge_data(test_base, test_static_0, test_static_cb_0)","metadata":{"execution":{"iopub.status.busy":"2024-06-30T15:06:39.490012Z","iopub.execute_input":"2024-06-30T15:06:39.490699Z","iopub.status.idle":"2024-06-30T15:06:50.631758Z","shell.execute_reply.started":"2024-06-30T15:06:39.490662Z","shell.execute_reply":"2024-06-30T15:06:50.630762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocess data","metadata":{}},{"cell_type":"code","source":"def preprocess_data(data):\n    data.fillna(-999, inplace=True)  # Handle missing values\n    for col in data.select_dtypes(include=['object']).columns:\n        data[col], _ = data[col].factorize()  # Encode categorical features\n    return data","metadata":{"execution":{"iopub.status.busy":"2024-06-30T15:06:54.748693Z","iopub.execute_input":"2024-06-30T15:06:54.749177Z","iopub.status.idle":"2024-06-30T15:06:54.755742Z","shell.execute_reply.started":"2024-06-30T15:06:54.749138Z","shell.execute_reply":"2024-06-30T15:06:54.754546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = preprocess_data(train)\ntest = preprocess_data(test)","metadata":{"execution":{"iopub.status.busy":"2024-06-30T15:06:56.665341Z","iopub.execute_input":"2024-06-30T15:06:56.666264Z","iopub.status.idle":"2024-06-30T15:07:22.643985Z","shell.execute_reply.started":"2024-06-30T15:06:56.666226Z","shell.execute_reply":"2024-06-30T15:07:22.642952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split data into features and target","metadata":{}},{"cell_type":"code","source":"X = train.drop(['case_id', 'target'], axis=1)\ny = train['target']","metadata":{"execution":{"iopub.status.busy":"2024-06-30T15:07:26.259586Z","iopub.execute_input":"2024-06-30T15:07:26.260408Z","iopub.status.idle":"2024-06-30T15:07:27.424933Z","shell.execute_reply.started":"2024-06-30T15:07:26.260373Z","shell.execute_reply":"2024-06-30T15:07:27.423848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Subsample the training data","metadata":{}},{"cell_type":"code","source":"X_subset = X.sample(n=10000, random_state=42)\ny_subset = y.loc[X_subset.index]","metadata":{"execution":{"iopub.status.busy":"2024-06-30T15:07:29.752538Z","iopub.execute_input":"2024-06-30T15:07:29.753300Z","iopub.status.idle":"2024-06-30T15:07:29.848434Z","shell.execute_reply.started":"2024-06-30T15:07:29.753269Z","shell.execute_reply":"2024-06-30T15:07:29.847455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_train_subset = lgb.Dataset(X_subset, y_subset)","metadata":{"execution":{"iopub.status.busy":"2024-06-30T15:07:33.656405Z","iopub.execute_input":"2024-06-30T15:07:33.657344Z","iopub.status.idle":"2024-06-30T15:07:33.661655Z","shell.execute_reply.started":"2024-06-30T15:07:33.657309Z","shell.execute_reply":"2024-06-30T15:07:33.660540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LightGBM parameters","metadata":{}},{"cell_type":"code","source":"params = {\n    'objective': 'binary',\n    'metric': 'auc',\n    'boosting_type': 'gbdt',\n    'num_leaves': 31,\n    'learning_rate': 0.05,\n    'feature_fraction': 0.9\n}","metadata":{"execution":{"iopub.status.busy":"2024-06-30T15:07:37.838932Z","iopub.execute_input":"2024-06-30T15:07:37.839341Z","iopub.status.idle":"2024-06-30T15:07:37.844661Z","shell.execute_reply.started":"2024-06-30T15:07:37.839301Z","shell.execute_reply":"2024-06-30T15:07:37.843378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Perform cross-validation with early stopping using callback","metadata":{}},{"cell_type":"code","source":"# Cross-validation without early stopping\ncv_results = lgb.cv(\n    params,\n    lgb_train_subset,\n    num_boost_round=500,\n    nfold=3,\n    metrics=['auc'],  # Track AUC for reporting\n    seed=42\n)","metadata":{"execution":{"iopub.status.busy":"2024-06-30T15:07:47.127306Z","iopub.execute_input":"2024-06-30T15:07:47.128198Z","iopub.status.idle":"2024-06-30T15:08:07.122511Z","shell.execute_reply.started":"2024-06-30T15:07:47.128161Z","shell.execute_reply":"2024-06-30T15:08:07.121512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print the keys of cv_results\nprint(cv_results.keys())","metadata":{"execution":{"iopub.status.busy":"2024-06-30T15:13:39.676628Z","iopub.execute_input":"2024-06-30T15:13:39.677488Z","iopub.status.idle":"2024-06-30T15:13:39.682567Z","shell.execute_reply.started":"2024-06-30T15:13:39.677424Z","shell.execute_reply":"2024-06-30T15:13:39.681572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get the best number of boosting rounds","metadata":{}},{"cell_type":"code","source":"best_num_boost_round = np.argmax(cv_results['valid auc-mean']) + 1","metadata":{"execution":{"iopub.status.busy":"2024-06-30T15:15:38.636399Z","iopub.execute_input":"2024-06-30T15:15:38.636863Z","iopub.status.idle":"2024-06-30T15:15:38.642654Z","shell.execute_reply.started":"2024-06-30T15:15:38.636830Z","shell.execute_reply":"2024-06-30T15:15:38.641372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train the final model","metadata":{}},{"cell_type":"code","source":"model = lgb.train(params, lgb_train_subset, num_boost_round=best_num_boost_round)","metadata":{"execution":{"iopub.status.busy":"2024-06-30T15:15:56.894656Z","iopub.execute_input":"2024-06-30T15:15:56.895046Z","iopub.status.idle":"2024-06-30T15:15:58.009533Z","shell.execute_reply.started":"2024-06-30T15:15:56.895010Z","shell.execute_reply":"2024-06-30T15:15:58.008519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make predictions on the test set","metadata":{}},{"cell_type":"code","source":"X_test = test.drop(['case_id'], axis=1)\ntest_predictions = model.predict(X_test, num_iteration=model.best_iteration)","metadata":{"execution":{"iopub.status.busy":"2024-06-30T15:16:04.384024Z","iopub.execute_input":"2024-06-30T15:16:04.384404Z","iopub.status.idle":"2024-06-30T15:16:04.397975Z","shell.execute_reply.started":"2024-06-30T15:16:04.384375Z","shell.execute_reply":"2024-06-30T15:16:04.396767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create submission file","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame({'case_id': test['case_id'], 'score': test_predictions})\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Submission file created successfully!\")","metadata":{"execution":{"iopub.status.busy":"2024-06-30T15:16:06.319536Z","iopub.execute_input":"2024-06-30T15:16:06.319927Z","iopub.status.idle":"2024-06-30T15:16:06.330527Z","shell.execute_reply.started":"2024-06-30T15:16:06.319899Z","shell.execute_reply":"2024-06-30T15:16:06.329330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}