{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Hey there! 😊 If you found this notebook helpful, dropping an upvote would really make my day.","metadata":{}},{"cell_type":"markdown","source":"## 📌 1. Import Libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport shap\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import KFold\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom scipy.stats import pearsonr","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 📂 2. Load Dataset","metadata":{}},{"cell_type":"markdown","source":"The following technique/code is adapted from Kaggle Master Mahdi Ravaghi’s work.\nYou can find more details in his original [notebook](https://www.kaggle.com/code/ravaghi/drw-crypto-market-prediction-ensemble).","metadata":{}},{"cell_type":"code","source":"class CFG:\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n    ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reduce_mem_usage(dataframe, dataset):    \n    print('Reducing memory usage for:', dataset)\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    \n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                dataframe[col] = dataframe[col].astype(np.int8)\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                dataframe[col] = dataframe[col].astype(np.int16)\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                dataframe[col] = dataframe[col].astype(np.int64)\n        else:\n            if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                dataframe[col] = dataframe[col].astype(np.float16)\n            elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                dataframe[col] = dataframe[col].astype(np.float32)\n            else:\n                dataframe[col] = dataframe[col].astype(np.float64)\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    print('--- Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n    print('--- Memory usage after: {:.2f} MB'.format(final_mem_usage))\n    print('--- Decreased memory usage by {:.1f}%\\n'.format(100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage))\n\n    return dataframe","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_parquet(CFG.train_path).reset_index(drop=True)\ntest = pd.read_parquet(CFG.test_path).reset_index(drop=True)\nsample=pd.read_csv(\"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 📊 3. Selected Features Based on SHAP Values","metadata":{}},{"cell_type":"code","source":"selected_features = [\n\n    \"X863\", \"X856\", \"X344\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \"X674\",\n    \"X415\", \"X345\", \"X137\", \"X855\", \"X174\", \"X302\", \"X178\", \"X532\", \"X168\", \"X612\",\n    \"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"\n]\n\n\n\ntrain= train[selected_features + [\"label\"]]\ntest= test[selected_features]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = reduce_mem_usage(train, \"train\")\ntest = reduce_mem_usage(test, \"test\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Train=\",train.shape)\nprint(\"Test=\",test.shape)\nprint(\"Sample=\",sample.shape)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RMV = [\"label\"]\nFEATURES = [c for c in train.columns if not c in RMV]\nprint(f\"There are {len(FEATURES)} FEATURES: {FEATURES}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 🚀 4. Train XGBoost with KFold Cross-Validation","metadata":{}},{"cell_type":"code","source":"FOLDS = 5\nkf = KFold(n_splits=FOLDS, shuffle=True, random_state=42)\n\noof_xgb = np.zeros(len(train))\nxgb_preds = np.zeros(len(test))\n\nxgb_params = {\n    \"colsample_bylevel\": 0.4778015829774066,\n    \"colsample_bynode\": 0.362764358742407,\n    \"colsample_bytree\": 0.7107423488010493,\n    \"gamma\": 1.7094857725240398,\n    \"learning_rate\": 0.02213323588455387,\n    \"max_depth\": 20,\n    \"max_leaves\": 12,\n    \"min_child_weight\": 16,\n    \"n_estimators\": 1667,\n    \"n_jobs\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 39.352415706891264,\n    \"reg_lambda\": 75.44843704068275,\n    \"subsample\": 0.06566669853471274,\n    \"verbosity\": 0,\n    \"tree_method\": \"gpu_hist\"\n    \n}\nfor i, (train_idx, valid_idx) in enumerate(kf.split(train)):\n    print(\"#\" * 25)\n    print(f\"### Fold {i + 1}\")\n    print(\"#\" * 25)\n\n    X_train = train.iloc[train_idx][FEATURES]\n    y_train = train.iloc[train_idx][\"label\"]\n    X_valid = train.iloc[valid_idx][FEATURES]\n    y_valid = train.iloc[valid_idx][\"label\"]\n    X_test = test[FEATURES]\n\n    xgb_model = XGBRegressor(**xgb_params)\n\n    xgb_model.fit(\n        X_train, y_train,\n        eval_set=[(X_valid, y_valid)],\n        verbose=200\n    )\n\n    oof_xgb[valid_idx] = xgb_model.predict(X_valid)\n    xgb_preds += xgb_model.predict(X_test)\n\npearson_score = pearsonr(train[\"label\"], oof_xgb)[0]\nprint(\"Final Pearson Correlation = \", pearson_score)\n\nxgb_preds /= FOLDS","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 🚀 5. Train LightGBM with KFold Cross-Validation","metadata":{}},{"cell_type":"code","source":"FOLDS = 5\nkf = KFold(n_splits=FOLDS, shuffle=True, random_state=42)\n\nlgbm_params = {\n    \"boosting_type\": \"gbdt\",\n    \"colsample_bytree\": 0.5625888953382505,\n    \"learning_rate\": 0.029312951475451557,\n    \"min_child_samples\": 63,\n    \"min_child_weight\": 0.11456572852335424,\n    \"n_estimators\": 126,\n    \"num_leaves\": 37,\n    \"random_state\": 42,\n    \"reg_alpha\": 85.2476527854083,\n    \"reg_lambda\": 99.38305361388907,\n    \"subsample\": 0.450669817684892,\n    \"verbose\": -1,\n    \"device\": \"gpu\",                  \n    \"gpu_platform_id\": 0,             \n    \"gpu_device_id\": 0,\n    \"n_jobs\": -1                      \n}\n\noof_lgb = np.zeros(len(train))\nlgb_preds = np.zeros(len(test))\n\nfor i, (train_idx, valid_idx) in enumerate(kf.split(train)):\n    print(\"#\" * 25)\n    print(f\"### Fold {i + 1}\")\n    print(\"#\" * 25)\n\n    X_train = train.iloc[train_idx][FEATURES]\n    y_train = train.iloc[train_idx][\"label\"]\n    X_valid = train.iloc[valid_idx][FEATURES]\n    y_valid = train.iloc[valid_idx][\"label\"]\n    X_test = test[FEATURES]\n\n    lgb_model = LGBMRegressor(**lgbm_params)\n\n    lgb_model.fit(\n        X_train, y_train,\n        eval_set=[(X_valid, y_valid)]\n    )\n\n    oof_lgb[valid_idx] = lgb_model.predict(X_valid)\n    lgb_preds += lgb_model.predict(X_test)\n\npearson_score = pearsonr(train[\"label\"], oof_lgb)[0]\nprint(\"Final Pearson Correlation = \", pearson_score)\n\nlgb_preds /= FOLDS","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## ⚖️ 6. Weighted Ensemble of LightGBM and XGBoost Predictions","metadata":{}},{"cell_type":"code","source":"ensemble_preds=(0.55 * lgb_preds + 0.45 * xgb_preds)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 📤 7. Create Submission File","metadata":{}},{"cell_type":"code","source":"sample[\"prediction\"] = ensemble_preds\nsample.to_csv(\"submission.csv\", index=False)\nsample.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}