{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install koolbox scikit-learn==1.5.2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T20:07:31.535515Z","iopub.execute_input":"2025-07-18T20:07:31.536016Z","iopub.status.idle":"2025-07-18T20:07:35.864747Z","shell.execute_reply.started":"2025-07-18T20:07:31.535976Z","shell.execute_reply":"2025-07-18T20:07:35.863478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom sklearn.linear_model import Ridge\nfrom lightgbm import LGBMRegressor\nfrom scipy.stats import pearsonr as pr\nfrom xgboost import XGBRegressor\nfrom sklearn.base import clone\nfrom koolbox import Trainer\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas as pd\nimport numpy as np\nimport warnings\nimport optuna\nimport joblib\nimport glob\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-18T20:07:35.866945Z","iopub.execute_input":"2025-07-18T20:07:35.867325Z","iopub.status.idle":"2025-07-18T20:07:35.874433Z","shell.execute_reply.started":"2025-07-18T20:07:35.867292Z","shell.execute_reply":"2025-07-18T20:07:35.873479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CFG:\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\n    target = \"label\"\n    n_folds = 5\n    seed = 42\n\n    run_optuna = True\n    n_optuna_trials = 500","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T20:07:35.875675Z","iopub.execute_input":"2025-07-18T20:07:35.876401Z","iopub.status.idle":"2025-07-18T20:07:35.895797Z","shell.execute_reply.started":"2025-07-18T20:07:35.876369Z","shell.execute_reply":"2025-07-18T20:07:35.894537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reduce_mem_usage(dataframe, dataset):    \n    print('Reducing memory usage for:', dataset)\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    \n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                dataframe[col] = dataframe[col].astype(np.int8)\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                dataframe[col] = dataframe[col].astype(np.int16)\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                dataframe[col] = dataframe[col].astype(np.int64)\n        else:\n            if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                dataframe[col] = dataframe[col].astype(np.float16)\n            elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                dataframe[col] = dataframe[col].astype(np.float32)\n            else:\n                dataframe[col] = dataframe[col].astype(np.float64)\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    print('--- Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n    print('--- Memory usage after: {:.2f} MB'.format(final_mem_usage))\n    print('--- Decreased memory usage by {:.1f}%\\n'.format(100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage))\n\n    return dataframe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T20:07:35.898042Z","iopub.execute_input":"2025-07-18T20:07:35.898707Z","iopub.status.idle":"2025-07-18T20:07:35.913862Z","shell.execute_reply.started":"2025-07-18T20:07:35.898679Z","shell.execute_reply":"2025-07-18T20:07:35.912870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_parquet(CFG.train_path).reset_index(drop=True)\ntest = pd.read_parquet(CFG.test_path).reset_index(drop=True)\n\ntrain = reduce_mem_usage(train, \"train\")\ntest = reduce_mem_usage(test, \"test\")\n\nX = train.drop(CFG.target, axis=1)\ny = train[CFG.target]\nX_test = test.drop(CFG.target, axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T20:07:35.915008Z","iopub.execute_input":"2025-07-18T20:07:35.915808Z","iopub.status.idle":"2025-07-18T20:08:22.289006Z","shell.execute_reply.started":"2025-07-18T20:07:35.915783Z","shell.execute_reply":"2025-07-18T20:08:22.288231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = [\n    \"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"\n]\n\nX = X[features]\nX_test = X_test[features]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T20:08:22.289932Z","iopub.execute_input":"2025-07-18T20:08:22.290247Z","iopub.status.idle":"2025-07-18T20:08:22.462039Z","shell.execute_reply.started":"2025-07-18T20:08:22.290211Z","shell.execute_reply":"2025-07-18T20:08:22.461175Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def pearsonr(y_true, y_pred):\n    return pr(y_true, y_pred)[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T20:08:22.462908Z","iopub.execute_input":"2025-07-18T20:08:22.463204Z","iopub.status.idle":"2025-07-18T20:08:22.467838Z","shell.execute_reply.started":"2025-07-18T20:08:22.463177Z","shell.execute_reply":"2025-07-18T20:08:22.466990Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_params = {\n    \"boosting_type\": \"gbdt\",\n    \"colsample_bytree\": 0.5625888953382505,\n    \"learning_rate\": 0.029312951475451557,\n    \"min_child_samples\": 63,\n    \"min_child_weight\": 0.11456572852335424,\n    \"n_estimators\": 126,\n    \"n_jobs\": -1,\n    \"num_leaves\": 37,\n    \"random_state\": 42,\n    \"reg_alpha\": 85.2476527854083,\n    \"reg_lambda\": 99.38305361388907,\n    \"subsample\": 0.450669817684892,\n    \"verbose\": -1\n}\n\nlgbm_goss_params = {\n    \"boosting_type\": \"goss\",\n    \"colsample_bytree\": 0.34695458228489784,\n    \"learning_rate\": 0.031023014900595287,\n    \"min_child_samples\": 30,\n    \"min_child_weight\": 0.4727729225033618,\n    \"n_estimators\": 220,\n    \"n_jobs\": -1,\n    \"num_leaves\": 58,\n    \"random_state\": 42,\n    \"reg_alpha\": 38.665994901468224,\n    \"reg_lambda\": 92.76991677464294,\n    \"subsample\": 0.4810891284493255,\n    \"verbose\": -1\n}\n\nxgb_params = {\n    \"colsample_bylevel\": 0.4778015829774066,\n    \"colsample_bynode\": 0.362764358742407,\n    \"colsample_bytree\": 0.7107423488010493,\n    \"gamma\": 1.7094857725240398,\n    \"learning_rate\": 0.02213323588455387,\n    \"max_depth\": 20,\n    \"max_leaves\": 12,\n    \"min_child_weight\": 16,\n    \"n_estimators\": 1667,\n    \"n_jobs\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 39.352415706891264,\n    \"reg_lambda\": 75.44843704068275,\n    \"subsample\": 0.06566669853471274,\n    \"verbosity\": 0\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T20:08:22.468894Z","iopub.execute_input":"2025-07-18T20:08:22.469203Z","iopub.status.idle":"2025-07-18T20:08:22.483601Z","shell.execute_reply.started":"2025-07-18T20:08:22.469165Z","shell.execute_reply":"2025-07-18T20:08:22.482805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fold_scores = {}\noverall_scores = {}\n\noof_preds = {}\ntest_preds = {}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T20:08:22.484796Z","iopub.execute_input":"2025-07-18T20:08:22.485539Z","iopub.status.idle":"2025-07-18T20:08:22.497717Z","shell.execute_reply.started":"2025-07-18T20:08:22.485512Z","shell.execute_reply":"2025-07-18T20:08:22.496894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_trainer = Trainer(\n    LGBMRegressor(**lgbm_params),\n    cv=KFold(n_splits=5, shuffle=False),\n    metric=pearsonr,\n    task=\"regression\",\n    metric_precision=8\n)\n\nlgbm_trainer.fit(X, y)\n\nfold_scores[\"LightGBM (gbdt)\"] = lgbm_trainer.fold_scores\noverall_scores[\"LightGBM (gbdt)\"] = [pearsonr(lgbm_trainer.oof_preds, y)]\noof_preds[\"LightGBM (gbdt)\"] = lgbm_trainer.oof_preds\ntest_preds[\"LightGBM (gbdt)\"] = lgbm_trainer.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T20:08:22.500284Z","iopub.execute_input":"2025-07-18T20:08:22.500540Z","iopub.status.idle":"2025-07-18T20:08:44.212962Z","shell.execute_reply.started":"2025-07-18T20:08:22.500520Z","shell.execute_reply":"2025-07-18T20:08:44.212172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb_trainer = Trainer(\n    XGBRegressor(**xgb_params),\n    cv=KFold(n_splits=5, shuffle=False),\n    metric=pearsonr,\n    task=\"regression\",\n    metric_precision=8\n)\n\nxgb_trainer.fit(X, y)\n\nfold_scores[\"XGBoost\"] = xgb_trainer.fold_scores\noverall_scores[\"XGBoost\"] = [pearsonr(xgb_trainer.oof_preds, y)]\noof_preds[\"XGBoost\"] = xgb_trainer.oof_preds\ntest_preds[\"XGBoost\"] = xgb_trainer.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T20:08:44.213932Z","iopub.execute_input":"2025-07-18T20:08:44.214371Z","iopub.status.idle":"2025-07-18T20:09:49.552159Z","shell.execute_reply.started":"2025-07-18T20:08:44.214346Z","shell.execute_reply":"2025-07-18T20:09:49.551321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_weights(weights, title):\n    sorted_indices = np.argsort(weights[0])[::-1]\n    sorted_coeffs = np.array(weights[0])[sorted_indices]\n    sorted_model_names = np.array(list(oof_preds.keys()))[sorted_indices]\n\n    plt.figure(figsize=(10, weights.shape[1] * 0.5))\n    ax = sns.barplot(x=sorted_coeffs, y=sorted_model_names, palette=\"RdYlGn_r\")\n\n    for i, (value, name) in enumerate(zip(sorted_coeffs, sorted_model_names)):\n        if value >= 0:\n            ax.text(value, i, f\"{value:.3f}\", va=\"center\", ha=\"left\", color=\"black\")\n        else:\n            ax.text(value, i, f\"{value:.3f}\", va=\"center\", ha=\"right\", color=\"black\")\n\n    xlim = ax.get_xlim()\n    ax.set_xlim(xlim[0] - 0.1 * abs(xlim[0]), xlim[1] + 0.1 * abs(xlim[1]))\n\n    plt.title(title)\n    plt.xlabel(\"\")\n    plt.ylabel(\"\")\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T20:09:49.553227Z","iopub.execute_input":"2025-07-18T20:09:49.553767Z","iopub.status.idle":"2025-07-18T20:09:49.561880Z","shell.execute_reply.started":"2025-07-18T20:09:49.553736Z","shell.execute_reply":"2025-07-18T20:09:49.560933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = pd.DataFrame(oof_preds)\nX_test = pd.DataFrame(test_preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T20:09:49.562817Z","iopub.execute_input":"2025-07-18T20:09:49.563099Z","iopub.status.idle":"2025-07-18T20:09:49.584899Z","shell.execute_reply.started":"2025-07-18T20:09:49.563073Z","shell.execute_reply":"2025-07-18T20:09:49.583929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"joblib.dump(X, \"oof_preds.pkl\")\njoblib.dump(X_test, \"test_preds.pkl\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T20:09:49.585912Z","iopub.execute_input":"2025-07-18T20:09:49.586253Z","iopub.status.idle":"2025-07-18T20:09:49.615463Z","shell.execute_reply.started":"2025-07-18T20:09:49.586220Z","shell.execute_reply":"2025-07-18T20:09:49.614439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def objective(trial):    \n    params = {\n        \"random_state\": CFG.seed,\n        \"alpha\": trial.suggest_float(\"alpha\", 0, 1),\n        \"tol\": trial.suggest_float(\"tol\", 1e-6, 1e-2),\n        \"fit_intercept\": trial.suggest_categorical(\"fit_intercept\", [True, False]),\n        \"positive\": trial.suggest_categorical(\"positive\", [True, False])\n    }\n\n    trainer = Trainer(\n        Ridge(**params),\n        cv=KFold(n_splits=5, shuffle=False),\n        metric=pearsonr,\n        task=\"regression\",\n        verbose=False\n    )\n    trainer.fit(X, y)\n    \n    return pearsonr(trainer.oof_preds, y)\n\nif CFG.run_optuna:\n    sampler = optuna.samplers.TPESampler(seed=CFG.seed, multivariate=True, n_startup_trials=CFG.n_optuna_trials // 10)\n    study = optuna.create_study(direction=\"maximize\", sampler=sampler)\n    study.optimize(objective, n_trials=CFG.n_optuna_trials, n_jobs=-1, catch=(ValueError,))\n    best_params = study.best_params\n\n    ridge_params = {\n        \"random_state\": CFG.seed,\n        \"alpha\": best_params[\"alpha\"],\n        \"tol\": best_params[\"tol\"],\n        \"fit_intercept\": best_params[\"fit_intercept\"],\n        \"positive\": best_params[\"positive\"]\n    }\nelse:\n    ridge_params = {\n        \"random_state\": CFG.seed\n    }\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T20:09:49.616472Z","iopub.execute_input":"2025-07-18T20:09:49.616791Z","iopub.status.idle":"2025-07-18T20:17:29.889925Z","shell.execute_reply.started":"2025-07-18T20:09:49.616761Z","shell.execute_reply":"2025-07-18T20:17:29.889122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ridge_trainer = Trainer(\n    Ridge(**ridge_params),\n    cv=KFold(n_splits=5, shuffle=False),\n    metric=pearsonr,\n    task=\"regression\",\n    metric_precision=6\n)\n\nridge_trainer.fit(X, y)\n\nfold_scores[\"Ridge (ensemble)\"] = ridge_trainer.fold_scores\noverall_scores[\"Ridge (ensemble)\"] = [pearsonr(ridge_trainer.oof_preds, y)]\nridge_test_preds = ridge_trainer.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T20:17:29.890875Z","iopub.execute_input":"2025-07-18T20:17:29.891124Z","iopub.status.idle":"2025-07-18T20:17:31.002110Z","shell.execute_reply.started":"2025-07-18T20:17:29.891103Z","shell.execute_reply":"2025-07-18T20:17:31.001112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ridge_coeffs = np.zeros((1, X.shape[1]))\nfor m in ridge_trainer.estimators:\n    ridge_coeffs += m.coef_\nridge_coeffs = ridge_coeffs / len(ridge_trainer.estimators)\n\nplot_weights(ridge_coeffs, \"Ridge Coefficients\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T20:17:31.003095Z","iopub.execute_input":"2025-07-18T20:17:31.003795Z","iopub.status.idle":"2025-07-18T20:17:31.154623Z","shell.execute_reply.started":"2025-07-18T20:17:31.003766Z","shell.execute_reply":"2025-07-18T20:17:31.153632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.read_csv(CFG.sample_sub_path)\nsub[\"prediction\"] = ridge_test_preds\nsub.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-18T20:27:21.230509Z","iopub.execute_input":"2025-07-18T20:27:21.230852Z","iopub.status.idle":"2025-07-18T20:27:22.700604Z","shell.execute_reply.started":"2025-07-18T20:27:21.230819Z","shell.execute_reply":"2025-07-18T20:27:22.699678Z"}},"outputs":[],"execution_count":null}]}