{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"},{"sourceId":241449882,"sourceType":"kernelVersion"},{"sourceId":241643814,"sourceType":"kernelVersion"}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **FOREWORD**\n\nThis is a baseline kernel for the [DRW - Crypto Market Prediction](https://www.kaggle.com/competitions/drw-crypto-market-prediction/code?competitionId=96164). This is a regression problem with pearson correlation eval-metric. This is a forecasting competition, with a public test set for syntax checks and a private test set for later periods. <br>\n\nFeature engineering is a key differentiator in such competitions. In this starter, we delve into the data, drop meaningless features and train simple models to start the process. <br>\n\nIn this version, we use the model parameters from the reference kernel [here](https://www.kaggle.com/code/ravaghi/drw-crypto-market-prediction-ensemble) and refit the model on the full dataset. \n","metadata":{}},{"cell_type":"markdown","source":"# **IMPORTS**","metadata":{}},{"cell_type":"code","source":"!uv pip install -q --system -r /kaggle/input/drw2025-public-imports-v1/req_kaggle.txt\n\nexec( open(f\"/kaggle/input/drw2025-public-imports-v1/myimports.py\", \"r\").read() )\nexec( open(f\"/kaggle/input/drw2025-public-imports-v1/myutils.py\", \"r\").read() )\nexec( open(f\"/kaggle/input/drw2025-public-imports-v1/training.py\", \"r\").read() )\n\nprint()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T13:58:28.346809Z","iopub.execute_input":"2025-05-24T13:58:28.347139Z","iopub.status.idle":"2025-05-24T13:59:11.774961Z","shell.execute_reply.started":"2025-05-24T13:58:28.347112Z","shell.execute_reply":"2025-05-24T13:59:11.773842Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **CONFIGURATION**","metadata":{}},{"cell_type":"code","source":"%%time\n\nclass CFG:\n    \"\"\"\n    Configuration class for parameters and CV strategy for tuning and training\n    Some parameters may be unused here as this is a general configuration class\n    \"\"\";\n\n    # Data preparation:-\n    version_nb         = 1\n    model_id           = \"V1_2\"\n    model_label        = \"ML\"\n    test_req           = False\n    test_iter          = 200\n    gpu_switch         = \"ON\" if torch.cuda.is_available() else \"OFF\"\n    state              = 42\n    target             = f'label'\n    grouper            = f\"\"\n    tgt_mapper         = {}\n    ip_path            = f\"/kaggle/input/drw-crypto-market-prediction\"\n    op_path            = f\"/kaggle/working\"\n    orig_path          = f\"\"\n    data_path          = f\"\"\n    dtl_preproc_req    = True\n    ftre_plots_req     = True\n    ftre_imp_req       = True\n    nb_orig            = 0\n    orig_all_folds     = False\n\n    # Model Training:-\n    pstprcs_oof        = False\n    pstprcs_train      = False\n    pstprcs_test       = False\n    ML                 = True\n    test_preds_req     = True\n    n_splits           = 5\n    n_repeats          = 1\n    nbrnd_erly_stp     = 0\n    mdlcv_mthd         = 'KF'\n    metric_obj         = 'maximize'\n\n    # Global variables for plotting:-\n    grid_specs = {'visible'  : True,\n                  'which'    : 'both',\n                  'linestyle': '--',\n                  'color'    : 'lightgrey',\n                  'linewidth': 0.75\n                 }\n\n    title_specs = {'fontsize'   : 9,\n                   'fontweight' : 'bold',\n                   'color'      : '#992600',\n                  }\n\ncv_selector = \\\n{\n \"RKF\"   : RKF(n_splits   = CFG.n_splits, n_repeats= CFG.n_repeats, random_state= CFG.state),\n \"RSKF\"  : RSKF(n_splits  = CFG.n_splits, n_repeats= CFG.n_repeats, random_state= CFG.state),\n \"SKF\"   : SKF(n_splits   = CFG.n_splits, shuffle = False, ),\n \"KF\"    : KFold(n_splits = CFG.n_splits, shuffle = False, ),\n \"GKF\"   : GKF(n_splits   = CFG.n_splits)\n}\n\ncollect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T13:59:26.892977Z","iopub.execute_input":"2025-05-24T13:59:26.893351Z","iopub.status.idle":"2025-05-24T13:59:27.127898Z","shell.execute_reply.started":"2025-05-24T13:59:26.893322Z","shell.execute_reply":"2025-05-24T13:59:27.126381Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **KEY CONFIG PARAMETERS**\n\n|Configuration parameter| Explanation| Data type| Sample values |  \n| ---------------------- | ------------------------------- | --------------------- | --------------- |\n| version_nb    | Version Number | int | 1 | \n| model_id      | Model ID    | string | V1_1 | \n| model_label   | Model Label | string | ML | \n| test_req      | Test Required| bool | True / False | \n| test_iter     | Test case iterations for models | int | 50 |\n| gpu_switch      | Do we need GPU support | bool | True / False |\n| state           | Random state | int | 42 |\n| target          | Target column | str |  |\n| grouper         | CV grouper column | str |  |\n| ip_path, op_path | Data paths  | str | |\n| pstprcs_* | Do we need post-processing  | bool |True / False |\n| ML| Do we need machine learning models  | bool |True / False |\n| test_preds_req| Do we need test set predictions (training in inference kernel)  | bool |True / False |\n| n_splits/ n_repeats | N-splits and repeats for CV scheme | int | 3/5/10|\n| nbrnd_erly_stp | Early stopping rounds | int | 40|\n| mdlcv_mthd | Model CV method | str | RSKF|\n| ensemble_req | Do we need ensemble | bool | True / False |\n| metric_obj   | Metric direction | str | minimize/ maximize |","metadata":{}},{"cell_type":"markdown","source":"# **PREPROCESSING**","metadata":{}},{"cell_type":"code","source":"%%time \n\ndrop_cols = \\\n[\n    'X697', 'X698', 'X699', 'X700', 'X701', 'X702', 'X703', 'X704', 'X705', 'X706', \n    'X707', 'X708', 'X709', 'X710', 'X711', 'X712', 'X713', 'X714', 'X715', 'X716',\n    'X717', 'X864', 'X867', 'X869', 'X870', 'X871', 'X872', 'X104', 'X110', 'X116',\n    'X122', 'X128', 'X134', 'X140', 'X146', 'X152', 'X158', 'X164', 'X170', 'X176',\n    'X182', 'X351', 'X357', 'X363', 'X369', 'X375', 'X381', 'X387', 'X393', 'X399',\n    'X405', 'X411', 'X417', 'X423', 'X429',\n    'timestamp'\n]\n\nXtrain = \\\n(\n    pl.scan_parquet(\n        os.path.join(CFG.ip_path, \"train.parquet\")\n    ).\n    drop( pl.col(drop_cols), strict = False).\n    collect(engine = \"streaming\").\n    select(pl.all().shrink_dtype()).\n    to_pandas()\n)\n\nXtest = \\\n(\n    pl.scan_parquet(\n        os.path.join(CFG.ip_path, \"test.parquet\")\n    ).\n    drop( pl.col(drop_cols), strict = False).\n    collect(engine = \"streaming\").\n    select(pl.all().shrink_dtype()).\n    to_pandas()\n)[Xtrain.columns]\n\nsub_fl = pd.read_csv(os.path.join(CFG.ip_path, \"sample_submission.csv\"), index_col = [\"ID\"])\n\nXtrain[\"Source\"], Xtest[\"Source\"] = (\"Competition\", \"Competition\")\nytrain = Xtrain[CFG.target]\ndel Xtrain[CFG.target]\ndel Xtest[CFG.target]\n\n# Initializing the CV scheme\nygrp = np.zeros(len(ytrain))\ncv   = cv_selector[CFG.mdlcv_mthd]\n\nfor fold_nb, (_, dev_idx) in enumerate(cv.split(Xtrain, ytrain)) :\n    ygrp[dev_idx] = fold_nb\nygrp = pd.Series(ygrp).astype(np.uint8)\n\nPrintColor(\n    f\"\\n---> Shapes = {Xtrain.shape} {Xtest.shape} {ytrain.shape} {ygrp.shape}\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T13:59:31.251023Z","iopub.execute_input":"2025-05-24T13:59:31.251388Z","iopub.status.idle":"2025-05-24T14:00:05.482984Z","shell.execute_reply.started":"2025-05-24T13:59:31.251361Z","shell.execute_reply":"2025-05-24T14:00:05.481545Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **MODEL TRAINING**","metadata":{}},{"cell_type":"markdown","source":"## **OFFLINE CV**","metadata":{}},{"cell_type":"code","source":"%%time \n\nMdl_Master = \\\n{ \n f\"XGB1R\"   : XGBR(**{  \"objective\"              : \"reg:squarederror\",\n                        \"device\"                 : \"cuda\" if CFG.gpu_switch == \"ON\" else \"cpu\", \n                        \"n_estimators\"           : 679 if CFG.test_req == False else 25,\n                        \"learning_rate\"          : 0.096,\n                        \"colsample_bylevel\"      : 0.46,\n                        \"colsample_bynode\"       : 0.60,\n                        \"colsample_bytree\"       : 0.115,\n                        \"gamma\"                  : 1.039,\n                        \"max_depth\"              : 40,\n                        \"max_leaves\"             : 19,\n                        \"min_child_weight\"       : 76,\n                        \"n_jobs\"                 : -1,\n                        \"random_state\"           : CFG.state,\n                        \"reg_alpha\"              : 65.41,\n                        \"reg_lambda\"             : 19.907,\n                        \"subsample\"              : 0.0144,\n                        \"verbosity\"              : 0,\n                    }\n                  ),\n    \n f'LGBM1R'  : LGBMR(**{ \"objective\"          : \"regression_l2\",\n                        'device'             : \"gpu\" if CFG.gpu_switch == \"ON\" else \"cpu\",\n                        \"n_estimators\"       : 441 if CFG.test_req == False else 25,\n                        \"learning_rate\"      : 0.0145,\n                        \"colsample_bytree\"   : 0.524,               \n                        \"min_child_samples\"  : 47,\n                        \"min_child_weight\"   : 0.193,\n                        \"n_jobs\"             : -1,\n                        \"num_leaves\"         : 65,\n                        \"random_state\"       : CFG.state,\n                        \"reg_alpha\"          : 76.69,\n                        \"reg_lambda\"         : 78.57,\n                        \"subsample\"          : 0.35,\n                        \"verbosity\"          : -1\n                      }\n                   ),\n\n f'LGBM2R'  : LGBMR(**{ \"objective\"              : \"regression_l2\",\n                        \"data_sample_strategy\"   : \"goss\",\n                        \"n_estimators\"           : 268 if CFG.test_req == False else 25,\n                        'device'                 : \"gpu\" if CFG.gpu_switch == \"ON\" else \"cpu\",\n                        \"learning_rate\"          : 0.0136,\n                        \"colsample_bytree\"       : 0.32,\n                        \"min_child_samples\"      : 47,\n                        \"min_child_weight\"       : 0.652,\n                        \"n_jobs\"                 : -1,\n                        \"num_leaves\"             : 25,\n                        \"random_state\"           : CFG.state,\n                        \"reg_alpha\"              : 24.43,\n                        \"reg_lambda\"             : 39.82,\n                        \"subsample\"              : 0.21,\n                        \"verbosity\"              : -1\n                     }\n                   ),\n}\n\n# Initializing model outputs\nOOF_Preds    = {}\nMdl_Preds    = {}\nFtreImp      = {}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T14:00:32.151204Z","iopub.execute_input":"2025-05-24T14:00:32.151532Z","iopub.status.idle":"2025-05-24T14:00:32.162585Z","shell.execute_reply.started":"2025-05-24T14:00:32.151505Z","shell.execute_reply":"2025-05-24T14:00:32.160450Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\ndrop_cols = [\"Source\", \"id\", \"Id\", \"Label\", CFG.target, \"fold_nb\"]\n\nfor method, mymodel in tqdm(Mdl_Master.items()):\n\n    PrintColor(\n        f\"\\n{'=' * 20} {method.upper()} MODEL TRAINING {'=' * 20}\\n\"\n    )\n\n    md = \\\n    ModelTrainer(\n        problem_type   = \"regression\",\n        es             = CFG.nbrnd_erly_stp,\n        target         = CFG.target,\n        orig_req       = False,\n        orig_all_folds = CFG.orig_all_folds,\n        metric_lbl     = \"pearsonr\",\n        drop_cols      = drop_cols,\n        pp_preds       = CFG.pstprcs_oof,\n    )\n \n    _, oof_preds, test_preds, ftreimp, _ =  \\\n    md.MakeOfflineModel(\n        Xtrain,\n        ytrain,\n        ygrp,\n        Xtest,\n        mymodel,\n        method,\n        test_preds_req   = True,\n        ftreimp_plot_req = CFG.ftre_plots_req,\n        ntop = 50,\n    )\n\n    OOF_Preds[method]    = oof_preds\n    Mdl_Preds[method]    = test_preds\n    FtreImp[method]      = ftreimp\n\n    del oof_preds, test_preds, ftreimp\n    print()\n    collect()\n\n\n_ = utils.CleanMemory()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T14:00:38.682590Z","iopub.execute_input":"2025-05-24T14:00:38.683023Z","iopub.status.idle":"2025-05-24T14:21:25.581401Z","shell.execute_reply.started":"2025-05-24T14:00:38.682974Z","shell.execute_reply":"2025-05-24T14:21:25.580343Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **OFFLINE ENSEMBLE**","metadata":{}},{"cell_type":"code","source":"%%time\n\nlen_train = Xtrain.loc[Xtrain.Source == \"Competition\"].shape[0]\nmethod    = \"L21R\"\nmodel     = \\\nPipeline(\n    steps = [(\"SS\", StandardScaler()),\n             (\"M\" , Ridge(max_iter = 10000, random_state = CFG.state))\n            ]\n)\n\nmd = \\\nModelTrainer(\n    problem_type   = \"regression\",\n    es             = CFG.nbrnd_erly_stp,\n    target         = CFG.target,\n    orig_req       = False,\n    orig_all_folds = CFG.orig_all_folds,\n    metric_lbl     = \"pearsonr\",\n    drop_cols      = drop_cols,\n    pp_preds       = CFG.pstprcs_oof,\n)\n\nfitted_models, oof_ens_preds, _, _, _ =  \\\nmd.MakeOfflineModel(\n    pd.DataFrame(OOF_Preds).assign(Source = \"Competition\"),\n    ytrain,\n    ygrp,\n    pd.DataFrame(Mdl_Preds).assign(Source = \"Competition\"),\n    model,\n    method,\n    test_preds_req   = True,\n    ftreimp_plot_req = False,\n    ntop             = 50,\n)\n\nscore = utils.ScoreMetric(ytrain, oof_ens_preds)\nPrintColor(f\"\\n\\n---> Overall score = {score:,.8f}\")\n\ncoefs = pd.DataFrame(columns = list(Mdl_Master.keys()))\nfor i, model in enumerate( fitted_models ):\n    coefs.loc[i] = model[\"M\"].coef_\n\ncoefs  = coefs.mean().values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T14:21:34.719400Z","iopub.execute_input":"2025-05-24T14:21:34.719751Z","iopub.status.idle":"2025-05-24T14:21:37.415364Z","shell.execute_reply.started":"2025-05-24T14:21:34.719721Z","shell.execute_reply":"2025-05-24T14:21:37.414167Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **FULL-REFIT**\n\nWe increase the estimators by a small factor from the offline models and refit on the complete dataset","metadata":{}},{"cell_type":"code","source":"%%time \n\nMdl_Master = \\\n{ \n f\"XGB1R\"   : XGBR(**{  \"objective\"              : \"reg:squarederror\",\n                        \"device\"                 : \"cuda\" if CFG.gpu_switch == \"ON\" else \"cpu\", \n                        \"n_estimators\"           : 725 if CFG.test_req == False else 25,\n                        \"learning_rate\"          : 0.096,\n                        \"colsample_bylevel\"      : 0.46,\n                        \"colsample_bynode\"       : 0.60,\n                        \"colsample_bytree\"       : 0.115,\n                        \"gamma\"                  : 1.039,\n                        \"max_depth\"              : 40,\n                        \"max_leaves\"             : 19,\n                        \"min_child_weight\"       : 76,\n                        \"n_jobs\"                 : -1,\n                        \"random_state\"           : CFG.state,\n                        \"reg_alpha\"              : 65.41,\n                        \"reg_lambda\"             : 19.907,\n                        \"subsample\"              : 0.0144,\n                        \"verbosity\"              : 0,\n                    }\n                  ),\n    \n f'LGBM1R'  : LGBMR(**{ \"objective\"          : \"regression_l2\",\n                        'device'             : \"gpu\" if CFG.gpu_switch == \"ON\" else \"cpu\",\n                        \"n_estimators\"       : 500 if CFG.test_req == False else 25,\n                        \"learning_rate\"      : 0.0145,\n                        \"colsample_bytree\"   : 0.524,               \n                        \"min_child_samples\"  : 47,\n                        \"min_child_weight\"   : 0.193,\n                        \"n_jobs\"             : -1,\n                        \"num_leaves\"         : 65,\n                        \"random_state\"       : CFG.state,\n                        \"reg_alpha\"          : 76.69,\n                        \"reg_lambda\"         : 78.57,\n                        \"subsample\"          : 0.35,\n                        \"verbosity\"          : -1\n                      }\n                   ),\n\n f'LGBM2R'  : LGBMR(**{ \"objective\"              : \"regression_l2\",\n                        \"data_sample_strategy\"   : \"goss\",\n                        \"n_estimators\"           : 300 if CFG.test_req == False else 25,\n                        'device'                 : \"gpu\" if CFG.gpu_switch == \"ON\" else \"cpu\",\n                        \"learning_rate\"          : 0.0136,\n                        \"colsample_bytree\"       : 0.32,\n                        \"min_child_samples\"      : 47,\n                        \"min_child_weight\"       : 0.652,\n                        \"n_jobs\"                 : -1,\n                        \"num_leaves\"             : 25,\n                        \"random_state\"           : CFG.state,\n                        \"reg_alpha\"              : 24.43,\n                        \"reg_lambda\"             : 39.82,\n                        \"subsample\"              : 0.21,\n                        \"verbosity\"              : -1\n                     }\n                   ),\n}\n\nFullFitPreds  = {}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T14:28:47.917950Z","iopub.execute_input":"2025-05-24T14:28:47.918377Z","iopub.status.idle":"2025-05-24T14:28:47.929043Z","shell.execute_reply.started":"2025-05-24T14:28:47.918349Z","shell.execute_reply":"2025-05-24T14:28:47.927950Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time \n\nfor method, mymodel in tqdm(Mdl_Master.items()):\n\n    PrintColor(f\"{method.upper()} FULL REFIT\")\n\n    md = \\\n    ModelTrainer(\n        problem_type   = \"regression\",\n        es             = CFG.nbrnd_erly_stp,\n        target         = CFG.target,\n        orig_req       = False,\n        orig_all_folds = CFG.orig_all_folds,\n        metric_lbl     = \"pearsonr\",\n        drop_cols      = drop_cols,\n        pp_preds       = CFG.pstprcs_oof,\n    )\n\n    _, _, test_preds = \\\n    md.MakeOnlineModel(\n        Xtrain.drop(drop_cols, axis=1, errors = \"ignore\"), \n        ytrain, \n        Xtest.drop(drop_cols, axis=1, errors = \"ignore\"), \n        clone(mymodel), \n        method,\n        test_preds_req = True,\n    )\n\n    try:\n        print(f\"---> Shape of test preds = {test_preds.shape}\")\n    except:\n        PrintColor(f\"---> Check - test preds not collated\", color = Fore.RED)\n\n    FullFitPreds[method] = test_preds\n    collect()\n    print()\n\n_ = utils.CleanMemory()\nprint()    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T14:36:15.390869Z","iopub.execute_input":"2025-05-24T14:36:15.391227Z","iopub.status.idle":"2025-05-24T14:41:17.705713Z","shell.execute_reply.started":"2025-05-24T14:36:15.391202Z","shell.execute_reply":"2025-05-24T14:41:17.704864Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **CLOSURE**","metadata":{}},{"cell_type":"code","source":"%%time\n\npd.DataFrame(OOF_Preds).\\\nto_parquet(\n    os.path.join(CFG.op_path, f\"OOF_Preds_{CFG.model_label}{CFG.model_id}.parquet\")\n)\n\npd.DataFrame(Mdl_Preds).\\\nto_parquet(\n    os.path.join(CFG.op_path, f\"Mdl_Preds_{CFG.model_label}{CFG.model_id}.parquet\")\n)\n\njoblib.dump(\n    fitted_models, \n    os.path.join(CFG.op_path, f\"L21R_{CFG.model_label}{CFG.model_id}.joblib\")\n)\n\nif len( list(Mdl_Master.keys()) ) > 1 :\n    sub_fl[\"prediction\"] = \\\n    np.average( pd.DataFrame(FullFitPreds).values, axis = 1, weights = coefs)\n    print(f\"---> Collated ensemble predictions\")\nelse:\n    sub_fl[\"prediction\"] = FullFitPreds[method]\n    print(f\"---> Collated single model predictions\")\n\n\nsub = \\\npd.read_csv(\n    f\"/kaggle/input/drw-crypto-market-prediction-ensemble/sub_ridge_0.122975.csv\"\n)[\"prediction\"].values.flatten()\nsub_fl[\"prediction\"] = 0.70 * sub + 0.3 * sub_fl[\"prediction\"].values.flatten()\n\nsub_fl.to_csv(os.path.join(CFG.op_path, f\"submission.csv\"))\n\nprint()\n!ls\nprint()\n!head submission.csv\n\n_ = utils.CleanMemory()\nprint()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T14:43:28.019295Z","iopub.execute_input":"2025-05-24T14:43:28.019757Z","iopub.status.idle":"2025-05-24T14:43:30.570217Z","shell.execute_reply.started":"2025-05-24T14:43:28.019728Z","shell.execute_reply":"2025-05-24T14:43:30.568899Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null}]}