{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":11305158,"sourceType":"competition"},{"sourceId":203900450,"sourceType":"kernelVersion"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Train Ridge Regressor\n* [1. CFG & Model](#s1)\n* [2. Load Data, Fit & Save](#2)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import polars as pl # Lazy read/preprocess\nimport pandas as pd\nimport numpy as np\nimport random\nfrom sklearn.linear_model import Ridge # Ridge\n###############\nimport os\nimport warnings\nwarnings.filterwarnings('ignore')\n#############################\n# import kaggle_evaluation.jane_street_inference_server # for submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T19:24:17.340891Z","iopub.execute_input":"2025-05-22T19:24:17.342655Z","iopub.status.idle":"2025-05-22T19:24:17.348892Z","shell.execute_reply.started":"2025-05-22T19:24:17.342611Z","shell.execute_reply":"2025-05-22T19:24:17.347479Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### 1. CFG and Model Setup <a id=\"s1\"></a>","metadata":{}},{"cell_type":"code","source":"# R2 wtd\n\ndef r2_val(y_true, y_pred, sample_weight):\n    nom = np.average((y_pred - y_true) ** 2, weights=sample_weight)\n    denom = (np.average((y_true) ** 2, weights=sample_weight) + 1e-38)\n    r2 = 1 -  nom/denom \n    return r2\n\n# Configuration\n\nclass CFG: \n    # Note: this is convenient for \n    # updating data/versioning due to different input\n    # which is a very common use in Kaggle community\n    seed = 42\n    target_col = \"responder_6\"\n    feature_cols = [\"symbol_id\", \"time_id\"] \\\n        + [f\"feature_{idx:02d}\" for idx in range(79)] \\\n        + [f\"responder_{idx}_lag_1\" for idx in range(9)] # use the lag data\n    categorical_cols = []\n\n# ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T19:24:20.481396Z","iopub.execute_input":"2025-05-22T19:24:20.481769Z","iopub.status.idle":"2025-05-22T19:24:20.489316Z","shell.execute_reply.started":"2025-05-22T19:24:20.481744Z","shell.execute_reply":"2025-05-22T19:24:20.488016Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### 2. Load Data and Fit <a id=\"s2\"></a>","metadata":{}},{"cell_type":"code","source":"DT_GT = 1350 \n# Note: Kaggle doesn't have enough RAM for full-size data\n# Full-size can be done using chunk-wise run in Colab Pro+\n\ntrain = pl.scan_parquet( # use lag-1 dataset from a kaggler; not inventing the wheels\n    \"/kaggle/input/js24-preprocessing-create-lags/training.parquet\"\n).filter(pl.col(\"date_id\") > DT_GT).collect().to_pandas()\n\nvalid = pl.scan_parquet(\n    \"/kaggle/input/js24-preprocessing-create-lags/validation.parquet\"\n).filter(pl.col(\"date_id\") > DT_GT).collect().to_pandas()\n\n# train.shape, valid.shape\ntrain = pd.concat([train, valid]).reset_index(drop=True)\ntrain = train.fillna(method = 'ffill').fillna(0)\nvalid = valid.fillna(method = 'ffill').fillna(0)\n\n# Train vs Valid\n\nX_train = train[ CFG.feature_cols ]\ny_train = train[ CFG.target_col ]\nw_train = train[ \"weight\" ]\nX_valid = valid[ CFG.feature_cols ]\ny_valid = valid[ CFG.target_col ]\nw_valid = valid[ \"weight\" ]\n\n(X_train.shape, y_train.shape, w_train.shape, X_valid.shape, y_valid.shape, w_valid.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T19:24:37.717545Z","iopub.execute_input":"2025-05-22T19:24:37.717905Z","iopub.status.idle":"2025-05-22T19:26:04.525448Z","shell.execute_reply.started":"2025-05-22T19:24:37.717882Z","shell.execute_reply":"2025-05-22T19:26:04.523979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# TRAIN = False\nTRAIN = True\n\nif TRAIN:\n    model = Ridge() # use the default\n    model.fit(X_train,y_train)\n    #############################\n    train_pred, valid_pred = model.predict(X_train), model.predict(X_valid)\n    #############################\n    r2_train = r2_val(y_train, train_pred, w_train)\n    r2_validate = r2_val(y_valid, valid_pred, w_valid)\n    ##############################\n    print(f\"Train R2: {r2_train}, Validation R2: {r2_validate}\")\n    # joblib.dump(model, \"js24_ridge-base.pkl\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T19:26:38.788068Z","iopub.execute_input":"2025-05-22T19:26:38.788615Z","iopub.status.idle":"2025-05-22T19:27:46.356934Z","shell.execute_reply.started":"2025-05-22T19:26:38.788580Z","shell.execute_reply":"2025-05-22T19:27:46.355553Z"}},"outputs":[],"execution_count":null}]}