{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":209102443,"sourceType":"kernelVersion"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"markdown","source":"In this notebook I'll show how to train a xgboost model with the data provided in this competition plus additional lags data generated in the EDA notebook https://www.kaggle.com/code/simonedegasperis/starter-eda","metadata":{}},{"cell_type":"code","source":"# Imports\nimport polars as pl\nimport numpy as np\nimport os\nimport copy\nfrom pathlib import Path","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:45:05.283120Z","iopub.execute_input":"2024-11-29T16:45:05.283513Z","iopub.status.idle":"2024-11-29T16:45:05.390918Z","shell.execute_reply.started":"2024-11-29T16:45:05.283475Z","shell.execute_reply":"2024-11-29T16:45:05.389475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# define columns to read\nfeature_cols = [f'feature_{x:02}' for x in range(79)]\nresponder_cols = [f'responder_{i}' for i in range(9)]\nresponder_lags = [f'responder_{i}_lag_1' for i in range(9)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:45:05.392488Z","iopub.execute_input":"2024-11-29T16:45:05.392969Z","iopub.status.idle":"2024-11-29T16:45:05.400689Z","shell.execute_reply.started":"2024-11-29T16:45:05.392917Z","shell.execute_reply":"2024-11-29T16:45:05.398927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# define base dir\nDATA_DIR = Path('/kaggle/input/')\nN_PARTITION = 10","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:45:05.402393Z","iopub.execute_input":"2024-11-29T16:45:05.402795Z","iopub.status.idle":"2024-11-29T16:45:05.411452Z","shell.execute_reply.started":"2024-11-29T16:45:05.402757Z","shell.execute_reply":"2024-11-29T16:45:05.410353Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Preparation","metadata":{}},{"cell_type":"markdown","source":"In order to create the new train and validation dataframes we have to chunk the lags otherwise the memory will explode.","metadata":{}},{"cell_type":"markdown","source":"# Lags\nGenerate lags for each train dataframe","metadata":{}},{"cell_type":"code","source":"fields = []\nfields.extend(['date_id', 'time_id', 'symbol_id', 'weight'])\nfields.extend(responder_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:45:05.414978Z","iopub.execute_input":"2024-11-29T16:45:05.415472Z","iopub.status.idle":"2024-11-29T16:45:05.427777Z","shell.execute_reply.started":"2024-11-29T16:45:05.415431Z","shell.execute_reply":"2024-11-29T16:45:05.426110Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_parquets = [DATA_DIR / f\"jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}/part-0.parquet\" for i in range(N_PARTITION)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:45:05.429856Z","iopub.execute_input":"2024-11-29T16:45:05.430381Z","iopub.status.idle":"2024-11-29T16:45:05.441839Z","shell.execute_reply.started":"2024-11-29T16:45:05.430339Z","shell.execute_reply":"2024-11-29T16:45:05.440641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# chunk the original lags dataframe to reduce memory consumption\nfor i, _f in enumerate(train_parquets):\n    print(f\"Processing dataframe {i}\")\n    lags = pl.read_parquet(DATA_DIR / 'starter-eda/train_lags.parquet')\n\n    pl_train = pl.read_parquet(_f, columns=fields)\n    min_date = pl_train['date_id'].min()\n    max_date = pl_train['date_id'].max()\n\n    lags = lags.filter(pl.col('date_id')<=max_date)\n    lags = lags.filter(pl.col('date_id')>=min_date)   \n\n    lags.write_parquet(f\"train_lags_{i}.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:45:05.443775Z","iopub.execute_input":"2024-11-29T16:45:05.444182Z","iopub.status.idle":"2024-11-29T16:45:27.939857Z","shell.execute_reply.started":"2024-11-29T16:45:05.444143Z","shell.execute_reply":"2024-11-29T16:45:27.938436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.listdir()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:45:27.941258Z","iopub.execute_input":"2024-11-29T16:45:27.941686Z","iopub.status.idle":"2024-11-29T16:45:27.952186Z","shell.execute_reply.started":"2024-11-29T16:45:27.941648Z","shell.execute_reply":"2024-11-29T16:45:27.950674Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Join features and lags","metadata":{}},{"cell_type":"code","source":"# read all training data\nfields = []\ntarget_col = \"responder_6\"\nfields.extend(['date_id', 'time_id', 'symbol_id', 'weight'])\nfields.extend([target_col])\nfields.extend(feature_cols)\nfor i, _f in enumerate(train_parquets):\n    print(f\"Processing dataframe {i}\")\n    pl_train = pl.read_parquet(_f, columns=fields)\n    # let's keep the last 2 entirely and subset the others\n    if i < 8:\n        pl_train = pl_train[10000:60000]\n    lags = pl.read_parquet(f\"train_lags_{i}.parquet\")\n    lags = lags.unique(subset=[\"date_id\", \"symbol_id\"])\n\n    # print(f\"Shape features dataset {i}\")\n    # print(pl_train.shape)\n    # print(f\"Shape lags dataset {i}\")\n    # print(lags.shape)\n\n    # join\n    pl_train = pl_train.join(lags, on=[\"date_id\", \"symbol_id\"],  how=\"inner\")\n\n    print(f\"Shape joined dataset {i}\")\n    print(pl_train.shape)\n    print(\"-----------\")\n\n    pl_train.write_parquet(f\"dataset_{i}.parquet\")\nlags = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:45:27.953856Z","iopub.execute_input":"2024-11-29T16:45:27.954345Z","iopub.status.idle":"2024-11-29T16:46:32.753292Z","shell.execute_reply.started":"2024-11-29T16:45:27.954307Z","shell.execute_reply":"2024-11-29T16:46:32.751347Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# XGB model","metadata":{}},{"cell_type":"code","source":"train_data = [f\"dataset_{i}.parquet\" for i in range(N_PARTITION)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:46:32.755284Z","iopub.execute_input":"2024-11-29T16:46:32.755794Z","iopub.status.idle":"2024-11-29T16:46:32.766090Z","shell.execute_reply.started":"2024-11-29T16:46:32.755740Z","shell.execute_reply":"2024-11-29T16:46:32.764384Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pl_train = pl.concat([pl.read_parquet(_f) for _f in train_data])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:46:32.767947Z","iopub.execute_input":"2024-11-29T16:46:32.768584Z","iopub.status.idle":"2024-11-29T16:46:36.989919Z","shell.execute_reply.started":"2024-11-29T16:46:32.768519Z","shell.execute_reply":"2024-11-29T16:46:36.988381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pl_train.tail()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:46:36.991300Z","iopub.execute_input":"2024-11-29T16:46:36.991726Z","iopub.status.idle":"2024-11-29T16:46:37.017532Z","shell.execute_reply.started":"2024-11-29T16:46:36.991674Z","shell.execute_reply":"2024-11-29T16:46:37.015980Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_col = \"responder_6\"\ntrain_columns = copy.copy(feature_cols)\ntrain_columns.extend(responder_lags)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:46:37.018868Z","iopub.execute_input":"2024-11-29T16:46:37.019219Z","iopub.status.idle":"2024-11-29T16:46:37.549225Z","shell.execute_reply.started":"2024-11-29T16:46:37.019186Z","shell.execute_reply":"2024-11-29T16:46:37.547929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# columns used for training\nprint(train_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:46:37.554366Z","iopub.execute_input":"2024-11-29T16:46:37.554796Z","iopub.status.idle":"2024-11-29T16:46:38.537022Z","shell.execute_reply.started":"2024-11-29T16:46:37.554760Z","shell.execute_reply":"2024-11-29T16:46:38.535522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# target col\nprint(target_col)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:46:38.538418Z","iopub.execute_input":"2024-11-29T16:46:38.538839Z","iopub.status.idle":"2024-11-29T16:46:38.554268Z","shell.execute_reply.started":"2024-11-29T16:46:38.538791Z","shell.execute_reply":"2024-11-29T16:46:38.553043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = pl_train.select(train_columns).to_numpy()\ny = pl_train.select(target_col).to_numpy().flatten()\nweights = pl_train.select(\"weight\").to_numpy().flatten()\n# X = np.nan_to_num(X)\n# y = np.nan_to_num(y)\n# weights = np.nan_to_num(weights)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:46:38.555725Z","iopub.execute_input":"2024-11-29T16:46:38.556092Z","iopub.status.idle":"2024-11-29T16:46:40.589436Z","shell.execute_reply.started":"2024-11-29T16:46:38.556058Z","shell.execute_reply":"2024-11-29T16:46:40.587935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:46:40.591439Z","iopub.execute_input":"2024-11-29T16:46:40.591866Z","iopub.status.idle":"2024-11-29T16:46:41.974863Z","shell.execute_reply.started":"2024-11-29T16:46:40.591827Z","shell.execute_reply":"2024-11-29T16:46:41.973276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\nweights_train, weights_test = train_test_split(weights, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:46:41.976327Z","iopub.execute_input":"2024-11-29T16:46:41.976857Z","iopub.status.idle":"2024-11-29T16:48:24.532794Z","shell.execute_reply.started":"2024-11-29T16:46:41.976820Z","shell.execute_reply":"2024-11-29T16:48:24.531405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dtrain = xgb.DMatrix(X_train, label=y_train, weight=weights_train)  # Training set\ndtest = xgb.DMatrix(X_test, label=y_test, weight=weights_test)    # Test set","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:48:24.534265Z","iopub.execute_input":"2024-11-29T16:48:24.534713Z","iopub.status.idle":"2024-11-29T16:48:42.129307Z","shell.execute_reply.started":"2024-11-29T16:48:24.534674Z","shell.execute_reply":"2024-11-29T16:48:42.127796Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def r2_metric(predt: np.ndarray, dtrain: xgb.DMatrix):\n    '''Compute r2 metric for xgboost'''\n    y_true = dtrain.get_label()\n    weights = dtrain.get_weight()\n    numerator = np.sum(weights * (y_true - predt) ** 2)\n    denominator = np.sum(weights * (y_true ** 2))\n    r2_score = 1 - (numerator / denominator)\n    return ('R2', float(r2_score))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:48:42.130701Z","iopub.execute_input":"2024-11-29T16:48:42.131077Z","iopub.status.idle":"2024-11-29T16:48:42.139013Z","shell.execute_reply.started":"2024-11-29T16:48:42.131035Z","shell.execute_reply":"2024-11-29T16:48:42.137374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train with custom metric and capture the trained model\nparams = {\n    'objective': 'reg:squarederror',\n    'max_depth': 4,\n    'eta': 0.1,\n}\n\nmodel = xgb.train(\n    params,\n    dtrain,\n    num_boost_round=50,\n    evals=[(dtest, 'test')],  # Test R2 metric on test datasets\n    custom_metric=r2_metric,  # Custom metric function\n    verbose_eval=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:48:42.140777Z","iopub.execute_input":"2024-11-29T16:48:42.141363Z","iopub.status.idle":"2024-11-29T16:51:23.225096Z","shell.execute_reply.started":"2024-11-29T16:48:42.141307Z","shell.execute_reply":"2024-11-29T16:51:23.223579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def r2_score(y_true, y_pred, weights):\n    \"\"\"\n    Calculate the sample weighted zero-mean R-squared score (R2).\n\n    Parameters:\n    - y_true (pd.Series or np.array): Ground truth values.\n    - y_pred (pd.Series or np.array): Predicted values.\n    - weights (pd.Series or np.array): Sample weights.\n\n    Returns:\n    - float: R2 score.\n    \"\"\"\n    numerator = np.sum(weights * (y_true - y_pred) ** 2)\n    denominator = np.sum(weights * (y_true ** 2))\n    r2_score = 1 - (numerator / denominator)\n    return r2_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:51:23.226629Z","iopub.execute_input":"2024-11-29T16:51:23.226995Z","iopub.status.idle":"2024-11-29T16:51:23.234041Z","shell.execute_reply.started":"2024-11-29T16:51:23.226945Z","shell.execute_reply":"2024-11-29T16:51:23.232681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(dtest)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:51:23.235527Z","iopub.execute_input":"2024-11-29T16:51:23.235920Z","iopub.status.idle":"2024-11-29T16:51:24.245629Z","shell.execute_reply.started":"2024-11-29T16:51:23.235885Z","shell.execute_reply":"2024-11-29T16:51:24.244608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate the model\nr2_xgb = r2_score(y_test, y_pred, weights_test)\nprint(f\"R2: {r2_xgb}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:51:24.246882Z","iopub.execute_input":"2024-11-29T16:51:24.248157Z","iopub.status.idle":"2024-11-29T16:51:24.267522Z","shell.execute_reply.started":"2024-11-29T16:51:24.248103Z","shell.execute_reply":"2024-11-29T16:51:24.266257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# save the model\nmodel.save_model(\"xgboost_model.json\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:51:24.269030Z","iopub.execute_input":"2024-11-29T16:51:24.269506Z","iopub.status.idle":"2024-11-29T16:51:24.280364Z","shell.execute_reply.started":"2024-11-29T16:51:24.269457Z","shell.execute_reply":"2024-11-29T16:51:24.279452Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LightGBM","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:51:24.281520Z","iopub.execute_input":"2024-11-29T16:51:24.282427Z","iopub.status.idle":"2024-11-29T16:51:25.406559Z","shell.execute_reply.started":"2024-11-29T16:51:24.282377Z","shell.execute_reply":"2024-11-29T16:51:25.405176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a LightGBM Dataset\ntrain_data = lgb.Dataset(X_train, label=y_train)\ntest_data = lgb.Dataset(X_test, label=y_test, reference=train_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:51:25.408001Z","iopub.execute_input":"2024-11-29T16:51:25.408640Z","iopub.status.idle":"2024-11-29T16:51:25.414672Z","shell.execute_reply.started":"2024-11-29T16:51:25.408599Z","shell.execute_reply":"2024-11-29T16:51:25.413264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define Parameters\nparams = {\n    'objective': 'regression',  # Use 'binary' for binary classification\n    'metric': 'rmse',          # Root Mean Squared Error\n    'boosting_type': 'gbdt',   # Gradient Boosted Decision Trees\n    'num_leaves': 31,\n    'learning_rate': 0.1,\n    'feature_fraction': 0.8\n}\n\n# Train the model\nlgbm_model = lgb.train(\n    params,\n    train_data,\n    valid_sets=[train_data, test_data],\n    num_boost_round=50\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:51:25.416212Z","iopub.execute_input":"2024-11-29T16:51:25.416709Z","iopub.status.idle":"2024-11-29T16:53:20.578091Z","shell.execute_reply.started":"2024-11-29T16:51:25.416669Z","shell.execute_reply":"2024-11-29T16:53:20.576404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = lgbm_model.predict(X_test, num_iteration=lgbm_model.best_iteration)\n# Evaluate the model\nr2_light = r2_score(y_test, y_pred, weights_test)\nprint(f\"R2: {r2_light}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:53:20.579815Z","iopub.execute_input":"2024-11-29T16:53:20.580359Z","iopub.status.idle":"2024-11-29T16:53:23.741338Z","shell.execute_reply.started":"2024-11-29T16:53:20.580286Z","shell.execute_reply":"2024-11-29T16:53:23.739877Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_model.save_model(\"lgbm_model.json\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:53:23.742726Z","iopub.execute_input":"2024-11-29T16:53:23.743081Z","iopub.status.idle":"2024-11-29T16:53:23.754921Z","shell.execute_reply.started":"2024-11-29T16:53:23.743046Z","shell.execute_reply":"2024-11-29T16:53:23.753623Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Combine models","metadata":{}},{"cell_type":"markdown","source":"## Average","metadata":{}},{"cell_type":"code","source":"# lightgbm\ny_pred1 = lgbm_model.predict(X_test, num_iteration=lgbm_model.best_iteration)\n# xgb\ndtest = xgb.DMatrix(X_test, label=y, weight=weights_test)    # Test set\ny_pred2 = model.predict(dtest)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:53:23.756564Z","iopub.execute_input":"2024-11-29T16:53:23.757642Z","iopub.status.idle":"2024-11-29T16:53:35.411959Z","shell.execute_reply.started":"2024-11-29T16:53:23.757589Z","shell.execute_reply":"2024-11-29T16:53:35.410732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = (y_pred1+y_pred2)/2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:53:35.413573Z","iopub.execute_input":"2024-11-29T16:53:35.413975Z","iopub.status.idle":"2024-11-29T16:53:35.429679Z","shell.execute_reply.started":"2024-11-29T16:53:35.413939Z","shell.execute_reply":"2024-11-29T16:53:35.428244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"r2 = r2_score(y_test, y_pred, weights_test)\nprint(f\"R2: {r2}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:53:35.430995Z","iopub.execute_input":"2024-11-29T16:53:35.431326Z","iopub.status.idle":"2024-11-29T16:53:35.461622Z","shell.execute_reply.started":"2024-11-29T16:53:35.431296Z","shell.execute_reply":"2024-11-29T16:53:35.460203Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Stacking","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import KFold","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T16:53:35.463223Z","iopub.execute_input":"2024-11-29T16:53:35.463738Z","iopub.status.idle":"2024-11-29T16:53:35.873606Z","shell.execute_reply.started":"2024-11-29T16:53:35.463685Z","shell.execute_reply":"2024-11-29T16:53:35.872512Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Meta-model\nmeta_model = RandomForestRegressor(n_estimators=20, max_depth=3, random_state=4)\n# meta_model.fit(stacked_features, y_test)  # y_val: True labels of validation set\n\n# Combine predictions into a feature matrix\nstacked_features = np.column_stack((y_pred1, y_pred2))  # Shape: (n_samples, 2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T17:10:12.268966Z","iopub.execute_input":"2024-11-29T17:10:12.270079Z","iopub.status.idle":"2024-11-29T17:10:12.302893Z","shell.execute_reply.started":"2024-11-29T17:10:12.270028Z","shell.execute_reply":"2024-11-29T17:10:12.301595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the RandomForestRegressor meta-model\nmeta_model.fit(stacked_features, y_test)\n\nfinal_predictions = meta_model.predict(stacked_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T17:10:13.501359Z","iopub.execute_input":"2024-11-29T17:10:13.502698Z","iopub.status.idle":"2024-11-29T17:11:09.836204Z","shell.execute_reply.started":"2024-11-29T17:10:13.502651Z","shell.execute_reply":"2024-11-29T17:11:09.834497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"r2 = r2_score(y_test, final_predictions, weights_test)\nprint(f\"R2: {r2}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T17:11:09.838405Z","iopub.execute_input":"2024-11-29T17:11:09.838859Z","iopub.status.idle":"2024-11-29T17:11:09.867035Z","shell.execute_reply.started":"2024-11-29T17:11:09.838818Z","shell.execute_reply":"2024-11-29T17:11:09.865714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\n\nwith open(\"meta_model.pkl\", \"wb\") as f:\n    pickle.dump(meta_model, f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T17:06:47.300636Z","iopub.execute_input":"2024-11-29T17:06:47.301070Z","iopub.status.idle":"2024-11-29T17:06:47.310002Z","shell.execute_reply.started":"2024-11-29T17:06:47.301035Z","shell.execute_reply":"2024-11-29T17:06:47.308432Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Clean the data","metadata":{}},{"cell_type":"code","source":"os.listdir()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T17:11:39.977157Z","iopub.execute_input":"2024-11-29T17:11:39.977725Z","iopub.status.idle":"2024-11-29T17:11:39.988464Z","shell.execute_reply.started":"2024-11-29T17:11:39.977666Z","shell.execute_reply":"2024-11-29T17:11:39.987158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# remove other data\nimport glob\nfor file_to_remove in glob.glob(\"*.parquet\"):\n    print(f\"Remove file {file_to_remove}\")\n    os.remove(file_to_remove)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T17:11:44.407394Z","iopub.execute_input":"2024-11-29T17:11:44.408437Z","iopub.status.idle":"2024-11-29T17:11:44.482401Z","shell.execute_reply.started":"2024-11-29T17:11:44.408392Z","shell.execute_reply":"2024-11-29T17:11:44.481162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.listdir()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T17:11:47.846721Z","iopub.execute_input":"2024-11-29T17:11:47.847178Z","iopub.status.idle":"2024-11-29T17:11:47.855876Z","shell.execute_reply.started":"2024-11-29T17:11:47.847138Z","shell.execute_reply":"2024-11-29T17:11:47.854649Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}