{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":11305158,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":7.594014,"end_time":"2024-10-10T11:58:36.355301","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-10-10T11:58:28.761287","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Libs Needed","metadata":{}},{"cell_type":"code","source":"import os\nimport joblib \n######################\nimport pandas as pd\nimport polars as pl\nimport numpy as np \n#################################\n# from joblib import Parallel, delayed\n##########################################\nimport kaggle_evaluation.jane_street_inference_server # API 4 submission\n#####################\nRUN_PREPROCESS = True # change to TRUE \n                       # if re-run preprocess, \n                       # WARNING: this takes time","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-21T12:38:08.523907Z","iopub.execute_input":"2025-05-21T12:38:08.524488Z","iopub.status.idle":"2025-05-21T12:38:08.529695Z","shell.execute_reply.started":"2025-05-21T12:38:08.524440Z","shell.execute_reply":"2025-05-21T12:38:08.528791Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1. Memo Reduce","metadata":{}},{"cell_type":"code","source":"def memo_reduce(df):\n\n    #### display memo size old \n    memo_size_old = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(memo_size_old))\n\n    #### memo reduced by type change\n    for col in df.columns:\n        col_type = df[col].dtype\n        #####\n        if col_type != object and str(col_type)!='category':# cannot be categorical\n            ######\n            c_min,c_max = df[col].min(),df[col].max() # min & max\n            ######\n            if str(col_type)[:3] == 'int': # if integer\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8) # downgrade to int8\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16) # downgrade to int16\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32) # downgrade to int32\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64) # downgrade to int64\n            ####\n            else: # if float\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)  \n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n    \n    #### new memo size\n    memo_size_new = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(memo_size_new))\n    print('Old vs Ne, decreased by {:.1f}%'.format(\n        100 * (memo_size_old - memo_size_new) / memo_size_old)\n         )\n\n    return df","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 2. Add Lags\n* Some kagglers found the add *lag-1 responders* as the features could increase predict performance\n* Here we add *lag-1 responders* to features","metadata":{}},{"cell_type":"code","source":"class LAG_CFG:\n    #################\n    target_col = \"responder_6\"\n    lag_cols_old = [\"date_id\", \"symbol_id\"] + [f\"responder_{idx}\" for idx in range(9)]\n    lag_cols_rename = { f\"responder_{idx}\" : f\"responder_{idx}_lag_1\" for idx in range(9)}\n    #################\n    valid_ratio = 0.05 # hold-out set\n    #################\n    start_dt = 1100 \n    # after date 1100, data become full-sized and stable, \n    # e.g., sysmbol_id, time_id\n###########################\ndef add_lag1():\n    #####################\n    df = pl.scan_parquet( # lazy read\n        f\"/kaggle/input/jane-street-realtime-marketdata-forecasting/train.parquet\"\n    ).select(\n        pl.int_range(pl.len(), dtype=pl.UInt32).alias(\"id\"), # add column id\n        pl.all(),\n    ).with_columns( # new col label = responder_6 x 2\n        (pl.col(LAG_CFG.target_col)*2).cast(pl.Int32).alias(\"label\"),\n    ).filter(\n        pl.col(\"date_id\").gt(LAG_CFG.start_dt)\n    )\n    ######################\n    df_lag = df.select(pl.col(LAG_CFG.lag_cols_old))\n    df_lag = df_lag.rename(LAG_CFG.lag_cols_rename)\n    df_lag = df_lag.with_columns(\n        date_id = pl.col('date_id') + 1,  # lagged by 1 day\n    )\n    df_lag = df_lag.group_by(\n        [\"date_id\", \"symbol_id\"], maintain_order=True).last()  \n    # use the close\\last value for previous date\n    # one can also use mean, median, max, min\n    # Mean average is another option\n    ##########################\n    out = df.join(df_lag, on=[\"date_id\", \"symbol_id\"],  how=\"left\")\n    len_train   = out.select(pl.col(\"date_id\")).collect().shape[0]\n    valid_records = int(len_train * LAG_CFG.valid_ratio)\n    len_ofl_mdl = len_train - valid_records # n rows for offline training\n    last_tr_dt  = out.select(pl.col(\"date_id\")).collect().row(len_ofl_mdl)[0]\n    ###########\n    training_data = out.filter(pl.col(\"date_id\").le(last_tr_dt))\n    validation_data   = out.filter(pl.col(\"date_id\").gt(last_tr_dt))\n    return training_data, validation_data\n########################################\nif RUN_PREPROCESS:\n    train_dat, val_dat = add_lag1()\n    train_dat.collect().write_parquet(\n        f\"training.parquet\", partition_by = \"date_id\",\n    )\n    val_dat.collect().write_parquet(\n        f\"validation.parquet\", partition_by = \"date_id\",\n    )\n    ### after run, pls download from the output or save the output in Kaggle\n#########################################","metadata":{"execution":{"iopub.status.busy":"2025-05-21T12:42:39.803643Z","iopub.execute_input":"2025-05-21T12:42:39.804001Z","iopub.status.idle":"2025-05-21T12:45:16.119451Z","shell.execute_reply.started":"2025-05-21T12:42:39.803972Z","shell.execute_reply":"2025-05-21T12:45:16.117089Z"},"trusted":true},"outputs":[],"execution_count":null}]}