{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Notebook for Preprocessing Data**\n**from the Jane Street Real-Time Market Data Forecasting Competition**\n\n[Link to competition](https://www.kaggle.com/competitions/jane-street-real-time-market-data-forecasting)\n\nThis notebook preprocesses the competition data and saves it in a way that allows it to fit into memory when opened for further analysis.\n\n**Public Dataset**\n\nThe processed output dataset can be found [here](https://www.kaggle.com/datasets/daniilvolkov/js24-fits-ram).\n\n**File List:**\n\n- `partition_id=0.pkl`\n- `partition_id=1.pkl`\n- `partition_id=2.pkl`\n- `partition_id=3.pkl`\n- `partition_id=4.pkl`\n- `partition_id=5.pkl`\n- `partition_id=6.pkl`\n- `partition_id=7.pkl`\n- `partition_id=8.pkl`\n- `partition_id=9.pkl`\n- `partition_id=all.pkl`\n\nYou can load each part separately or all parts together—it's up to you.\n","metadata":{}},{"cell_type":"code","source":"# # Open files\n# file_path = '/kaggle/input/js24-prepare-data-pkl-gz-output/partition_id=all.pkl.gz'\n# with gzip.open(file_path, 'rb') as f:\n#     df = pickle.load(f)\n\n# # Display dataset info (approximately 8GB in RAM)\n# df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T10:48:13.827556Z","iopub.execute_input":"2024-11-09T10:48:13.828404Z","iopub.status.idle":"2024-11-09T10:48:13.895776Z","shell.execute_reply.started":"2024-11-09T10:48:13.828353Z","shell.execute_reply":"2024-11-09T10:48:13.894797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np, pandas as pd, polars as pl\nimport pickle,gzip\nimport os, gc\n\nimport warnings\n\nwarnings.filterwarnings('ignore')\n\n\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T09:16:19.490585Z","iopub.execute_input":"2024-11-09T09:16:19.491101Z","iopub.status.idle":"2024-11-09T09:16:19.501687Z","shell.execute_reply.started":"2024-11-09T09:16:19.491056Z","shell.execute_reply":"2024-11-09T09:16:19.500218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.\n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n\n    for col in df.columns:\n        col_type = df[col].dtype.name\n\n        if col_type not in ['object', 'category', 'datetime64[ns, UTC]']:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T09:13:08.658858Z","iopub.execute_input":"2024-11-09T09:13:08.659324Z","iopub.status.idle":"2024-11-09T09:13:08.675934Z","shell.execute_reply.started":"2024-11-09T09:13:08.659279Z","shell.execute_reply":"2024-11-09T09:13:08.674370Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"read and resave data\")\n\ndata_frames = []\n\nfor i in range(0,10):\n    print(f'read part {i}')\n    train_tmp = pd.read_parquet(f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}/part-0.parquet\")\n\n    # Optimise DF\n    print(f'optimize part {i}')\n    train_tmp = reduce_mem_usage(train_tmp)\n\n\n    # Save every part\n    print(f'save part {i}')\n    with gzip.open(f'partition_id={i}.pkl.gz', 'wb') as f:\n        pickle.dump(train_tmp, f)\n\n    # Make list of DFs\n    data_frames.append(train_tmp)\n\n    print('clean RAM')\n    del train_tmp\n    gc.collect() \n    print('------------------')\n\n\n# Save all DF\nprint('Concat all DFs in one and save pkl.gz')\ndf = pd.concat(data_frames, ignore_index=True)\ndel data_frames\nwith gzip.open('partition_id=all.pkl.gz', 'wb') as f:\n    pickle.dump(df, f)\n\ndel df\ngc.collect() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T09:27:44.917712Z","iopub.execute_input":"2024-11-09T09:27:44.918223Z","iopub.status.idle":"2024-11-09T09:37:34.118464Z","shell.execute_reply.started":"2024-11-09T09:27:44.918175Z","shell.execute_reply":"2024-11-09T09:37:34.117406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T09:06:43.031477Z","iopub.execute_input":"2024-11-09T09:06:43.031946Z","iopub.status.idle":"2024-11-09T09:06:43.158149Z","shell.execute_reply.started":"2024-11-09T09:06:43.031905Z","shell.execute_reply":"2024-11-09T09:06:43.156934Z"}},"outputs":[],"execution_count":null},{"cell_type":"raw","source":"data_frames = []\n\n# Read data from saved pkl\nfor i in range(10):\n    with gzip.open(f'partition_id={i}.pkl.gz', 'rb') as f:\n        loaded_data = pickle.load(f)\n        data_frames.append(loaded_data)\n\n# Cancat all parts\ndf = pd.concat(data_frames, ignore_index=True)\ndel data_frames\n\n# Save all DF\nwith gzip.open('partition_id=all.pkl.gz', 'wb') as f:\n    pickle.dump(df, f)","metadata":{"execution":{"iopub.status.busy":"2024-11-09T09:39:02.189298Z","iopub.execute_input":"2024-11-09T09:39:02.189830Z","iopub.status.idle":"2024-11-09T09:51:38.931306Z","shell.execute_reply.started":"2024-11-09T09:39:02.189788Z","shell.execute_reply":"2024-11-09T09:51:38.929768Z"}}},{"cell_type":"code","source":"print('finish')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-09T09:55:47.706192Z","iopub.execute_input":"2024-11-09T09:55:47.706771Z","iopub.status.idle":"2024-11-09T09:55:47.713396Z","shell.execute_reply.started":"2024-11-09T09:55:47.706725Z","shell.execute_reply":"2024-11-09T09:55:47.712109Z"}},"outputs":[],"execution_count":null}]}