{"cells": [{"cell_type": "markdown", "metadata": {}, "source": "# Exploratory Data Analysis for Jane Street Forecasting"}, {"cell_type": "code", "execution_count": null, "metadata": {"_cell_guid": "b1076dfc-b9ad-4769-8c92-a6c4dae69d19", "_uuid": "8f2839f25d086af736a60e9eeb907d3b93b6e0e5", "execution": {"iopub.execute_input": "2024-10-21T14:12:47.141165Z", "iopub.status.busy": "2024-10-21T14:12:47.140737Z", "iopub.status.idle": "2024-10-21T14:12:50.539413Z", "shell.execute_reply": "2024-10-21T14:12:50.537474Z", "shell.execute_reply.started": "2024-10-21T14:12:47.141123Z"}, "trusted": true}, "outputs": [], "source": "import os\nfrom pathlib import Path\n\nimport pandas as pd\nimport numpy as np\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import train_test_split, TimeSeriesSplit\nfrom sklearn.preprocessing import StandardScaler\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\n\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")"}, {"cell_type": "markdown", "metadata": {}, "source": "## Paths"}, {"cell_type": "code", "execution_count": null, "metadata": {"execution": {"iopub.execute_input": "2024-10-21T14:12:50.542001Z", "iopub.status.busy": "2024-10-21T14:12:50.541384Z", "iopub.status.idle": "2024-10-21T14:12:50.548796Z", "shell.execute_reply": "2024-10-21T14:12:50.547402Z", "shell.execute_reply.started": "2024-10-21T14:12:50.541958Z"}, "trusted": true}, "outputs": [], "source": "INPUT_DIR = Path('/kaggle/input/jane-street-real-time-market-data-forecasting')"}, {"cell_type": "markdown", "metadata": {}, "source": "## Data reading"}, {"cell_type": "code", "execution_count": null, "metadata": {"execution": {"iopub.execute_input": "2024-10-21T14:12:50.551071Z", "iopub.status.busy": "2024-10-21T14:12:50.550557Z", "iopub.status.idle": "2024-10-21T14:12:55.481115Z", "shell.execute_reply": "2024-10-21T14:12:55.47961Z", "shell.execute_reply.started": "2024-10-21T14:12:50.551015Z"}, "trusted": true}, "outputs": [], "source": "file_name = 'train.parquet/partition_id=0/part-0.parquet'\n\ndf = pd.read_parquet(os.path.join(INPUT_DIR, file_name))\nprint(len(df))\ndf.head(10)"}], "metadata": {"kaggle": {"accelerator": "none", "dataSources": [{"databundleVersionId": 9871156, "sourceId": 84493, "sourceType": "competition"}], "dockerImageVersionId": 30786, "isGpuEnabled": false, "isInternetEnabled": true, "language": "python", "sourceType": "notebook"}, "kernelspec": {"display_name": "Python 3", "language": "python", "name": "python3"}, "language_info": {"codemirror_mode": {"name": "ipython", "version": 3}, "file_extension": ".py", "mimetype": "text/x-python", "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "version": "3.10.14"}}, "nbformat": 4, "nbformat_minor": 4}