{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from os.path import exists, join\n\nimport numpy as np\nimport plotly.express as px\nimport polars as pl\nimport polars.selectors as cs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T18:33:18.713417Z","iopub.execute_input":"2024-11-26T18:33:18.713898Z","iopub.status.idle":"2024-11-26T18:33:20.359176Z","shell.execute_reply.started":"2024-11-26T18:33:18.713837Z","shell.execute_reply":"2024-11-26T18:33:20.357958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_folder = \"/kaggle/input/jane-street-real-time-market-data-forecasting\"\nif not exists(data_folder): # I download the data inside a local folder named \"data\"\n    data_folder = \"data\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T18:33:20.361653Z","iopub.execute_input":"2024-11-26T18:33:20.362397Z","iopub.status.idle":"2024-11-26T18:33:20.368129Z","shell.execute_reply.started":"2024-11-26T18:33:20.362345Z","shell.execute_reply":"2024-11-26T18:33:20.366785Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train and test parquet analysis\n\n## Load the data\n\nNote: The test.parquet file contains only zeros in the downloaded data, so data analysis was skipped. However, it can be used to test the submission code.","metadata":{}},{"cell_type":"code","source":"# Initialize a list to hold samples from each file\nsamples = []\n\n# Load a sample from each file\nfor i in range(10):\n    file_path = join(data_folder, f\"train.parquet/partition_id={i}/part-0.parquet\")\n    chunk = pl.read_parquet(file_path)\n    samples.append(chunk)\n\n# Concatenate all samples into one DataFrame if needed\ntrain_df = pl.concat(samples)\n\n# Separate features and responders\ntrain_features = train_df.select(cs.starts_with(\"feature_\"))\ntrain_responders = train_df.select(cs.starts_with(\"responder_\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T18:33:20.369524Z","iopub.execute_input":"2024-11-26T18:33:20.369877Z","iopub.status.idle":"2024-11-26T18:34:56.414855Z","shell.execute_reply.started":"2024-11-26T18:33:20.369812Z","shell.execute_reply":"2024-11-26T18:34:56.413433Z"}},"outputs":[],"execution_count":null},{"cell_type":"raw","source":"file_path = join(data_folder, \"test.parquet/date_id=0/part-0.parquet\")\ntest_df = pl.read_parquet(file_path)\n\n# Separate features and responders\ntest_features = test_df.select(cs.starts_with(\"feature_\"))\ntest_responders = test_df.select(cs.starts_with(\"responder_\"))","metadata":{"vscode":{"languageId":"raw"}}},{"cell_type":"markdown","source":"## Feature analysis","metadata":{}},{"cell_type":"code","source":"train_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T18:34:56.418451Z","iopub.execute_input":"2024-11-26T18:34:56.419023Z","iopub.status.idle":"2024-11-26T18:34:56.457960Z","shell.execute_reply.started":"2024-11-26T18:34:56.418962Z","shell.execute_reply":"2024-11-26T18:34:56.456783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_features.min()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T18:34:56.459395Z","iopub.execute_input":"2024-11-26T18:34:56.459740Z","iopub.status.idle":"2024-11-26T18:34:59.342839Z","shell.execute_reply.started":"2024-11-26T18:34:56.459706Z","shell.execute_reply":"2024-11-26T18:34:59.341582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_features.max()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T18:34:59.344688Z","iopub.execute_input":"2024-11-26T18:34:59.345025Z","iopub.status.idle":"2024-11-26T18:35:00.007622Z","shell.execute_reply.started":"2024-11-26T18:34:59.344993Z","shell.execute_reply":"2024-11-26T18:35:00.006296Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Some features are discrete (e.g., features 9 to 11), while most are continuous. Different normalization strategies can be applied to handle these distinct types of data effectively.","metadata":{}},{"cell_type":"code","source":"# compute percentage of nans per column\ntrain_features.select([\n    (((pl.col(col).is_nan() | pl.col(col).is_null()).sum() / len(train_features)).alias(col)) for col in train_features.columns\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T18:35:00.009249Z","iopub.execute_input":"2024-11-26T18:35:00.009658Z","iopub.status.idle":"2024-11-26T18:35:02.705711Z","shell.execute_reply.started":"2024-11-26T18:35:00.009622Z","shell.execute_reply":"2024-11-26T18:35:02.704204Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Several columns have no missing values, while others have significant amounts of missing data (e.g., features 26, 27, 31, etc.). Strategies need to be devised to address this issue effectively.","metadata":{}},{"cell_type":"markdown","source":"## Responders analysis","metadata":{}},{"cell_type":"code","source":"train_responders","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T18:35:02.707582Z","iopub.execute_input":"2024-11-26T18:35:02.708025Z","iopub.status.idle":"2024-11-26T18:35:02.717503Z","shell.execute_reply.started":"2024-11-26T18:35:02.707977Z","shell.execute_reply":"2024-11-26T18:35:02.716311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_responders.std()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T18:35:02.718970Z","iopub.execute_input":"2024-11-26T18:35:02.719388Z","iopub.status.idle":"2024-11-26T18:35:04.848849Z","shell.execute_reply.started":"2024-11-26T18:35:02.719353Z","shell.execute_reply":"2024-11-26T18:35:04.847646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_responders.min()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T18:35:04.851774Z","iopub.execute_input":"2024-11-26T18:35:04.852141Z","iopub.status.idle":"2024-11-26T18:35:04.936365Z","shell.execute_reply.started":"2024-11-26T18:35:04.852107Z","shell.execute_reply":"2024-11-26T18:35:04.935109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_responders.max()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T18:35:04.938012Z","iopub.execute_input":"2024-11-26T18:35:04.938388Z","iopub.status.idle":"2024-11-26T18:35:05.023917Z","shell.execute_reply.started":"2024-11-26T18:35:04.938352Z","shell.execute_reply":"2024-11-26T18:35:05.022601Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The values appear to be standardized between -5.0 and 5.0. Applying a clamp to the model's output could help prevent nonsensical predictions for this challenge.","metadata":{}},{"cell_type":"code","source":"# compute percentage of nans per column\ntrain_responders.select([\n    (((pl.col(col).is_nan() | pl.col(col).is_null()).sum() / len(train_responders)).alias(col)) for col in train_responders.columns\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T18:35:05.025391Z","iopub.execute_input":"2024-11-26T18:35:05.025739Z","iopub.status.idle":"2024-11-26T18:35:05.355701Z","shell.execute_reply.started":"2024-11-26T18:35:05.025707Z","shell.execute_reply":"2024-11-26T18:35:05.354388Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Since all training data includes complete responder values, semi-supervised techniques will not be required.","metadata":{}},{"cell_type":"markdown","source":"## Some quick key numbers","metadata":{}},{"cell_type":"code","source":"print(f'Number of unique dates: {len(train_df[\"date_id\"].unique())}. With min value of {train_df[\"date_id\"].min()}, max value of {train_df[\"date_id\"].max()}')\nprint(f'Number of unique symbol id: {len(train_df[\"symbol_id\"].unique())}. With min value of {train_df[\"symbol_id\"].min()}, max value of {train_df[\"symbol_id\"].max()}')\nprint(f'Number of unique time id: {len(train_df[\"time_id\"].unique())}. With min value of {train_df[\"time_id\"].min()}, max value of {train_df[\"time_id\"].max()}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T18:35:05.357384Z","iopub.execute_input":"2024-11-26T18:35:05.357875Z","iopub.status.idle":"2024-11-26T18:35:06.447110Z","shell.execute_reply.started":"2024-11-26T18:35:05.357820Z","shell.execute_reply":"2024-11-26T18:35:06.445983Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Further analysis is required to examine the (date_id, time_id) pairs in the data.\n\nNote: The provided lags.parquet file is a sample dataset intended to illustrate the format used in the prediction function.","metadata":{}},{"cell_type":"markdown","source":"## Correlations\nAnalyze and compute correlations between features and responders to identify significant relationships.","metadata":{}},{"cell_type":"code","source":"def compute_correlation_matrix(df: pl.DataFrame)  -> tuple[np.ndarray[np.floating], list[str]]:\n    \"\"\"Compute a correlation (pearson) matrix of the dataframe using int and float columns.\n\n    Parameters\n    ----------\n    df : pl.DataFrame\n        the dataframe of the dataset to compute the correlation\n\n    Returns\n    -------\n    tuple[np.ndarray[np.floating], list[str]]\n        the matrix and the columns/row names\n    \"\"\"\n    numeric_cols = df.select(pl.col(pl.Float32, pl.Int8, pl.Int16)).columns\n    res_matrix = np.zeros((len(numeric_cols), len(numeric_cols)))\n\n    for i, col1 in enumerate(numeric_cols):\n        for j, col2 in enumerate(numeric_cols[i:]):\n            correlation = df.select(pl.corr(col1, col2, method=\"pearson\")).item()\n            res_matrix[i, j + i] = correlation  # Fill upper triangle\n            res_matrix[j + i, i] = correlation  # Mirror to lower triangle\n\n    return res_matrix, numeric_cols\n\ndef show_matrix_heatmap(res_matrix: np.ndarray, numeric_cols: list[str], data_name: str) -> None:\n    \"\"\"Create heatmap for correlation matrix.\n\n    Parameters\n    ----------\n    res_matrix : np.ndarray\n        matrix of correlation.\n    numeric_cols : list[str]\n        labels for the row/columns\n    data_name : str\n        dataset name (e.g. \"train\" or \"jane street\")\n    \"\"\"\n    # Create heatmap using Plotly\n    fig = px.imshow(\n        res_matrix,\n        x=numeric_cols,\n        y=numeric_cols,\n        text_auto=True,\n        color_continuous_scale=\"PuOr\",\n        title=f\"Heatmap of correlation between all {data_name} data\",\n        labels={\"color\": \"Value\"},\n        width=1500,\n        height=1500,\n        zmin=-1,  # Minimum value for the color scale\n        zmax=1   # Maximum value for the color scale\n    )\n\n    fig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T18:35:06.449121Z","iopub.execute_input":"2024-11-26T18:35:06.449600Z","iopub.status.idle":"2024-11-26T18:35:06.462172Z","shell.execute_reply.started":"2024-11-26T18:35:06.449550Z","shell.execute_reply":"2024-11-26T18:35:06.460702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"res_matrix, numeric_cols = compute_correlation_matrix(train_df)\nshow_matrix_heatmap(res_matrix, numeric_cols, \"train\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T18:35:06.463603Z","iopub.execute_input":"2024-11-26T18:35:06.463933Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Some features exhibit strong correlations (> 0.8 or < -0.8). Leveraging these relationships by learning a linear function between them can effectively impute missing values in the data.","metadata":{}}]}