{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# EDA by date\n\nThis notebook explores the train data, to evaluate whether there are patterns related to specific months or times of the year.\n\nIn this competition, the test data is anonymzed and shuffled, meaning that we do not have access to the timestamps. Yet, if there are patterns related to specific times of the year, we may try to infer them and use them to improve predictions. (see also this notebook for the prediction https://www.kaggle.com/code/dalloliogm/rev-engineering-the-original-order-of-test-rows)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import polars as pl\nimport pandas as pd\nimport numpy as np\nfrom sklearn.decomposition import PCA\nimport umap\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# --- 1. Load and shrink data ---\ntrain_df = pl.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\ntrain_df = train_df.select(pl.all().shrink_dtype()).to_pandas()\n\n# --- 2. Add calendar features ---\ntrain_df['date'] = pd.to_datetime(train_df['timestamp'], unit='s')\ntrain_df['month'] = train_df['date'].dt.month\ntrain_df['dayofweek'] = train_df['date'].dt.dayofweek\ntrain_df['is_tax_season'] = train_df['month'].isin([3, 4]).astype(int)\ntrain_df['is_q_end'] = train_df['date'].dt.is_quarter_end.astype(int)\n\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T13:15:54.744174Z","iopub.execute_input":"2025-05-28T13:15:54.744983Z","iopub.status.idle":"2025-05-28T13:16:05.793660Z","shell.execute_reply.started":"2025-05-28T13:15:54.744949Z","shell.execute_reply":"2025-05-28T13:16:05.792844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T13:16:05.794998Z","iopub.execute_input":"2025-05-28T13:16:05.795279Z","iopub.status.idle":"2025-05-28T13:16:05.855996Z","shell.execute_reply.started":"2025-05-28T13:16:05.795256Z","shell.execute_reply":"2025-05-28T13:16:05.854867Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T13:16:14.384660Z","iopub.execute_input":"2025-05-28T13:16:14.384953Z","iopub.status.idle":"2025-05-28T13:16:14.390921Z","shell.execute_reply.started":"2025-05-28T13:16:14.384934Z","shell.execute_reply":"2025-05-28T13:16:14.390115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.month.value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T13:19:05.131993Z","iopub.execute_input":"2025-05-28T13:19:05.132957Z","iopub.status.idle":"2025-05-28T13:19:05.147170Z","shell.execute_reply.started":"2025-05-28T13:19:05.132927Z","shell.execute_reply":"2025-05-28T13:19:05.146178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 3. Select numeric features ---\nfeature_cols = [col for col in train_df.columns if col.startswith('X') or col in ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']]\nX = train_df[feature_cols].replace([np.inf, -np.inf], np.nan).fillna(0).astype(np.float32)\n\n# Optional: Subsample to 20k rows to plot faster\n#X = X.sample(n=20000, random_state=42)\nmeta = train_df.loc[X.index, ['month', 'dayofweek', 'is_tax_season', 'is_q_end']]\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T13:19:20.156497Z","iopub.execute_input":"2025-05-28T13:19:20.156919Z","iopub.status.idle":"2025-05-28T13:19:29.947100Z","shell.execute_reply.started":"2025-05-28T13:19:20.156888Z","shell.execute_reply":"2025-05-28T13:19:29.945728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T13:17:14.897776Z","iopub.execute_input":"2025-05-28T13:17:14.898190Z","iopub.status.idle":"2025-05-28T13:17:14.938810Z","shell.execute_reply.started":"2025-05-28T13:17:14.898160Z","shell.execute_reply":"2025-05-28T13:17:14.937166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 4. Dimensionality reduction (PCA + UMAP) ---\npca = PCA(n_components=50).fit_transform(X)\nembedding = umap.UMAP(n_neighbors=30, min_dist=0.3, random_state=42).fit_transform(pca)\n\n# --- 5. Plot helper ---\ndef plot_embedding(embedding, labels, title):\n    plt.figure(figsize=(8, 6))\n    sns.scatterplot(x=embedding[:,0], y=embedding[:,1], hue=labels, palette='Spectral', s=10, alpha=0.6)\n    plt.title(title)\n    plt.legend(title='', bbox_to_anchor=(1.05, 1), loc='upper left')\n    plt.tight_layout()\n    plt.show()\n\n# --- 6. Visualizations ---\nplot_embedding(embedding, meta['month'], 'PCA+UMAP by Month')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T13:19:29.948757Z","iopub.execute_input":"2025-05-28T13:19:29.949830Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_embedding(embedding, meta['dayofweek'], 'PCA+UMAP by Day of Week')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T13:18:22.001972Z","iopub.execute_input":"2025-05-28T13:18:22.002262Z","iopub.status.idle":"2025-05-28T13:18:23.043562Z","shell.execute_reply.started":"2025-05-28T13:18:22.002243Z","shell.execute_reply":"2025-05-28T13:18:23.042465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_embedding(embedding, meta['is_tax_season'], 'PCA+UMAP by Tax Season')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T13:18:28.938569Z","iopub.execute_input":"2025-05-28T13:18:28.938941Z","iopub.status.idle":"2025-05-28T13:18:29.583728Z","shell.execute_reply.started":"2025-05-28T13:18:28.938917Z","shell.execute_reply":"2025-05-28T13:18:29.582787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_embedding(embedding, meta['is_q_end'], 'PCA+UMAP by Quarter End')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T13:18:32.455190Z","iopub.execute_input":"2025-05-28T13:18:32.455534Z","iopub.status.idle":"2025-05-28T13:18:33.087587Z","shell.execute_reply.started":"2025-05-28T13:18:32.455509Z","shell.execute_reply":"2025-05-28T13:18:33.086590Z"}},"outputs":[],"execution_count":null}]}