{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:41:27.814175Z","iopub.execute_input":"2025-07-12T18:41:27.814900Z","iopub.status.idle":"2025-07-12T18:41:28.147599Z","shell.execute_reply.started":"2025-07-12T18:41:27.814870Z","shell.execute_reply":"2025-07-12T18:41:28.146930Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/train.parquet\")\ndf_test = pd.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/test.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:41:28.149069Z","iopub.execute_input":"2025-07-12T18:41:28.149394Z","iopub.status.idle":"2025-07-12T18:42:10.209895Z","shell.execute_reply.started":"2025-07-12T18:41:28.149374Z","shell.execute_reply":"2025-07-12T18:42:10.208944Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.read_csv(\"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:42:10.210841Z","iopub.execute_input":"2025-07-12T18:42:10.211142Z","iopub.status.idle":"2025-07-12T18:42:10.515559Z","shell.execute_reply.started":"2025-07-12T18:42:10.211118Z","shell.execute_reply":"2025-07-12T18:42:10.514693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:42:10.516492Z","iopub.execute_input":"2025-07-12T18:42:10.516713Z","iopub.status.idle":"2025-07-12T18:42:10.522663Z","shell.execute_reply.started":"2025-07-12T18:42:10.516685Z","shell.execute_reply":"2025-07-12T18:42:10.521824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:42:10.524902Z","iopub.execute_input":"2025-07-12T18:42:10.525152Z","iopub.status.idle":"2025-07-12T18:42:10.561234Z","shell.execute_reply.started":"2025-07-12T18:42:10.525131Z","shell.execute_reply":"2025-07-12T18:42:10.560489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:42:10.562072Z","iopub.execute_input":"2025-07-12T18:42:10.562597Z","iopub.status.idle":"2025-07-12T18:42:10.567365Z","shell.execute_reply.started":"2025-07-12T18:42:10.562568Z","shell.execute_reply":"2025-07-12T18:42:10.566697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:42:10.568146Z","iopub.execute_input":"2025-07-12T18:42:10.568419Z","iopub.status.idle":"2025-07-12T18:42:10.600891Z","shell.execute_reply.started":"2025-07-12T18:42:10.568398Z","shell.execute_reply":"2025-07-12T18:42:10.600032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\nTrain columns and data types:\")\nprint(df_train.dtypes)\n\nprint(\"\\nTest columns and data types:\")\nprint(df_test.dtypes)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:42:10.601801Z","iopub.execute_input":"2025-07-12T18:42:10.602086Z","iopub.status.idle":"2025-07-12T18:42:10.620581Z","shell.execute_reply.started":"2025-07-12T18:42:10.602054Z","shell.execute_reply":"2025-07-12T18:42:10.620025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#  Check for missing values\nprint(\"\\nMissing values in train data:\")\nprint(df_train.isnull().sum())\n\nprint(\"\\nMissing values in test data:\")\nprint(df_test.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:42:10.621263Z","iopub.execute_input":"2025-07-12T18:42:10.621500Z","iopub.status.idle":"2025-07-12T18:42:13.084575Z","shell.execute_reply.started":"2025-07-12T18:42:10.621475Z","shell.execute_reply":"2025-07-12T18:42:13.083748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\n\n# statistics summary\nprint(\"\\nTrain data statistics:\")\nprint(df_train.describe())\n\nprint(\"\\nTest data statistics:\")\nprint(df_test.describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:42:13.085454Z","iopub.execute_input":"2025-07-12T18:42:13.085700Z","iopub.status.idle":"2025-07-12T18:42:42.485080Z","shell.execute_reply.started":"2025-07-12T18:42:13.085679Z","shell.execute_reply":"2025-07-12T18:42:42.484252Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Univariate Analysis (Feature Distributions)","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:42:42.485948Z","iopub.execute_input":"2025-07-12T18:42:42.486246Z","iopub.status.idle":"2025-07-12T18:42:43.064422Z","shell.execute_reply.started":"2025-07-12T18:42:42.486216Z","shell.execute_reply":"2025-07-12T18:42:43.063815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"public_features = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\n\nplt.figure(figsize=(15, 10))\nfor i, feature in enumerate(public_features, 1):\n    plt.subplot(2, 3, i)\n    sns.histplot(df_train[feature], bins=50, kde=True)\n    plt.title(f'Distribution of {feature}')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:42:43.065153Z","iopub.execute_input":"2025-07-12T18:42:43.065606Z","iopub.status.idle":"2025-07-12T18:42:54.266674Z","shell.execute_reply.started":"2025-07-12T18:42:43.065586Z","shell.execute_reply":"2025-07-12T18:42:54.265864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(6,4))\nsns.histplot(df_train['label'], bins=100, kde=True)\nplt.title('Distribution of Target (label)')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:42:54.267575Z","iopub.execute_input":"2025-07-12T18:42:54.267871Z","iopub.status.idle":"2025-07-12T18:42:56.554572Z","shell.execute_reply.started":"2025-07-12T18:42:54.267834Z","shell.execute_reply":"2025-07-12T18:42:56.553815Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Bivariate Analysis (Feature vs Target)","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 10))\nfor i, feature in enumerate(public_features, 1):\n    plt.subplot(2, 3, i)\n    plt.hexbin(df_train[feature], df_train['label'], gridsize=50, cmap='Blues')\n    plt.xlabel(feature)\n    plt.ylabel('label')\n    plt.title(f'{feature} vs label')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:42:56.557722Z","iopub.execute_input":"2025-07-12T18:42:56.557930Z","iopub.status.idle":"2025-07-12T18:42:57.584066Z","shell.execute_reply.started":"2025-07-12T18:42:56.557912Z","shell.execute_reply":"2025-07-12T18:42:57.583199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corr_matrix = df_train[public_features + ['label']].corr()\nplt.figure(figsize=(8,6))\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm')\nplt.title('Correlation Matrix')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:42:57.585021Z","iopub.execute_input":"2025-07-12T18:42:57.585269Z","iopub.status.idle":"2025-07-12T18:42:57.882253Z","shell.execute_reply.started":"2025-07-12T18:42:57.585250Z","shell.execute_reply":"2025-07-12T18:42:57.881525Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Time Series Analysis","metadata":{}},{"cell_type":"code","source":"\nplt.figure(figsize=(15,5))\ndf_train['label'].plot()\nplt.title('Target (label) over Time')\nplt.xlabel('Timestamp')\nplt.ylabel('label')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:42:57.883198Z","iopub.execute_input":"2025-07-12T18:42:57.883512Z","iopub.status.idle":"2025-07-12T18:42:59.198313Z","shell.execute_reply.started":"2025-07-12T18:42:57.883484Z","shell.execute_reply":"2025-07-12T18:42:59.197521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rolling_window = 60  # e.g., 60 minutes\nplt.figure(figsize=(15,5))\ndf_train['label'].rolling(window=rolling_window).mean().plot(label='Rolling Mean')\ndf_train['label'].rolling(window=rolling_window).std().plot(label='Rolling Std')\nplt.legend()\nplt.title(f'Rolling Mean and Std of label (window={rolling_window})')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:42:59.199235Z","iopub.execute_input":"2025-07-12T18:42:59.199445Z","iopub.status.idle":"2025-07-12T18:43:01.856704Z","shell.execute_reply.started":"2025-07-12T18:42:59.199427Z","shell.execute_reply":"2025-07-12T18:43:01.856034Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Outlier Detection","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,10))\nfor i, feature in enumerate(public_features, 1):\n    plt.subplot(2, 3, i)\n    sns.boxplot(x=df_train[feature])\n    plt.title(f'Boxplot of {feature}')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:43:01.857695Z","iopub.execute_input":"2025-07-12T18:43:01.858017Z","iopub.status.idle":"2025-07-12T18:43:02.783093Z","shell.execute_reply.started":"2025-07-12T18:43:01.857987Z","shell.execute_reply":"2025-07-12T18:43:02.782248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import zscore\n\nz_scores = zscore(df_train['volume'])\noutliers = df_train[np.abs(z_scores) > 3]\nprint(f\"Number of outliers in volume: {len(outliers)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:43:02.783880Z","iopub.execute_input":"2025-07-12T18:43:02.784123Z","iopub.status.idle":"2025-07-12T18:43:02.922952Z","shell.execute_reply.started":"2025-07-12T18:43:02.784103Z","shell.execute_reply":"2025-07-12T18:43:02.922186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.preprocessing import RobustScaler\n\n# Log-transform volume to reduce skewness\ndf_train['volume_log'] = np.log1p(df_train['volume'])\n\n# Create outlier flag based on z-score\nfrom scipy.stats import zscore\nz_scores = zscore(df_train['volume'])\ndf_train['volume_outlier'] = (np.abs(z_scores) > 3).astype(int)\n\n# Use RobustScaler for scaling\nscaler = RobustScaler()\ndf_train['volume_scaled'] = scaler.fit_transform(df_train[['volume']])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:43:02.923931Z","iopub.execute_input":"2025-07-12T18:43:02.924250Z","iopub.status.idle":"2025-07-12T18:43:03.023038Z","shell.execute_reply.started":"2025-07-12T18:43:02.924221Z","shell.execute_reply":"2025-07-12T18:43:03.022399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']:\n    df_train[f'{col}_log'] = np.log1p(df_train[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:43:03.023760Z","iopub.execute_input":"2025-07-12T18:43:03.024209Z","iopub.status.idle":"2025-07-12T18:43:03.041720Z","shell.execute_reply.started":"2025-07-12T18:43:03.024189Z","shell.execute_reply":"2025-07-12T18:43:03.041003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Scale the log-transformed features:\nscaler = RobustScaler()\nfor col in ['bid_qty_log', 'ask_qty_log', 'buy_qty_log', 'sell_qty_log', 'volume_log']:\n    df_train[f'{col}_scaled'] = scaler.fit_transform(df_train[[col]])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:43:03.042536Z","iopub.execute_input":"2025-07-12T18:43:03.042756Z","iopub.status.idle":"2025-07-12T18:43:03.176483Z","shell.execute_reply.started":"2025-07-12T18:43:03.042731Z","shell.execute_reply":"2025-07-12T18:43:03.175797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Log-transform\n\ndf_test['volume_log'] = np.log1p(df_test['volume'])\n\n# Step 2: Create outlier flag (same as before)\ntest_z_scores = (df_test['volume'] - df_train['volume'].mean()) / df_train['volume'].std()\ndf_test['volume_outlier'] = (np.abs(test_z_scores) > 3).astype(int)\n\n# Step 3: Apply scaler on the log-transformed volume (volume_log)\ndf_test['volume_scaled'] = scaler.transform(df_test[['volume_log']])\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:43:03.177256Z","iopub.execute_input":"2025-07-12T18:43:03.177575Z","iopub.status.idle":"2025-07-12T18:43:03.204515Z","shell.execute_reply.started":"2025-07-12T18:43:03.177547Z","shell.execute_reply":"2025-07-12T18:43:03.203736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"window_sizes = [5, 15, 60]  # Example windows in minutes\n\nfor feature in ['volume', 'bid_qty', 'ask_qty', 'buy_qty', 'sell_qty']:\n    for window in window_sizes:\n        df_train[f'{feature}_rollmean_{window}'] = df_train[feature].rolling(window).mean()\n        df_train[f'{feature}_rollstd_{window}'] = df_train[feature].rolling(window).std()\n        df_train[f'{feature}_lag_{window}'] = df_train[feature].shift(window)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:43:03.205316Z","iopub.execute_input":"2025-07-12T18:43:03.205593Z","iopub.status.idle":"2025-07-12T18:43:03.712084Z","shell.execute_reply.started":"2025-07-12T18:43:03.205569Z","shell.execute_reply":"2025-07-12T18:43:03.711250Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['label_rollstd_60'] = df_train['label'].rolling(60).std()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:43:03.712940Z","iopub.execute_input":"2025-07-12T18:43:03.713240Z","iopub.status.idle":"2025-07-12T18:43:03.738805Z","shell.execute_reply.started":"2025-07-12T18:43:03.713212Z","shell.execute_reply":"2025-07-12T18:43:03.738014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_cols = [col for col in df_train.columns if col not in ['label', 'timestamp', 'ID']]\n\n# Filter to columns existing in both train and test\nfeature_cols = [col for col in feature_cols if col in df_test.columns]\n\nX = df_train[feature_cols]\ny = df_train['label']\nX_test = df_test[feature_cols]\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:53:33.118050Z","iopub.execute_input":"2025-07-12T18:53:33.118841Z","iopub.status.idle":"2025-07-12T18:53:35.258876Z","shell.execute_reply.started":"2025-07-12T18:53:33.118816Z","shell.execute_reply":"2025-07-12T18:53:35.258256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"split_idx = int(len(df_train) * 0.8)\nX_train, X_val = X.iloc[:split_idx], X.iloc[split_idx:]\ny_train, y_val = y.iloc[:split_idx], y.iloc[split_idx:]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:53:38.786091Z","iopub.execute_input":"2025-07-12T18:53:38.786376Z","iopub.status.idle":"2025-07-12T18:53:38.790820Z","shell.execute_reply.started":"2025-07-12T18:53:38.786352Z","shell.execute_reply":"2025-07-12T18:53:38.790037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\nmodel = lgb.LGBMRegressor()\nmodel.fit(X_train, y_train)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:53:41.726734Z","iopub.execute_input":"2025-07-12T18:53:41.727447Z","iopub.status.idle":"2025-07-12T18:55:18.066955Z","shell.execute_reply.started":"2025-07-12T18:53:41.727422Z","shell.execute_reply":"2025-07-12T18:55:18.066239Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import pearsonr\ny_pred = model.predict(X_val)\ncorr, _ = pearsonr(y_val, y_pred)\nprint(\"Validation Pearson correlation:\", corr)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:58:07.093636Z","iopub.execute_input":"2025-07-12T18:58:07.094846Z","iopub.status.idle":"2025-07-12T18:58:07.931275Z","shell.execute_reply.started":"2025-07-12T18:58:07.094814Z","shell.execute_reply":"2025-07-12T18:58:07.930411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.fit(X, y)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:58:13.166894Z","iopub.execute_input":"2025-07-12T18:58:13.167219Z","iopub.status.idle":"2025-07-12T19:00:00.584817Z","shell.execute_reply.started":"2025-07-12T18:58:13.167195Z","shell.execute_reply":"2025-07-12T19:00:00.584111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_preds = model.predict(X_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T19:00:09.764084Z","iopub.execute_input":"2025-07-12T19:00:09.764374Z","iopub.status.idle":"2025-07-12T19:00:15.340469Z","shell.execute_reply.started":"2025-07-12T19:00:09.764353Z","shell.execute_reply":"2025-07-12T19:00:15.339813Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission File","metadata":{}},{"cell_type":"code","source":"submission['label'] = test_preds\nsubmission.to_csv('submission.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T19:00:27.044623Z","iopub.execute_input":"2025-07-12T19:00:27.045254Z","iopub.status.idle":"2025-07-12T19:00:28.809765Z","shell.execute_reply.started":"2025-07-12T19:00:27.045227Z","shell.execute_reply":"2025-07-12T19:00:28.809189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}