{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"cd92c88e","cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.decomposition import PCA\nfrom sklearn.manifold import TSNE\nfrom sklearn.feature_selection import SelectKBest, f_regression, RFE\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.cluster import KMeans\nfrom sklearn.preprocessing import StandardScaler\nimport os \nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T17:19:20.112044Z","iopub.execute_input":"2025-06-03T17:19:20.112856Z","iopub.status.idle":"2025-06-03T17:19:20.118015Z","shell.execute_reply.started":"2025-06-03T17:19:20.112792Z","shell.execute_reply":"2025-06-03T17:19:20.117080Z"}},"outputs":[],"execution_count":null},{"id":"d8222bd1","cell_type":"markdown","source":"## Initialize data and add clustering as additional features","metadata":{}},{"id":"0e4f902b","cell_type":"code","source":"# Load the data\nprint(\"Loading training data...\")\ndir = os.getcwd()\nprint(dir)\ntrain_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\nprint(f\"Training data info: {train_df.info}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T17:19:24.236779Z","iopub.execute_input":"2025-06-03T17:19:24.237466Z","iopub.status.idle":"2025-06-03T17:19:30.803749Z","shell.execute_reply.started":"2025-06-03T17:19:24.237438Z","shell.execute_reply":"2025-06-03T17:19:30.802853Z"}},"outputs":[],"execution_count":null},{"id":"a0db4752","cell_type":"code","source":"spread = train_df['bid_qty'] - train_df['ask_qty']\nplt.figure(figsize = (20, 20))\nplt.subplot(3, 1, 1)\nplt.plot(spread)\nplt.subplot(3, 1, 2)\nplt.plot(train_df['label'])\nsns.displot(train_df['label'], kind='kde')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T20:38:39.499826Z","iopub.status.idle":"2025-06-02T20:38:39.500240Z","shell.execute_reply.started":"2025-06-02T20:38:39.500038Z","shell.execute_reply":"2025-06-02T20:38:39.500057Z"}},"outputs":[],"execution_count":null},{"id":"6cbb69a0","cell_type":"code","source":"common_features = ['X175', 'X179', 'X137', 'X197', 'X22', 'X40', 'X181', 'X28', 'X169', 'X198', 'X173', 'X21'] #taken from https://www.kaggle.com/code/ahsuna123/anonymized-features-importance-selection-eda\ncorr = train_df[common_features].corr()\nfig = plt.figure(figsize=(20, 5))\nplt.subplot(1,3,1)\nplt.title(\"correlation of common eda features\")\nsns.heatmap(corr)\nplt.subplot(1,3,2) \nsns.heatmap(train_df[common_features].cov())\nplt.title(\"covariance of common eda features\")\nplt.subplot(1,3,3)\nplt.bar(common_features, height=train_df[common_features].var(axis=0))\nplt.title('variance of common eda features')\nplt.xticks(rotation=45)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T17:19:30.805016Z","iopub.execute_input":"2025-06-03T17:19:30.805260Z","iopub.status.idle":"2025-06-03T17:19:32.175661Z","shell.execute_reply.started":"2025-06-03T17:19:30.805240Z","shell.execute_reply":"2025-06-03T17:19:32.174849Z"}},"outputs":[],"execution_count":null},{"id":"c85a8e42","cell_type":"markdown","source":"## Testing for stationarity in data","metadata":{}},{"id":"bf38124b","cell_type":"code","source":"from statsmodels.tsa.stattools import adfuller\nseason_length = len(train_df['label'])/4\nvalues = [train_df.reset_index().loc[x*season_length:season_length+x*season_length, 'label'].values for x in range(4)]\nprint(values)\nfor i in values:\n    res = adfuller(i)\n    \n    # Printing the statistical result of the adfuller test\n    print('Augmneted Dickey_fuller Statistic: %f' % res[0])\n    print('p-value: %f' % res[1])\n    \n    # printing the critical values at different alpha levels.\n    print('critical values at different levels:')\n    for k, v in res[4].items():\n        print('\\t%s: %.3f' % (k, v))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T03:26:11.256555Z","iopub.execute_input":"2025-06-03T03:26:11.257143Z","iopub.status.idle":"2025-06-03T03:28:19.252335Z","shell.execute_reply.started":"2025-06-03T03:26:11.257109Z","shell.execute_reply":"2025-06-03T03:28:19.250713Z"}},"outputs":[],"execution_count":null},{"id":"8dc733e6-af29-4444-8f42-2a590f00f37a","cell_type":"markdown","source":"# Training a baseline model with common features","metadata":{}},{"id":"98fba045-db11-4fb0-b218-60fab9357de5","cell_type":"code","source":"import xgboost as xgb\nfrom sklearn.model_selection import GridSearchCV\nimport multiprocessing\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import TimeSeriesSplit\nxgb_label = train_df['label']\nxgb_data = train_df.drop('label', axis=1)[common_features, 'bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(xgb_train)\nxgb_model = xgb.XGBRegressor(\n        n_jobs=multiprocessing.cpu_count() // 2, tree_method=\"hist\", objective=\"reg:squarederror\", \n    )\ntscv = TimeSeriesSplit()\nfor i, (train_index, test_index) in enumerate(tscv.split(X_scaled)):\n    print(f\"Fold {i}:\")\n    print(f\"  Train: index={train_index}\")\n    print(f\"  Test:  index={test_index}\")    \n    xgb_model.fit(X_scaled.iloc[train_index], xgb_label.iloc[train_index])\n    prediction = xgb_model.predict(X_scaled.iloc[test_index])\n    print(f\"  Score: {np.corrcoef(prediction, xgb_label.iloc[test_index].to_numpy())}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"dc717703-db91-4d40-b2d4-8ff64dbc5e60","cell_type":"markdown","source":"# Clustering and analysis\n* PCA for clustering\n* Creating new features from clusters","metadata":{}},{"id":"58d05086-798d-4894-b2ce-a959ee831ee0","cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import IncrementalPCA\nimport matplotlib.pyplot as plt\n\nprint(\"\\n\" + \"=\"*50)\nprint(\"3. DIMENSIONALITY REDUCTION\")\nprint(\"=\"*50)\n\n# 1. Use only numeric columns and limit to 200 features\nX_numeric = X_features.select_dtypes(include=[np.number]).iloc[:, :200]\n\n# 2. Fill missing values with column median\nX_clean = X_numeric.fillna(X_numeric.median())\n\n# 3. Downcast to float32 to save memory\nX_clean = X_clean.astype(np.float32)\n\n# 4. Standardize features\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X_clean)\n\n# 5. Perform Incremental PCA\nprint(\"Performing Incremental PCA...\")\nipca = IncrementalPCA(n_components=50, batch_size=200)\nX_pca = ipca.fit_transform(X_scaled)\n\n# 6. Explained variance analysis\nexplained_variance_ratio = ipca.explained_variance_ratio_\ncumsum_variance = np.cumsum(explained_variance_ratio)\nn_components_90 = np.argmax(cumsum_variance >= 0.90) + 1\nn_components_95 = np.argmax(cumsum_variance >= 0.95) + 1\n\nprint(f\"Number of components needed for 90% variance: {n_components_90}\")\nprint(f\"Number of components needed for 95% variance: {n_components_95}\")\n\n# 7. Plot explained variance\nplt.figure(figsize=(12, 5))\n\nplt.subplot(1, 2, 1)\nplt.plot(range(1, len(explained_variance_ratio) + 1),\n         explained_variance_ratio, 'bo-')\nplt.xlabel('Principal Component')\nplt.ylabel('Explained Variance Ratio')\nplt.title('PCA - Individual Explained Variance')\nplt.grid(True)\n\nplt.subplot(1, 2, 2)\nplt.plot(range(1, len(cumsum_variance) + 1),\n         cumsum_variance, 'ro-')\nplt.axhline(y=0.90, color='g', linestyle='--', label='90% variance')\nplt.axhline(y=0.95, color='b', linestyle='--', label='95% variance')\nplt.xlabel('Number of Components')\nplt.ylabel('Cumulative Explained Variance')\nplt.title('PCA - Cumulative Explained Variance')\nplt.legend()\nplt.grid(True)\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"b4f32942","cell_type":"markdown","source":"## Boosted Tree Feature Selection","metadata":{}},{"id":"d71bffbb","cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}