{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.cross_decomposition import PLSRegression as PLS","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:17:04.550040Z","iopub.execute_input":"2025-08-18T00:17:04.550469Z","iopub.status.idle":"2025-08-18T00:17:04.557785Z","shell.execute_reply.started":"2025-08-18T00:17:04.550440Z","shell.execute_reply":"2025-08-18T00:17:04.555532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/train.parquet\")\ntest = pd.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/test.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:17:04.560136Z","iopub.execute_input":"2025-08-18T00:17:04.560487Z","iopub.status.idle":"2025-08-18T00:17:27.474572Z","shell.execute_reply.started":"2025-08-18T00:17:04.560462Z","shell.execute_reply":"2025-08-18T00:17:27.473480Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" def transform_df(df, n=6):\n    x=[i for i in df.columns if i not in [\"label\"]]#\"bid_qty\",\"ask_qty\", \"buy_qty\",\"sell_qty\", \"volume\", \n    temp=df[x]\n    pls=PLS(n_components=n)\n    xpls=pls.fit_transform(temp, df[\"label\"])\n    df1=pd.DataFrame()\n    df1[[f\"c{i}\" for i in range(1, n+1)]]=xpls[0]\n    df1=pd.concat([df1, df.reset_index()[[\"bid_qty\",\"ask_qty\", \"buy_qty\",\"sell_qty\", \"volume\",\"label\"]]], axis=1)\n    return df1, pls\n    \ndef transform_test(df, pls, n=6):\n    x=[i for i in df.columns if i not in [\"label\"]]\n    temp=df[x]\n    xpls=pls.transform(temp)\n    print(xpls)\n    df1=pd.DataFrame()\n    df1[[f\"c{i}\" for i in range(1, n+1)]]=xpls\n    df1=pd.concat([df1, df.reset_index()[[\"bid_qty\",\"ask_qty\", \"sell_qty\", \"buy_qty\",\"volume\"]]], axis=1)\n    return df1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:17:27.475665Z","iopub.execute_input":"2025-08-18T00:17:27.476037Z","iopub.status.idle":"2025-08-18T00:17:27.486619Z","shell.execute_reply.started":"2025-08-18T00:17:27.476007Z","shell.execute_reply":"2025-08-18T00:17:27.485387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:17:27.488873Z","iopub.execute_input":"2025-08-18T00:17:27.489202Z","iopub.status.idle":"2025-08-18T00:17:27.545700Z","shell.execute_reply.started":"2025-08-18T00:17:27.489174Z","shell.execute_reply":"2025-08-18T00:17:27.543703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"temp, pls=transform_df(train, n=7)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:17:27.546548Z","iopub.execute_input":"2025-08-18T00:17:27.546934Z","iopub.status.idle":"2025-08-18T00:18:47.348896Z","shell.execute_reply.started":"2025-08-18T00:17:27.546889Z","shell.execute_reply":"2025-08-18T00:18:47.343514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best=sorted(list(zip(abs(pls.coef_), [i for i in train.columns if i not in [\"label\"]])), key=lambda x: x[0], reverse=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:18:47.353741Z","iopub.execute_input":"2025-08-18T00:18:47.354829Z","iopub.status.idle":"2025-08-18T00:18:47.376788Z","shell.execute_reply.started":"2025-08-18T00:18:47.354772Z","shell.execute_reply":"2025-08-18T00:18:47.375155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#features by weights from PLS\nbest_sel=[i[1] for i in best[:25]]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:18:47.378180Z","iopub.execute_input":"2025-08-18T00:18:47.378566Z","iopub.status.idle":"2025-08-18T00:18:47.420700Z","shell.execute_reply.started":"2025-08-18T00:18:47.378537Z","shell.execute_reply":"2025-08-18T00:18:47.419155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_sel","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:18:47.422554Z","iopub.execute_input":"2025-08-18T00:18:47.422993Z","iopub.status.idle":"2025-08-18T00:18:47.461151Z","shell.execute_reply.started":"2025-08-18T00:18:47.422956Z","shell.execute_reply":"2025-08-18T00:18:47.459708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_features=best_sel+[\"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\", \"bid_qty\", \"label\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:18:47.462627Z","iopub.execute_input":"2025-08-18T00:18:47.462962Z","iopub.status.idle":"2025-08-18T00:18:47.496581Z","shell.execute_reply.started":"2025-08-18T00:18:47.462932Z","shell.execute_reply":"2025-08-18T00:18:47.495118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"weights=[i[1] for i in best[:25]]+[1,1,1,1,1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:18:47.502640Z","iopub.execute_input":"2025-08-18T00:18:47.503052Z","iopub.status.idle":"2025-08-18T00:18:47.530358Z","shell.execute_reply.started":"2025-08-18T00:18:47.503015Z","shell.execute_reply":"2025-08-18T00:18:47.528880Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train[base_features]\ntest = test[base_features]\ntrain","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:18:47.531524Z","iopub.execute_input":"2025-08-18T00:18:47.531849Z","iopub.status.idle":"2025-08-18T00:18:48.030892Z","shell.execute_reply.started":"2025-08-18T00:18:47.531824Z","shell.execute_reply":"2025-08-18T00:18:48.029632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def add_features(df):\n    # Original features\n    df['bid_ask_interaction'] = df['bid_qty'] * df['ask_qty']\n    df['bid_buy_interaction'] = df['bid_qty'] * df['buy_qty']\n    df['bid_sell_interaction'] = df['bid_qty'] * df['sell_qty']\n    df['ask_buy_interaction'] = df['ask_qty'] * df['buy_qty']\n    df['ask_sell_interaction'] = df['ask_qty'] * df['sell_qty']\n\n    df['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-10)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-10)\n    df['log_volume'] = np.log1p(df['volume'])\n\n    df['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-10)\n    df['bid_ask_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['liquidity_ratio'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'] + 1e-10)\n    \n    # === NEW MICROSTRUCTURE FEATURES ===\n    \n    # Price Pressure Indicators\n    df['net_order_flow'] = df['buy_qty'] - df['sell_qty']\n    df['normalized_net_flow'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['buying_pressure'] = df['buy_qty'] / (df['volume'] + 1e-10)\n    df['volume_weighted_buy'] = df['buy_qty'] * df['volume']\n    \n    # Liquidity Depth Measures\n    df['total_depth'] = df['bid_qty'] + df['ask_qty']\n    df['depth_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['relative_spread'] = np.abs(df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['log_depth'] = np.log1p(df['total_depth'])\n    \n    # Order Flow Toxicity Proxies\n    df['kyle_lambda'] = np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['flow_toxicity'] = np.abs(df['order_flow_imbalance']) * df['volume']\n    df['aggressive_flow_ratio'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    \n    # Market Activity Indicators\n   # df['volume_depth_ratio'] = df['volume'] / (df['total_depth'] + 1e-10)\n    df['activity_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + 1e-10)\n    df['log_buy_qty'] = np.log1p(df['buy_qty'])\n    df['log_sell_qty'] = np.log1p(df['sell_qty'])\n    df['log_bid_qty'] = np.log1p(df['bid_qty'])\n    df['log_ask_qty'] = np.log1p(df['ask_qty'])\n    \n    # Microstructure Volatility Proxies\n    df['realized_spread_proxy'] = 2 * np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['price_impact_proxy'] = df['net_order_flow'] / (df['total_depth'] + 1e-10)\n    df['quote_volatility_proxy'] = np.abs(df['depth_imbalance'])\n    \n    # Complex Interaction Terms\n  #  df['flow_depth_interaction'] = df['net_order_flow'] * df['total_depth']\n  #  df['imbalance_volume_interaction'] = df['order_flow_imbalance'] * df['volume']\n #   df['depth_volume_interaction'] = df['total_depth'] * df['volume']\n    df['buy_sell_spread'] = np.abs(df['buy_qty'] - df['sell_qty'])\n    df['bid_ask_spread'] = np.abs(df['bid_qty'] - df['ask_qty'])\n    \n    # Information Asymmetry Measures\n   # df['trade_informativeness'] = df['net_order_flow'] / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    #df['execution_shortfall_proxy'] = df['buy_sell_spread'] / (df['volume'] + 1e-10)\n    #df['adverse_selection_proxy'] = df['net_order_flow'] / (df['total_depth'] + 1e-10) * df['volume']\n    \n    # Market Efficiency Indicators\n    df['fill_probability'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n   # df['execution_rate'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n   # df['market_efficiency'] = df['volume'] / (df['bid_ask_spread'] + 1e-10)\n    \n    # Non-linear Transformations\n    #df['sqrt_volume'] = np.sqrt(df['volume'])\n   # df['sqrt_depth'] = np.sqrt(df['total_depth'])\n   # df['volume_squared'] = df['volume'] ** 2\n   # df['imbalance_squared'] = df['order_flow_imbalance'] ** 2\n    \n    # Relative Measures\n    df['bid_ratio'] = df['bid_qty'] / (df['total_depth'] + 1e-10)\n    df['ask_ratio'] = df['ask_qty'] / (df['total_depth'] + 1e-10)\n    df['buy_ratio'] = df['buy_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['sell_ratio'] = df['sell_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    \n    # Market Stress Indicators\n   # df['liquidity_consumption'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n   # df['market_stress'] = df['volume'] / (df['total_depth'] + 1e-10) * np.abs(df['order_flow_imbalance'])\n    df['depth_depletion'] = df['volume'] / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    \n    # Directional Indicators\n   ## df['net_buying_ratio'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    #df['directional_volume'] = df['net_order_flow'] * np.log1p(df['volume'])\n   # df['signed_volume'] = np.sign(df['net_order_flow']) * df['volume']\n    \n    # Replace infinities and NaNs\n    df = df.replace([np.inf, -np.inf], 0).fillna(0)\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:18:48.031976Z","iopub.execute_input":"2025-08-18T00:18:48.032273Z","iopub.status.idle":"2025-08-18T00:18:48.055637Z","shell.execute_reply.started":"2025-08-18T00:18:48.032250Z","shell.execute_reply":"2025-08-18T00:18:48.054391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#x_train = add_features(train)\nx_train=train\nx_train.shape\n\nimport warnings \n\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:18:48.056641Z","iopub.execute_input":"2025-08-18T00:18:48.056964Z","iopub.status.idle":"2025-08-18T00:18:48.081565Z","shell.execute_reply.started":"2025-08-18T00:18:48.056932Z","shell.execute_reply":"2025-08-18T00:18:48.080364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#scaler = StandardScaler()\n#x_train.iloc[:, :] = scaler.fit_transform(X_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:18:48.082604Z","iopub.execute_input":"2025-08-18T00:18:48.082905Z","iopub.status.idle":"2025-08-18T00:18:48.104803Z","shell.execute_reply.started":"2025-08-18T00:18:48.082873Z","shell.execute_reply":"2025-08-18T00:18:48.103647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#x_test = add_features(test)\nx_test=test\nx_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:18:48.106228Z","iopub.execute_input":"2025-08-18T00:18:48.106620Z","iopub.status.idle":"2025-08-18T00:18:48.131973Z","shell.execute_reply.started":"2025-08-18T00:18:48.106595Z","shell.execute_reply":"2025-08-18T00:18:48.130596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ensure the index is a datetime type\nx_train.index = pd.to_datetime(x_train.index)\n\n# Sort the dataset by timestamp (index)\nx_train = x_train.sort_index()\n\n# Define the split point (80% train, 20% validation)\nsplit_index = int(len(x_train) * 0.8)\n\n# Slice by index — NO shuffling\ntrain_split = x_train.iloc[:split_index]\nval_split = x_train.iloc[split_index:]\n\n# Training features and labels\nX_train = train_split.drop(columns=['label'])\ny_train = train_split['label']\n\n# Validation features and labels\nX_val = val_split.drop(columns=['label'])\ny_val = val_split['label']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:18:48.133482Z","iopub.execute_input":"2025-08-18T00:18:48.133948Z","iopub.status.idle":"2025-08-18T00:18:48.272670Z","shell.execute_reply.started":"2025-08-18T00:18:48.133910Z","shell.execute_reply":"2025-08-18T00:18:48.271370Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:18:48.274548Z","iopub.execute_input":"2025-08-18T00:18:48.274877Z","iopub.status.idle":"2025-08-18T00:18:48.283819Z","shell.execute_reply.started":"2025-08-18T00:18:48.274851Z","shell.execute_reply":"2025-08-18T00:18:48.282758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBRegressor\nfrom sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error\n\nXGB_PARAMS = {\n    \"tree_method\": \"hist\",\n    \"device\": \"gpu\",\n    \"colsample_bylevel\": 0.4778,\n    \"colsample_bynode\": 0.3628,\n    \"colsample_bytree\": 0.7107,\n    \"gamma\": 1.7095,\n    \"learning_rate\": 0.02213,\n    \"max_depth\": 20,\n    \"max_leaves\": 12,\n    \"min_child_weight\": 16,\n    \"n_estimators\": 1667,\n    \"subsample\": 0.06567,\n    \"reg_alpha\": 1,#39.3524,\n    \"reg_lambda\": 75.4484,\n    \"verbosity\": 0,\n    \"random_state\": 42,\n    \"n_jobs\": -1,\n    \"feature_weights\": weights\n}\n\nLEARNERS = [\n    {\"name\": \"xgb\", \"Estimator\": XGBRegressor, \"params\": XGB_PARAMS}\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:18:48.285214Z","iopub.execute_input":"2025-08-18T00:18:48.285551Z","iopub.status.idle":"2025-08-18T00:18:48.309927Z","shell.execute_reply.started":"2025-08-18T00:18:48.285520Z","shell.execute_reply":"2025-08-18T00:18:48.308192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBRegressor\nimport numpy as np\n\nmodel = XGBRegressor(**XGB_PARAMS)\nmodel.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:18:48.311470Z","iopub.execute_input":"2025-08-18T00:18:48.311811Z","iopub.status.idle":"2025-08-18T00:19:20.564319Z","shell.execute_reply.started":"2025-08-18T00:18:48.311778Z","shell.execute_reply":"2025-08-18T00:19:20.563279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_time_decay_weights(n: int, decay: float = 0.9) -> np.ndarray:\n    positions = np.arange(n)\n    normalized = positions / (n - 1)\n    weights = decay ** (1.0 - normalized)\n    return weights * n / weights.sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:19:20.565444Z","iopub.execute_input":"2025-08-18T00:19:20.565742Z","iopub.status.idle":"2025-08-18T00:19:20.572380Z","shell.execute_reply.started":"2025-08-18T00:19:20.565719Z","shell.execute_reply":"2025-08-18T00:19:20.570646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_val_pred = model.predict(X_val)\n\nfrom scipy.stats import pearsonr\n\n# Pearson correlation\npearson_corr, _ = pearsonr(y_val, y_val_pred)\n\nprint(f\"✅ Pearson Correlation: {pearson_corr:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:19:20.573610Z","iopub.execute_input":"2025-08-18T00:19:20.573896Z","iopub.status.idle":"2025-08-18T00:19:21.727613Z","shell.execute_reply.started":"2025-08-18T00:19:20.573874Z","shell.execute_reply":"2025-08-18T00:19:21.726111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = x_train.drop(columns=['label'])\ny = x_train['label']\nfeature_names = X.columns.tolist()  # ✅ Save exact feature names\n\nx_test = x_test[feature_names]  # ✅ Ensure same columns, same order\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:19:21.729154Z","iopub.execute_input":"2025-08-18T00:19:21.729581Z","iopub.status.idle":"2025-08-18T00:19:21.894270Z","shell.execute_reply.started":"2025-08-18T00:19:21.729544Z","shell.execute_reply":"2025-08-18T00:19:21.893100Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#x_test=x_test.drop(columns=['label'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:19:21.896009Z","iopub.execute_input":"2025-08-18T00:19:21.896429Z","iopub.status.idle":"2025-08-18T00:19:21.902686Z","shell.execute_reply.started":"2025-08-18T00:19:21.896401Z","shell.execute_reply":"2025-08-18T00:19:21.899984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(x_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:19:21.904293Z","iopub.execute_input":"2025-08-18T00:19:21.904709Z","iopub.status.idle":"2025-08-18T00:19:28.909597Z","shell.execute_reply.started":"2025-08-18T00:19:21.904675Z","shell.execute_reply":"2025-08-18T00:19:28.908811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.read_csv(\"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\")\nsubmission[\"prediction\"] = y_pred\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(\"📁 Submission file saved as 'submission.csv'\")\nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-18T00:19:28.910306Z","iopub.execute_input":"2025-08-18T00:19:28.910590Z","iopub.status.idle":"2025-08-18T00:19:30.238103Z","shell.execute_reply.started":"2025-08-18T00:19:28.910566Z","shell.execute_reply":"2025-08-18T00:19:30.236804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}}]}