{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-15T09:31:45.577749Z","iopub.execute_input":"2025-09-15T09:31:45.578326Z","iopub.status.idle":"2025-09-15T09:31:45.875005Z","shell.execute_reply.started":"2025-09-15T09:31:45.578301Z","shell.execute_reply":"2025-09-15T09:31:45.874409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np, pandas as pd\nfrom sklearn.feature_selection import mutual_info_regression\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.inspection import permutation_importance\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T09:31:45.876175Z","iopub.execute_input":"2025-09-15T09:31:45.876576Z","iopub.status.idle":"2025-09-15T09:31:47.365454Z","shell.execute_reply.started":"2025-09-15T09:31:45.876548Z","shell.execute_reply":"2025-09-15T09:31:47.364865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\ntest = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T09:31:47.366486Z","iopub.execute_input":"2025-09-15T09:31:47.36686Z","iopub.status.idle":"2025-09-15T09:32:32.125249Z","shell.execute_reply.started":"2025-09-15T09:31:47.366831Z","shell.execute_reply":"2025-09-15T09:32:32.124387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub   = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T10:04:36.687146Z","iopub.execute_input":"2025-09-15T10:04:36.687448Z","iopub.status.idle":"2025-09-15T10:04:37.058376Z","shell.execute_reply.started":"2025-09-15T10:04:36.687426Z","shell.execute_reply":"2025-09-15T10:04:37.057536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T19:16:53.54722Z","iopub.execute_input":"2025-09-14T19:16:53.547433Z","iopub.status.idle":"2025-09-14T19:16:53.568912Z","shell.execute_reply.started":"2025-09-14T19:16:53.547415Z","shell.execute_reply":"2025-09-14T19:16:53.568403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.shape)\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T19:16:53.569533Z","iopub.execute_input":"2025-09-14T19:16:53.569769Z","iopub.status.idle":"2025-09-14T19:16:53.593003Z","shell.execute_reply.started":"2025-09-14T19:16:53.569749Z","shell.execute_reply":"2025-09-14T19:16:53.592209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isna().sum().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T19:16:53.593708Z","iopub.execute_input":"2025-09-14T19:16:53.593949Z","iopub.status.idle":"2025-09-14T19:16:54.818821Z","shell.execute_reply.started":"2025-09-14T19:16:53.593927Z","shell.execute_reply":"2025-09-14T19:16:54.818112Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## we have almost 800 features, too much for modeling, take a lot of time.\n## first we have to reduce the dimension by eliminating the meaningless features","metadata":{}},{"cell_type":"code","source":"\"\"\"let's see how important are the known features \"\"\"\n\nfeat = [\"bid_qty\",\"ask_qty\",\"buy_qty\",\"sell_qty\",\"volume\"]\n\ndf = train[feat + [\"label\"]].dropna()\nX = df[feat].astype(np.float32)\ny = df[\"label\"].astype(np.float32)\n\n# univariate\n\npearson  = X.corrwith(y)\nspearman = X.corrwith(y, method=\"spearman\")\nmi = pd.Series(mutual_info_regression(X, y, random_state=42), index=feat)\n\n# model\nuse_rf = False\ntry:\n    from xgboost import XGBRegressor\n    model = XGBRegressor(\n        n_estimators=500, learning_rate=0.05, max_depth=6,\n        subsample=0.9, colsample_bytree=1.0, random_state=42,\n        tree_method=\"gpu_hist\", n_jobs=-1  # will auto-fallback below if no GPU\n    )\nexcept Exception:\n    use_rf = True\n\nXtr, Xte, ytr, yte = train_test_split(X, y, test_size=0.2, random_state=42)\n\nif use_rf:\n    from sklearn.ensemble import RandomForestRegressor\n    model = RandomForestRegressor(n_estimators=300, random_state=42, n_jobs=-1)\n\ntry:\n    model.fit(Xtr, ytr)\nexcept Exception:\n\n    if \"XGBRegressor\" in type(model).__name__:\n        model.set_params(tree_method=\"hist\")\n    model.fit(Xtr, ytr)\n\n\ntry:\n    booster = model.get_booster()\n    gain = booster.get_score(importance_type=\"gain\")\n    model_imp = pd.Series({f: gain.get(f\"f{i}\", 0.0) for i, f in enumerate(feat)}, index=feat)\nexcept Exception:\n    model_imp = pd.Series(getattr(model, \"feature_importances_\", np.zeros(len(feat))), index=feat)\n\n# permutation importance\nperm = permutation_importance(model, Xte, yte, n_repeats=5, random_state=42, n_jobs=-1)\nperm_imp = pd.Series(perm.importances_mean, index=feat)\n\n# SHAP (small subset, safe) \ntry:\n    import shap\n    bg = Xtr.sample(min(2000, len(Xtr)), random_state=42)\n    Xte_s = Xte.sample(min(8000, len(Xte)), random_state=42)\n    explainer = shap.Explainer(model, bg)\n    shap_vals = explainer(Xte_s)\n    shap_imp = pd.Series(np.abs(shap_vals.values).mean(axis=0), index=feat)\nexcept Exception:\n    shap_imp = pd.Series(np.zeros(len(feat)), index=feat)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T19:33:58.137645Z","iopub.execute_input":"2025-09-14T19:33:58.138353Z","iopub.status.idle":"2025-09-14T19:36:16.583732Z","shell.execute_reply.started":"2025-09-14T19:33:58.138325Z","shell.execute_reply":"2025-09-14T19:36:16.583177Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scores = pd.concat([\n    pearson.rename(\"pearson\"),\n    spearman.rename(\"spearman\"),\n    mi.rename(\"mutual_info\"),\n    model_imp.rename(\"model_importance\"),\n    perm_imp.rename(\"perm_importance\"),\n    shap_imp.rename(\"shap_importance\"),\n], axis=1).sort_values(\"shap_importance\", ascending=False)\n\nscores.round(6)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T19:56:14.77979Z","iopub.execute_input":"2025-09-14T19:56:14.780233Z","iopub.status.idle":"2025-09-14T19:56:14.795192Z","shell.execute_reply.started":"2025-09-14T19:56:14.7802Z","shell.execute_reply":"2025-09-14T19:56:14.794347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[\"order_imb\"]   = train[\"bid_qty\"] - train[\"ask_qty\"]\ntrain[\"trade_imb\"]   = train[\"buy_qty\"] - train[\"sell_qty\"]\ntrain[\"rel_ord_imb\"] = (train[\"bid_qty\"] - train[\"ask_qty\"]) / (train[\"bid_qty\"] + train[\"ask_qty\"] + 1e-9)\ntrain[\"rel_trd_imb\"] = (train[\"buy_qty\"] - train[\"sell_qty\"]) / (train[\"buy_qty\"] + train[\"sell_qty\"] + 1e-9)\ntrain[\"log_volume\"]  = np.log1p(train[\"volume\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T10:06:35.355657Z","iopub.execute_input":"2025-09-15T10:06:35.356336Z","iopub.status.idle":"2025-09-15T10:06:35.451324Z","shell.execute_reply.started":"2025-09-15T10:06:35.356313Z","shell.execute_reply":"2025-09-15T10:06:35.450353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"let's see how important are the known features \"\"\"\n\nfeat = [\"trade_imb\",\"order_imb\",\"rel_ord_imb\",\"rel_trd_imb\",\"log_volume\"]\n\ndf = train[feat + [\"label\"]].dropna()\nX = df[feat].astype(np.float32)\ny = df[\"label\"].astype(np.float32)\n\n# univariate\n\npearson  = X.corrwith(y)\nspearman = X.corrwith(y, method=\"spearman\")\nmi = pd.Series(mutual_info_regression(X, y, random_state=42), index=feat)\n\n# model\nuse_rf = False\ntry:\n    from xgboost import XGBRegressor\n    model = XGBRegressor(\n        n_estimators=500, learning_rate=0.05, max_depth=6,\n        subsample=0.9, colsample_bytree=1.0, random_state=42,\n        tree_method=\"gpu_hist\", n_jobs=-1  # will auto-fallback below if no GPU\n    )\nexcept Exception:\n    use_rf = True\n\nXtr, Xte, ytr, yte = train_test_split(X, y, test_size=0.2, random_state=42)\n\nif use_rf:\n    from sklearn.ensemble import RandomForestRegressor\n    model = RandomForestRegressor(n_estimators=300, random_state=42, n_jobs=-1)\n\ntry:\n    model.fit(Xtr, ytr)\nexcept Exception:\n\n    if \"XGBRegressor\" in type(model).__name__:\n        model.set_params(tree_method=\"hist\")\n    model.fit(Xtr, ytr)\n\n\ntry:\n    booster = model.get_booster()\n    gain = booster.get_score(importance_type=\"gain\")\n    model_imp = pd.Series({f: gain.get(f\"f{i}\", 0.0) for i, f in enumerate(feat)}, index=feat)\nexcept Exception:\n    model_imp = pd.Series(getattr(model, \"feature_importances_\", np.zeros(len(feat))), index=feat)\n\n# permutation importance\nperm = permutation_importance(model, Xte, yte, n_repeats=5, random_state=42, n_jobs=-1)\nperm_imp = pd.Series(perm.importances_mean, index=feat)\n\n# SHAP (small subset, safe) \ntry:\n    import shap\n    bg = Xtr.sample(min(2000, len(Xtr)), random_state=42)\n    Xte_s = Xte.sample(min(8000, len(Xte)), random_state=42)\n    explainer = shap.Explainer(model, bg)\n    shap_vals = explainer(Xte_s)\n    shap_imp = pd.Series(np.abs(shap_vals.values).mean(axis=0), index=feat)\nexcept Exception:\n    shap_imp = pd.Series(np.zeros(len(feat)), index=feat)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T19:59:40.771303Z","iopub.execute_input":"2025-09-14T19:59:40.77185Z","iopub.status.idle":"2025-09-14T20:02:01.475009Z","shell.execute_reply.started":"2025-09-14T19:59:40.771827Z","shell.execute_reply":"2025-09-14T20:02:01.474216Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scores = pd.concat([\n    pearson.rename(\"pearson\"),\n    spearman.rename(\"spearman\"),\n    mi.rename(\"mutual_info\"),\n    model_imp.rename(\"model_importance\"),\n    perm_imp.rename(\"perm_importance\"),\n    shap_imp.rename(\"shap_importance\"),\n], axis=1).sort_values(\"shap_importance\", ascending=False)\n\nscores.round(6)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T20:02:01.476322Z","iopub.execute_input":"2025-09-14T20:02:01.47664Z","iopub.status.idle":"2025-09-14T20:02:01.488369Z","shell.execute_reply.started":"2025-09-14T20:02:01.476614Z","shell.execute_reply":"2025-09-14T20:02:01.487686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"keep = [\"rel_ord_imb\", \"rel_trd_imb\",'log_volume']\ntrain_fs = train[keep + [\"label\"]].copy()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T10:06:39.931508Z","iopub.execute_input":"2025-09-15T10:06:39.931848Z","iopub.status.idle":"2025-09-15T10:06:39.954332Z","shell.execute_reply.started":"2025-09-15T10:06:39.931826Z","shell.execute_reply":"2025-09-15T10:06:39.953602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_fs.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T10:06:41.347555Z","iopub.execute_input":"2025-09-15T10:06:41.348161Z","iopub.status.idle":"2025-09-15T10:06:41.356806Z","shell.execute_reply.started":"2025-09-15T10:06:41.348138Z","shell.execute_reply":"2025-09-15T10:06:41.355982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train2 = train.drop(columns = [\"bid_qty\",\"ask_qty\",\"buy_qty\",\"sell_qty\",\"volume\"] + feat)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T20:06:45.587881Z","iopub.execute_input":"2025-09-14T20:06:45.588259Z","iopub.status.idle":"2025-09-14T20:06:47.179617Z","shell.execute_reply.started":"2025-09-14T20:06:45.588235Z","shell.execute_reply":"2025-09-14T20:06:47.178836Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style = 'color :red'> first appraoch : variance filtring </span>","metadata":{}},{"cell_type":"markdown","source":"* features with low varaince has not enough inforamtion to predict the target, we can assume that they are not important","metadata":{}},{"cell_type":"code","source":"# Variance filtering: drop (near-)constant features\nimport numpy as np\nfrom sklearn.feature_selection import VarianceThreshold\n\n# numeric features only, exclude target\nfeat_cols = train.select_dtypes(include=[np.number]).columns.drop('label', errors='ignore')\nX = train[feat_cols].astype(np.float32)\n\nvt = VarianceThreshold(threshold=1e-4)\nmask = vt.fit(X).get_support()\n\nX_var = X.loc[:, mask]                 # features kept\ncols_kept = X_var.columns.tolist()\ncols_drop = feat_cols[~mask].tolist()\n\nprint(f\"Kept {len(cols_kept)} / {len(feat_cols)} features. Dropped: {len(cols_drop)}\")\n# If you want the filtered training set with label:\ntrain_var = pd.concat([X_var, train['label']], axis=1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T19:56:58.90657Z","iopub.execute_input":"2025-09-14T19:56:58.906848Z","iopub.status.idle":"2025-09-14T19:57:13.74684Z","shell.execute_reply.started":"2025-09-14T19:56:58.906825Z","shell.execute_reply":"2025-09-14T19:57:13.746276Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style = 'color :red'> second approach : correlation filtring </span>","metadata":{}},{"cell_type":"markdown","source":"* We can't test the correlation of all features individually, so we propose to perform clustering in order to group correlated features into the same cluster. Then, we can select one or two features from each cluster. If the number of clusters is not too large (around 8 to 10), we could apply PCA within each cluster to leverage the information contained in them.","metadata":{}},{"cell_type":"code","source":"corr_thresh = 0.8          # group features with |corr| >= this\nn_corr_rows = 300_000       # rows used to compute correlation (sampling for safety)\nn_mi_rows   = 100_000        # rows used for MI (sampling for safety)\nrng = np.random.default_rng(42)\n\n\nX_all = train2.select_dtypes(include=[np.number]).drop(columns=['label'], errors='ignore')\ny_all = train2['label'].astype(np.float32).values\nX_all = X_all.astype(np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T20:11:29.378956Z","iopub.execute_input":"2025-09-14T20:11:29.379489Z","iopub.status.idle":"2025-09-14T20:11:37.518016Z","shell.execute_reply.started":"2025-09-14T20:11:29.379464Z","shell.execute_reply":"2025-09-14T20:11:37.517216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"idx_corr = rng.choice(len(X_all), min(len(X_all), n_corr_rows), replace=False)\nidx_mi   = rng.choice(len(X_all), min(len(X_all), n_mi_rows),   replace=False)\nX_corr = X_all.iloc[idx_corr]\nX_mi   = X_all.iloc[idx_mi]\ny_mi   = y_all[idx_mi]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T20:13:58.267232Z","iopub.execute_input":"2025-09-14T20:13:58.267507Z","iopub.status.idle":"2025-09-14T20:14:00.034951Z","shell.execute_reply.started":"2025-09-14T20:13:58.267487Z","shell.execute_reply":"2025-09-14T20:14:00.034396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corr = X_corr.corr().abs()\n\ntry:\n    from scipy.cluster.hierarchy import linkage, fcluster\n    from scipy.spatial.distance import squareform\n\n    dist = 1.0 - corr.values\n    np.fill_diagonal(dist, 0.0)\n    Z = linkage(squareform(dist, checks=False), method='average')\n    cluster_id = fcluster(Z, t=1.0 - corr_thresh, criterion='distance')\n\n    clusters = {}\n    for col, cid in zip(corr.columns, cluster_id):\n        clusters.setdefault(cid, []).append(col)\nexcept Exception:\n    clusters, used = [], set()\n    for f in corr.columns:\n        if f in used: \n            continue\n        grp = corr.index[corr[f] >= corr_thresh].tolist()\n        clusters.append(grp)\n        used.update(grp)\n    clusters = {i+1: g for i, g in enumerate(clusters)}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T20:14:06.47215Z","iopub.execute_input":"2025-09-14T20:14:06.472615Z","iopub.status.idle":"2025-09-14T20:21:00.339203Z","shell.execute_reply.started":"2025-09-14T20:14:06.472595Z","shell.execute_reply":"2025-09-14T20:21:00.338302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mi = pd.Series(mutual_info_regression(X_mi, y_mi, random_state=42), index=X_mi.columns).fillna(0.0)\n\nselected = []\nfor cols in clusters.values():\n    cols = [c for c in cols if c in mi.index]\n    if not cols:\n        continue\n    selected.append(mi[cols].idxmax())\n\nselected = sorted(set(selected), key=lambda c: mi[c], reverse=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T20:21:00.340487Z","iopub.execute_input":"2025-09-14T20:21:00.340766Z","iopub.status.idle":"2025-09-14T20:29:42.368742Z","shell.execute_reply.started":"2025-09-14T20:21:00.340739Z","shell.execute_reply":"2025-09-14T20:29:42.368187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"reduced_train = pd.concat([train2[selected], train2['label']], axis=1)\nprint(f\"Clusters: {len(clusters)} | Selected: {len(selected)} features\")\nprint(\"First 10 kept:\", selected[:10])\nprint(\"reduced_train shape:\", reduced_train.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T20:29:42.369409Z","iopub.execute_input":"2025-09-14T20:29:42.369591Z","iopub.status.idle":"2025-09-14T20:29:43.068836Z","shell.execute_reply.started":"2025-09-14T20:29:42.369577Z","shell.execute_reply":"2025-09-14T20:29:43.068266Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style = 'color :red'> third appraoch : cross validation </span>","metadata":{}},{"cell_type":"code","source":"from collections import Counter\nfrom sklearn.model_selection import KFold, StratifiedKFold\nfrom sklearn.preprocessing import LabelEncoder\nimport warnings","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T20:44:39.679858Z","iopub.execute_input":"2025-09-14T20:44:39.680173Z","iopub.status.idle":"2025-09-14T20:44:39.684087Z","shell.execute_reply.started":"2025-09-14T20:44:39.680152Z","shell.execute_reply":"2025-09-14T20:44:39.683388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"warnings.filterwarnings(\"ignore\")\n\ndf = reduced_train.copy()\nX = df.drop(columns=[\"label\"]).astype(np.float32)\ny_raw = df[\"label\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T20:44:40.395952Z","iopub.execute_input":"2025-09-14T20:44:40.396675Z","iopub.status.idle":"2025-09-14T20:44:43.336783Z","shell.execute_reply.started":"2025-09-14T20:44:40.396648Z","shell.execute_reply":"2025-09-14T20:44:43.33624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"uniq = pd.unique(y_raw)\nif y_raw.nunique() == 2:\n    task = \"binary\"\n    y = (y_raw.astype(\"category\").cat.codes).astype(np.int32)\nelif (pd.api.types.is_integer_dtype(y_raw) or np.allclose(uniq, uniq.astype(int))) and y_raw.nunique() <= 20:\n    task = \"multiclass\"\n    le = LabelEncoder()\n    y = le.fit_transform(y_raw).astype(np.int32)\nelse:\n    task = \"reg\"\n    y = y_raw.astype(np.float32).values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T20:44:51.41738Z","iopub.execute_input":"2025-09-14T20:44:51.417648Z","iopub.status.idle":"2025-09-14T20:44:51.481649Z","shell.execute_reply.started":"2025-09-14T20:44:51.417627Z","shell.execute_reply":"2025-09-14T20:44:51.481099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"use_xgb = True\ntry:\n    from xgboost import XGBClassifier, XGBRegressor\nexcept Exception:\n    use_xgb = False\n\ndef make_model():\n    if use_xgb:\n        common = dict(\n            n_estimators=2000, max_depth=6, learning_rate=0.05,\n            subsample=0.9, colsample_bytree=0.8, reg_lambda=1.0,\n            tree_method=\"gpu_hist\", n_jobs=-1, random_state=42\n        )\n        if task == \"reg\":\n            return XGBRegressor(**common, eval_metric=\"rmse\")\n        elif task == \"binary\":\n            return XGBClassifier(**common, eval_metric=\"logloss\", objective=\"binary:logistic\")\n        else:\n            return XGBClassifier(**common, eval_metric=\"mlogloss\", objective=\"multi:softprob\", num_class=y_raw.nunique())\n    else:\n        # Fallback: small RandomForest (CPU)\n        from sklearn.ensemble import RandomForestRegressor, RandomForestClassifier\n        if task == \"reg\":\n            return RandomForestRegressor(n_estimators=400, max_depth=None, n_jobs=-1, random_state=42)\n        else:\n            return RandomForestClassifier(n_estimators=500, max_depth=None, n_jobs=-1, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T20:45:01.185599Z","iopub.execute_input":"2025-09-14T20:45:01.185868Z","iopub.status.idle":"2025-09-14T20:45:01.192171Z","shell.execute_reply.started":"2025-09-14T20:45:01.185847Z","shell.execute_reply":"2025-09-14T20:45:01.191444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- fixes & imports ---\nfrom collections import Counter\nimport numpy as np\nimport gc\n\n# SHAP settings (keep memory safe)\nn_bg = 2000        # background rows for SHAP model/TreeExplainer\nn_eval = 40000     # rows to compute SHAP on (per fold)\ntop_k = 60\nmin_freq = 4\n\nfeature_counts = Counter()\ncols = X.columns.to_list()\n\n# CV splitter (unchanged)\nif task == \"reg\":\n    splitter = KFold(n_splits=6, shuffle=True, random_state=42)\nelse:\n    splitter = StratifiedKFold(n_splits=6, shuffle=True, random_state=42)\n\nfor fold, (tr_idx, va_idx) in enumerate(splitter.split(X, y), 1):\n    X_tr, X_va = X.iloc[tr_idx], X.iloc[va_idx]\n    y_tr, y_va = y[tr_idx], y[va_idx]\n\n    model = make_model()\n\n    # (optional) hint to use GPU predictor if it's XGBoost\n    try:\n        model.set_params(predictor=\"gpu_predictor\")\n    except Exception:\n        pass\n\n    # fit with early stopping if supported\n    try:\n        model.fit(\n            X_tr, y_tr,\n            eval_set=[(X_va, y_va)],\n            verbose=False,\n            early_stopping_rounds=100\n        )\n    except TypeError:\n        # models without early stopping\n        model.fit(X_tr, y_tr)\n\n    # ---- SHAP (subsampled) ----\n    explainer = None\n    sv = None\n    vals = None\n    imp = None\n    try:\n        import shap\n        bg = X_tr.sample(min(n_bg, len(X_tr)), random_state=42)\n        X_eval = X_va.sample(min(n_eval, len(X_va)), random_state=42)\n\n        # fast path; for tree models shap.Explainer picks TreeExplainer\n        explainer = shap.Explainer(model, bg)\n        sv = explainer(X_eval, check_additivity=False)\n\n        vals = sv.values\n        # handle multi-class shape\n        if isinstance(vals, list):\n            vals = np.stack(vals, axis=-1)  # (n_samples, n_features, n_classes)\n        if vals.ndim == 3:\n            vals = np.mean(np.abs(vals), axis=2)  # avg over classes\n        imp = np.mean(np.abs(vals), axis=0)       # mean |SHAP| across samples\n\n        top_idx = np.argsort(imp)[::-1][:top_k]\n        feature_counts.update([cols[i] for i in top_idx])\n\n    except Exception:\n        # SHAP failed -> use model feature importance as a safe backup\n        try:\n            imp = getattr(model, \"feature_importances_\", None)\n            if imp is None:\n                # XGBoost booster gain (only for XGB models)\n                booster = model.get_booster()\n                gain = booster.get_score(importance_type=\"gain\")\n                imp = np.array([gain.get(f\"f{i}\", 0.0) for i in range(len(cols))], dtype=np.float32)\n            top_idx = np.argsort(imp)[::-1][:top_k]\n            feature_counts.update([cols[i] for i in top_idx])\n        except Exception:\n            pass\n\n    # cleanup per fold\n    del model\n    try:\n        del explainer, sv, vals, imp\n    except Exception:\n        pass\n    gc.collect()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T22:34:16.154176Z","iopub.execute_input":"2025-09-14T22:34:16.154829Z","iopub.status.idle":"2025-09-15T05:14:51.008263Z","shell.execute_reply.started":"2025-09-14T22:34:16.154808Z","shell.execute_reply":"2025-09-15T05:14:51.007317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected = [f for f, c in feature_counts.items() if c >= min_freq]\nselected = sorted(selected, key=lambda f: (-feature_counts[f], f))\n\nreduced_train_cv = pd.concat([X[selected], y_raw], axis=1)\n\nprint(f\"Top-{top_k} per fold, kept if appearing ≥{min_freq}/6 folds.\")\nprint(f\"Selected {len(selected)} features out of {X.shape[1]}.\")\nprint(\"Most frequent (first 50):\", selected[:50])\nprint(\"reduced_train_cv shape:\", reduced_train_cv.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T05:14:51.009974Z","iopub.execute_input":"2025-09-15T05:14:51.010344Z","iopub.status.idle":"2025-09-15T05:14:51.116036Z","shell.execute_reply.started":"2025-09-15T05:14:51.010321Z","shell.execute_reply":"2025-09-15T05:14:51.115003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"topfeat =  ['X136', 'X137', 'X138', 'X169', 'X180', 'X198', 'X218', 'X22', 'X269', 'X28', 'X281', 'X296', 'X342', 'X343', 'X344', 'X36', 'X382', 'X385', 'X387', 'X425', 'X426', 'X427', 'X428', 'X445', 'X466', 'X501', 'X508', 'X582', 'X586', 'X587', 'X590', 'X591', 'X604', 'X611', 'X612', 'X613', 'X614', 'X616', 'X623', 'X626', 'X646', 'X731', 'X750', 'X751', 'X752', 'X756', 'X757', 'X758', 'X759', 'X762']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T09:33:16.430954Z","iopub.execute_input":"2025-09-15T09:33:16.431676Z","iopub.status.idle":"2025-09-15T09:33:16.436046Z","shell.execute_reply.started":"2025-09-15T09:33:16.431648Z","shell.execute_reply":"2025-09-15T09:33:16.435228Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = train[topfeat + ['label']]\ntest = test[topfeat]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T09:33:19.121201Z","iopub.execute_input":"2025-09-15T09:33:19.121488Z","iopub.status.idle":"2025-09-15T09:33:19.654596Z","shell.execute_reply.started":"2025-09-15T09:33:19.121462Z","shell.execute_reply":"2025-09-15T09:33:19.653781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T09:33:20.701367Z","iopub.execute_input":"2025-09-15T09:33:20.701995Z","iopub.status.idle":"2025-09-15T09:33:20.733573Z","shell.execute_reply.started":"2025-09-15T09:33:20.701973Z","shell.execute_reply":"2025-09-15T09:33:20.732772Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"****\n### quick test","metadata":{}},{"cell_type":"code","source":"import warnings, optuna, xgboost as xgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error\nfrom tqdm import tqdm\n\n# Silence warnings\nwarnings.filterwarnings(\"ignore\")\n\n# Train/val split\nX = data.drop(columns=[\"label\"]).astype(np.float32)\ny = data[\"label\"].astype(np.float32)\n\nX_train, X_valid, y_train, y_valid = train_test_split(\n    X, y, test_size=0.2, random_state=42\n)\n\n# Objective for Optuna\ndef objective(trial):\n    params = {\n        \"booster\": \"gbtree\",\n        \"tree_method\": \"gpu_hist\",   # force GPU\n        \"predictor\": \"gpu_predictor\",\n        \"objective\": \"reg:squarederror\",\n        \"eval_metric\": \"rmse\",\n        \"random_state\": 42,\n        \"max_depth\": trial.suggest_int(\"max_depth\", 4, 10),\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.01, 0.3, log=True),\n        \"subsample\": trial.suggest_float(\"subsample\", 0.6, 1.0),\n        \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.6, 1.0),\n        \"lambda\": trial.suggest_float(\"lambda\", 1e-3, 10.0, log=True),\n        \"alpha\": trial.suggest_float(\"alpha\", 1e-3, 10.0, log=True),\n        \"min_child_weight\": trial.suggest_int(\"min_child_weight\", 1, 20),\n        \"n_estimators\": 2000,\n    }\n\n    model = xgb.XGBRegressor(**params)\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_valid, y_valid)],\n        early_stopping_rounds=100,\n        verbose=False\n    )\n    preds = model.predict(X_valid)\n    rmse = mean_squared_error(y_valid, preds, squared=False)\n    return rmse\n\n# Run Optuna with tqdm\nn_trials = 30\npbar = tqdm(total=n_trials, desc=\"Optuna tuning\")\n\ndef callback(study, trial):\n    pbar.update(1)\n    pbar.set_postfix(rmse=trial.value)\n\nstudy = optuna.create_study(direction=\"minimize\")\nstudy.optimize(objective, n_trials=n_trials, callbacks=[callback])\npbar.close()\n\nprint(\"Best params:\", study.best_trial.params)\nprint(\"Best RMSE:\", study.best_value)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T09:33:23.100543Z","iopub.execute_input":"2025-09-15T09:33:23.101259Z","iopub.status.idle":"2025-09-15T09:55:49.159057Z","shell.execute_reply.started":"2025-09-15T09:33:23.101233Z","shell.execute_reply":"2025-09-15T09:55:49.158435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#  params: {'max_depth': 10, 'learning_rate': 0.04449241431926484, 'subsample': 0.7292951043347027, 'colsample_bytree': 0.9657451881115909, 'lambda': 0.13206388158422397, 'alpha': 0.015985557976794247, 'min_child_weight': 9}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train final model with best params\nbest_params = study.best_trial.params\nbest_params.update({\n    \"booster\": \"gbtree\",\n    \"tree_method\": \"gpu_hist\",\n    \"predictor\": \"gpu_predictor\",\n    \"objective\": \"reg:squarederror\",\n    \"eval_metric\": \"rmse\",\n    \"random_state\": 42,\n    \"n_estimators\": 2000\n})\n\nfinal_model = xgb.XGBRegressor(**best_params)\nfinal_model.fit(X, y, verbose=False)\n\n# Predict on test\nX_test = test.astype(np.float32)\npreds = final_model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T10:01:49.948568Z","iopub.execute_input":"2025-09-15T10:01:49.949322Z","iopub.status.idle":"2025-09-15T10:02:55.970931Z","shell.execute_reply.started":"2025-09-15T10:01:49.949296Z","shell.execute_reply":"2025-09-15T10:02:55.970092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Plot feature importance (gain by default)\nxgb.plot_importance(final_model, importance_type=\"gain\", height=0.6)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T10:09:36.593794Z","iopub.execute_input":"2025-09-15T10:09:36.594115Z","iopub.status.idle":"2025-09-15T10:09:37.185458Z","shell.execute_reply.started":"2025-09-15T10:09:36.594091Z","shell.execute_reply":"2025-09-15T10:09:37.184778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save submission\nsubmission = pd.DataFrame({\n    \"ID\": np.arange(1,len(preds)+1),  # replace with actual ID if available\n    \"prediction\": preds\n})\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(\"Saved submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T10:05:09.319704Z","iopub.execute_input":"2025-09-15T10:05:09.319984Z","iopub.status.idle":"2025-09-15T10:05:10.155597Z","shell.execute_reply.started":"2025-09-15T10:05:09.319963Z","shell.execute_reply":"2025-09-15T10:05:10.154917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data2 = pd.concat([train_fs, data], axis=1)\nX = data.drop(columns=[\"label\"]).astype(np.float32)\ny = data[\"label\"].astype(np.float32)\ntest2 = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\ntest2 = test2[X.columns]\n\n#Train final model with best params\nbest_params = study.best_trial.params\nbest_params.update({\n    \"booster\": \"gbtree\",\n    \"tree_method\": \"gpu_hist\",\n    \"predictor\": \"gpu_predictor\",\n    \"objective\": \"reg:squarederror\",\n    \"eval_metric\": \"rmse\",\n    \"random_state\": 42,\n    \"n_estimators\": 2000\n})\n\nfinal_model = xgb.XGBRegressor(**best_params)\nfinal_model.fit(X, y, verbose=False)\n\n# Predict on test\nX_test = test2.astype(np.float32)\npreds2 = final_model.predict(X_test)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T10:07:17.762227Z","iopub.execute_input":"2025-09-15T10:07:17.762495Z","iopub.status.idle":"2025-09-15T10:08:28.702815Z","shell.execute_reply.started":"2025-09-15T10:07:17.762473Z","shell.execute_reply":"2025-09-15T10:08:28.702057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save submission\nsubmission = pd.DataFrame({\n    \"ID\": np.arange(1,len(preds2)+1),  # replace with actual ID if available\n    \"prediction\": preds2\n})\nsubmission.to_csv(\"submission2.csv\", index=False)\nprint(\"Saved submission2.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T10:08:47.250742Z","iopub.execute_input":"2025-09-15T10:08:47.251494Z","iopub.status.idle":"2025-09-15T10:08:48.028043Z","shell.execute_reply.started":"2025-09-15T10:08:47.251459Z","shell.execute_reply":"2025-09-15T10:08:48.0272Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style = 'color : red'> feature engineering </span>","metadata":{}},{"cell_type":"markdown","source":"### Since the features are anonymized, domain-driven engineering (like order/trade imbalance before) isn’t possible. A solid strategy is:\n\n* Autoencoder (AE) to compress the anonymized features into a low-dimensional latent space (say 8–10 dimensions).\n\n* Concatenate latent features with the engineered features we already identified (rel_ord_imb, rel_trd_imb, maybe log_volume).\n","metadata":{}},{"cell_type":"code","source":"    # \"\"\"Swish activation function - prevents dead neurons and provides smooth gradients\"\"\"\n    #   \"\"\"Gaussian noise layer for data augmentation and overfitting prevention\"\"\"\n\n\"\"\"import torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader, TensorDataset\n\n# Prep data\n\nX = data.drop(columns=[\"label\"]).astype(np.float32).values\ny = data[\"label\"].values.astype(np.float32)\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# normalize\n\nfrom sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n\ndataset = TensorDataset(torch.tensor(X_scaled))\nloader = DataLoader(dataset, batch_size=1024, shuffle=False, drop_last=True) \n\n# Define Autoencoder\n\nlatent_dim = 10\n\nimport torch\nimport torch.nn as nn\n\n\nclass Swish(nn.Module):\n    def forward(self, x):\n        return x * torch.sigmoid(x)\n\nclass GaussianNoise(nn.Module):\n  \n    def __init__(self, std=0.05):\n        super(GaussianNoise, self).__init__()\n        self.std = std\n        \n    def forward(self, x):\n        if self.training:\n            noise = torch.randn_like(x) * self.std\n            return x + noise\n        return x\n        \nclass AutoEncoder(nn.Module):\n    def __init__(self, input_size, encoding_size=128, dropout=0.3):\n        super(AutoEncoder, self).__init__()\n        \n        # Encoder pathway\n        self.encoder = nn.Sequential(\n            nn.Linear(input_size, input_size // 2),\n            nn.BatchNorm1d(input_size // 2),\n            Swish(),\n            nn.Dropout(dropout),\n            \n            nn.Linear(input_size // 2, input_size // 4),\n            nn.BatchNorm1d(input_size // 4),\n            Swish(),\n            nn.Dropout(dropout),\n            \n            nn.Linear(input_size // 4, encoding_size),\n            nn.BatchNorm1d(encoding_size),\n            Swish()\n        )\n        \n        # Decoder pathway\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_size, input_size // 4),\n            nn.BatchNorm1d(input_size // 4),\n            Swish(),\n            nn.Dropout(dropout),\n            \n            nn.Linear(input_size // 4, input_size // 2),\n            nn.BatchNorm1d(input_size // 2),\n            Swish(),\n            nn.Dropout(dropout),\n            \n            nn.Linear(input_size // 2, input_size)\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return encoded, decoded\n\n\nmodel = AutoEncoder(X.shape[1], latent_dim).to(device)\nopt = torch.optim.Adam(model.parameters(), lr=1e-3)\nloss_fn = nn.MSELoss()\n\n\"\"\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"import torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader, TensorDataset\n\n# Prep data\n\nX = data.drop(columns=[\"label\"]).astype(np.float32).values\ny = data[\"label\"].values.astype(np.float32)\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# normalize\n\nfrom sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n\ndataset = TensorDataset(torch.tensor(X_scaled))\nloader = DataLoader(dataset, batch_size=1024, shuffle=False, drop_last=True) \n\n# Define Autoencoder\n\nlatent_dim = 10\n\nimport torch\nimport torch.nn as nn\n\nclass RobustAutoEncoder(nn.Module):\n    def __init__(self, input_dim, latent_dim, dropout=0.2):\n        super().__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, 256),\n            nn.BatchNorm1d(256),\n            nn.LeakyReLU(0.2),\n            nn.Dropout(dropout),\n\n            nn.Linear(256, 128),\n            nn.BatchNorm1d(128),\n            nn.LeakyReLU(0.2),\n            nn.Dropout(dropout),\n\n            nn.Linear(128, 64),\n            nn.BatchNorm1d(64),\n            nn.LeakyReLU(0.2),\n            nn.Dropout(dropout),\n\n            nn.Linear(64, latent_dim)   # latent code\n        )\n\n        self.decoder = nn.Sequential(\n            nn.Linear(latent_dim, 64),\n            nn.BatchNorm1d(64),\n            nn.LeakyReLU(0.2),\n\n            nn.Linear(64, 128),\n            nn.BatchNorm1d(128),\n            nn.LeakyReLU(0.2),\n\n            nn.Linear(128, 256),\n            nn.BatchNorm1d(256),\n            nn.LeakyReLU(0.2),\n\n            nn.Linear(256, input_dim)   # reconstruction\n        )\n\n        # init weights for stability\n        self.apply(self._init_weights)\n\n    def forward(self, x, noise_std=0.0):\n        if noise_std > 0:   # Denoising AE\n            x = x + noise_std * torch.randn_like(x)\n        z = self.encoder(x)\n        x_rec = self.decoder(z)\n        return z, x_rec\n\n    @staticmethod\n    def _init_weights(m):\n        if isinstance(m, nn.Linear):\n            nn.init.xavier_uniform_(m.weight)\n            if m.bias is not None:\n                nn.init.zeros_(m.bias)\n\n\nmodel = AutoEncoder(X.shape[1], latent_dim).to(device)\nopt = torch.optim.Adam(model.parameters(), lr=1e-3)\nloss_fn = nn.MSELoss()\n\n\"\"\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader, TensorDataset\n\n# Prep data\n\nX = data.drop(columns=[\"label\"]).astype(np.float32).values\ny = data[\"label\"].values.astype(np.float32)\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# normalize\n\nfrom sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n\ndataset = TensorDataset(torch.tensor(X_scaled))\nloader = DataLoader(dataset, batch_size=1024, shuffle=False, drop_last=True) \n\n# Define Autoencoder\n\nlatent_dim = 10\n\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, latent_dim):\n        super().__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, 256), nn.ReLU(),\n            nn.Linear(256, 128), nn.ReLU(),\n            nn.Linear(128, 64),nn.ReLU(),\n            nn.Linear(64,latent_dim)\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(latent_dim,64),nn.ReLU(),\n            nn.Linear(64, 128), nn.ReLU(),\n            nn.Linear(128, 256), nn.ReLU(),\n            nn.Linear(256, input_dim)\n        )\n\n    def forward(self, x):\n        z = self.encoder(x)\n        x_rec = self.decoder(z)\n        return z, x_rec\n\nmodel = AutoEncoder(X.shape[1], latent_dim).to(device)\nopt = torch.optim.Adam(model.parameters(), lr=1e-3)\nloss_fn = nn.MSELoss()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:17:05.251288Z","iopub.execute_input":"2025-09-15T11:17:05.251594Z","iopub.status.idle":"2025-09-15T11:17:05.877194Z","shell.execute_reply.started":"2025-09-15T11:17:05.251572Z","shell.execute_reply":"2025-09-15T11:17:05.876508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train AE\n\nepochs = 20\nfor epoch in range(epochs):\n    model.train()\n    total_loss = 0\n    for (xb,) in loader:\n        xb = xb.to(device)\n        z, xb_rec = model(xb)\n        loss = loss_fn(xb_rec, xb)\n        opt.zero_grad()\n        loss.backward()\n        opt.step()\n        total_loss += loss.item() * xb.size(0)\n    print(f\"Epoch {epoch+1}/{epochs}, loss {total_loss/len(loader.dataset):.6f}\")\n\n# Extract latent features\n\nmodel.eval()\nwith torch.no_grad():\n    Z_train = model.encoder(torch.tensor(X_scaled).to(device)).cpu().numpy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:17:06.607269Z","iopub.execute_input":"2025-09-15T11:17:06.607542Z","iopub.status.idle":"2025-09-15T11:18:56.0421Z","shell.execute_reply.started":"2025-09-15T11:17:06.607521Z","shell.execute_reply":"2025-09-15T11:18:56.041239Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Put into DataFrame\nlatent_cols = [f\"z{i}\" for i in range(latent_dim)]\nlatent_df = pd.DataFrame(Z_train, index=data.index, columns=latent_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:18:56.043524Z","iopub.execute_input":"2025-09-15T11:18:56.043863Z","iopub.status.idle":"2025-09-15T11:18:56.048167Z","shell.execute_reply.started":"2025-09-15T11:18:56.043844Z","shell.execute_reply":"2025-09-15T11:18:56.047421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Combine with engineered features\n\nengineered = train[[\"rel_ord_imb\", \"rel_trd_imb\", \"log_volume\"]]\nlatent = pd.concat([latent_df, engineered, train[\"label\"]], axis=1)\n\nprint(\"feature engineering shape:\", latent.shape)\nlatent.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:18:56.048973Z","iopub.execute_input":"2025-09-15T11:18:56.049188Z","iopub.status.idle":"2025-09-15T11:18:56.112674Z","shell.execute_reply.started":"2025-09-15T11:18:56.049172Z","shell.execute_reply":"2025-09-15T11:18:56.111963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = test.astype(np.float32).values\nX_test_scaled = scaler.transform(X_test)\n\nwith torch.no_grad():\n    Z_test = model.encoder(torch.tensor(X_test_scaled).to(device)).cpu().numpy()\n\nlatent_test_df = pd.DataFrame(Z_test, index=test.index, columns=latent_cols)\n\n# Concat with original test\ntest_with_latent = pd.concat([test.reset_index(drop=True), latent_test_df.reset_index(drop=True)], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:18:56.114524Z","iopub.execute_input":"2025-09-15T11:18:56.114764Z","iopub.status.idle":"2025-09-15T11:18:56.736951Z","shell.execute_reply.started":"2025-09-15T11:18:56.114747Z","shell.execute_reply":"2025-09-15T11:18:56.736362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"te  = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T10:40:51.986353Z","iopub.execute_input":"2025-09-15T10:40:51.986627Z","iopub.status.idle":"2025-09-15T10:41:02.560885Z","shell.execute_reply.started":"2025-09-15T10:40:51.986604Z","shell.execute_reply":"2025-09-15T10:41:02.540271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_with_latent[\"rel_ord_imb\"] = (te[\"bid_qty\"] - te[\"ask_qty\"]) / (te[\"bid_qty\"] + te[\"ask_qty\"] + 1e-9)\ntest_with_latent[\"rel_trd_imb\"] = (te[\"buy_qty\"] - te[\"sell_qty\"]) / (te[\"buy_qty\"] + te[\"sell_qty\"] + 1e-9)\ntest_with_latent[\"log_volume\"]  = np.log1p(te[\"volume\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:18:56.737748Z","iopub.execute_input":"2025-09-15T11:18:56.737991Z","iopub.status.idle":"2025-09-15T11:18:56.796777Z","shell.execute_reply.started":"2025-09-15T11:18:56.73797Z","shell.execute_reply":"2025-09-15T11:18:56.795878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_with_latent = pd.concat([data.reset_index(drop=True), latent.drop(columns ='label').reset_index(drop=True)], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:18:56.797613Z","iopub.execute_input":"2025-09-15T11:18:56.797851Z","iopub.status.idle":"2025-09-15T11:18:56.988845Z","shell.execute_reply.started":"2025-09-15T11:18:56.797832Z","shell.execute_reply":"2025-09-15T11:18:56.988164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_with_latent.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:18:56.989767Z","iopub.execute_input":"2025-09-15T11:18:56.990115Z","iopub.status.idle":"2025-09-15T11:18:56.995968Z","shell.execute_reply.started":"2025-09-15T11:18:56.990082Z","shell.execute_reply":"2025-09-15T11:18:56.994878Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"***","metadata":{}},{"cell_type":"code","source":"train_with_latent.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:18:56.996789Z","iopub.execute_input":"2025-09-15T11:18:56.997124Z","iopub.status.idle":"2025-09-15T11:18:57.013848Z","shell.execute_reply.started":"2025-09-15T11:18:56.997102Z","shell.execute_reply":"2025-09-15T11:18:57.013082Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_with_latent.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:18:57.014758Z","iopub.execute_input":"2025-09-15T11:18:57.015241Z","iopub.status.idle":"2025-09-15T11:18:57.030115Z","shell.execute_reply.started":"2025-09-15T11:18:57.015214Z","shell.execute_reply":"2025-09-15T11:18:57.029254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = train_with_latent.iloc[int(len(train_with_latent) * 0.4):].reset_index(drop=True) ## only the last 40% rows\n## training on the last 40% rows improves significantly the performance\n## this is what's expected because the behaviors of the market changes through time so the last month not gonna be like the next month it's better to use the recent data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:12:07.699805Z","iopub.execute_input":"2025-09-15T11:12:07.700617Z","iopub.status.idle":"2025-09-15T11:12:07.83784Z","shell.execute_reply.started":"2025-09-15T11:12:07.700588Z","shell.execute_reply":"2025-09-15T11:12:07.837056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train_with_latent.drop(columns=[\"label\"]).astype(np.float32)\ny = train_with_latent[\"label\"].astype(np.float32)\n\n#Train final model with best params\nbest_params = study.best_trial.params\nbest_params.update({\n    \"booster\": \"gbtree\",\n    \"tree_method\": \"gpu_hist\",\n    \"predictor\": \"gpu_predictor\",\n    \"objective\": \"reg:squarederror\",\n    \"eval_metric\": \"rmse\",\n    \"random_state\": 42,\n    \"n_estimators\": 2000\n})\n\nfinal_model = xgb.XGBRegressor(**best_params)\nfinal_model.fit(X, y, verbose=False)\n\n# Predict on test\nX_test = test_with_latent.astype(np.float32)\npreds2 = final_model.predict(X_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:18:57.03228Z","iopub.execute_input":"2025-09-15T11:18:57.032509Z","iopub.status.idle":"2025-09-15T11:20:13.307228Z","shell.execute_reply.started":"2025-09-15T11:18:57.032491Z","shell.execute_reply":"2025-09-15T11:20:13.306387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame({\n    \"ID\": np.arange(1,len(preds2)+1),\n    \"prediction\": preds2\n})\nsubmission.to_csv(\"submission9.csv\", index=False)\nprint(\"Saved submission2.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:20:13.308083Z","iopub.execute_input":"2025-09-15T11:20:13.30828Z","iopub.status.idle":"2025-09-15T11:20:14.080989Z","shell.execute_reply.started":"2025-09-15T11:20:13.308265Z","shell.execute_reply":"2025-09-15T11:20:14.080208Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"***","metadata":{}},{"cell_type":"code","source":"class CryptoNetV3(nn.Module):\n    def __init__(self, input_dim):\n        super().__init__()\n\n        self.norm_input = nn.LayerNorm(input_dim)\n\n        self.block1 = nn.Sequential(\n            nn.Linear(input_dim, 512),\n            nn.GELU(),\n            nn.LayerNorm(512),\n            nn.Dropout(0.2)\n        )\n        self.block2 = nn.Sequential(\n            nn.Linear(512, 256),\n            nn.GELU(),\n            nn.LayerNorm(256),\n            nn.Dropout(0.3)\n        )\n        self.block3 = nn.Sequential(\n            nn.Linear(256, 128),\n            nn.GELU(),\n            nn.LayerNorm(128),\n            nn.Dropout(0.3)\n        )\n        self.block4 = nn.Sequential(\n            nn.Linear(128, 64),\n            nn.GELU(),\n            nn.LayerNorm(64),\n            nn.Dropout(0.2)\n        )\n        self.output_layer = nn.Linear(64, 1)\n\n        self._init_weights()\n\n    def forward(self, x):\n        x = self.norm_input(x)\n        x = self.block1(x)\n        x = self.block2(x)\n        x = self.block3(x)\n        x = self.block4(x)\n        return self.output_layer(x)\n\n    def _init_weights(self):\n        for m in self.modules():\n            if isinstance(m, nn.Linear):\n                nn.init.kaiming_normal_(m.weight, nonlinearity='relu')\n                nn.init.constant_(m.bias, 0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:29:04.445101Z","iopub.execute_input":"2025-09-15T11:29:04.445367Z","iopub.status.idle":"2025-09-15T11:29:04.453223Z","shell.execute_reply.started":"2025-09-15T11:29:04.445349Z","shell.execute_reply":"2025-09-15T11:29:04.452338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\n\ndef set_seed(seed=42):\n    torch.manual_seed(seed)\n    np.random.seed(seed)\n    random.seed(seed)\n    if torch.cuda.is_available():\n        torch.cuda.manual_seed(seed)\n\nset_seed(42)\n\nclass CryptoNetV4(nn.Module):\n    def __init__(self, inp):\n        super().__init__()\n        self.ln0 = nn.LayerNorm(inp)\n        def block(in_f, out_f, p):\n            return nn.Sequential(\n                nn.Linear(in_f, out_f),\n                nn.SiLU(),\n                nn.Dropout(p),\n                nn.LayerNorm(out_f),\n            )\n        self.b1 = block(inp, 512, 0.2)\n        self.b2 = block(512, 256, 0.3)\n        self.b3 = block(256, 128, 0.3)\n        self.b4 = block(128, 64, 0.2)\n        self.out = nn.Linear(64, 1)\n        self._init()\n    def _init(self):\n        for m in self.modules():\n            if isinstance(m, nn.Linear):\n                nn.init.kaiming_normal_(m.weight, nonlinearity='relu')\n                nn.init.zeros_(m.bias)\n    def forward(self, x):\n        x = self.ln0(x)\n        for b in (self.b1, self.b2, self.b3, self.b4):\n            x = b(x)\n        return self.out(x).squeeze(-1)\n\n\ndef pearson_corr(y_true, y_pred):\n    vx = y_true - y_true.mean()\n    vy = y_pred - y_pred.mean()\n    corr = (vx * vy).sum() / (torch.sqrt((vx ** 2).sum()) * torch.sqrt((vy ** 2).sum()))\n    return corr.item() if not torch.isnan(corr) else 0.0\n\ndef train_model(X_train, y_train, X_val, y_val, input_dim, batch_size=1024, lr=5e-4, epochs=200, patience=20):\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n    train_loader = DataLoader(TensorDataset(X_train, y_train), batch_size=batch_size, shuffle=True, pin_memory=True)\n    val_loader = DataLoader(TensorDataset(X_val, y_val), batch_size=batch_size, pin_memory=True)\n\n    model = CryptoNetV3(input_dim=input_dim).to(device)\n    optimizer = torch.optim.AdamW(model.parameters(), lr=lr, weight_decay=1e-4)\n    scheduler = torch.optim.lr_scheduler.CosineAnnealingLR(optimizer, T_max=epochs)\n    criterion = nn.MSELoss()\n\n    best_corr = -np.inf\n    wait = 0\n\n    for epoch in range(1, epochs + 1):\n        model.train()\n        total_loss = 0.0\n        for xb, yb in train_loader:\n            xb, yb = xb.to(device), yb.to(device)\n            optimizer.zero_grad()\n            preds = model(xb)\n            loss = criterion(preds, yb)\n            loss.backward()\n            torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=1.0)\n            optimizer.step()\n            total_loss += loss.item()\n\n        scheduler.step()\n\n        # Validation\n        model.eval()\n        val_preds, val_targets = [], []\n        with torch.no_grad():\n            for xb, yb in val_loader:\n                xb = xb.to(device)\n                preds = model(xb).cpu()\n                val_preds.append(preds)\n                val_targets.append(yb)\n\n        val_preds = torch.cat(val_preds)\n        val_targets = torch.cat(val_targets)\n        val_corr = pearson_corr(val_targets, val_preds)\n\n        print(f\"Epoch {epoch:03d} | Loss: {total_loss:.4f} | Val Corr: {val_corr:.5f}\")\n\n        if val_corr > best_corr:\n            best_corr = val_corr\n            wait = 0\n            torch.save(model.state_dict(), \"best_model.pt\")\n        else:\n            wait += 1\n            if wait >= patience:\n                print(f\"Early stopping at epoch {epoch}\")\n                break\n\n    model.load_state_dict(torch.load(\"best_model.pt\"))\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:29:17.61653Z","iopub.execute_input":"2025-09-15T11:29:17.61731Z","iopub.status.idle":"2025-09-15T11:29:17.633436Z","shell.execute_reply.started":"2025-09-15T11:29:17.617286Z","shell.execute_reply":"2025-09-15T11:29:17.632772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_tensor = torch.tensor(train_with_latent.drop(columns=[\"label\"]).values, dtype=torch.float32)\ny_tensor = torch.tensor(train_with_latent[\"label\"].values, dtype=torch.float32)\n\n# Split to train/val\nval_split = 0.2\nsplit_idx = int(len(X_tensor) * (1 - val_split))\nX_train, X_val = X_tensor[:split_idx], X_tensor[split_idx:]\ny_train, y_val = y_tensor[:split_idx], y_tensor[split_idx:]\n\nmodel = train_model(\n    X_train, y_train, X_val, y_val,\n    input_dim=X_tensor.shape[1],\n    lr=3e-4,\n    epochs=300,\n    batch_size=1024,\n    patience=10\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:29:21.59683Z","iopub.execute_input":"2025-09-15T11:29:21.597464Z","execution_failed":"2025-09-15T11:29:35.25Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_with_latent = test_with_latent.fillna(0.0)\n\nX_test = torch.tensor(test_with_latent.values, dtype=torch.float32).to(device)  # <-- move to GPU if model is on GPU\n\nmodel.eval()\nwith torch.no_grad():\n    y_pred_test = model(X_test).cpu().numpy().squeeze()  # bring back to CPU for numpy\n\nsubmission = pd.DataFrame({\n    \"ID\": np.arange(1,len(y_pred_test)+1),   # replace with actual test IDs if available\n    \"prediction\": y_pred_test\n})\n\nsubmission.to_csv(\"submission_latent2.csv\", index=False)\nprint(\"Saved submission_latent.csv\")\nsubmission.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T11:28:07.217731Z","iopub.execute_input":"2025-09-15T11:28:07.218061Z","iopub.status.idle":"2025-09-15T11:28:08.572366Z","shell.execute_reply.started":"2025-09-15T11:28:07.218Z","shell.execute_reply":"2025-09-15T11:28:08.571712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}