{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-28T12:23:38.516885Z","iopub.execute_input":"2025-06-28T12:23:38.517247Z","iopub.status.idle":"2025-06-28T12:23:42.681669Z","shell.execute_reply.started":"2025-06-28T12:23:38.517216Z","shell.execute_reply":"2025-06-28T12:23:42.680715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n# データの読み込み\ntrain = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\ntest = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\nsample_submission = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\n\n# データの確認\nprint(f\"Train Data Shape: {train.shape}\")\nprint(f\"Test Data Shape: {test.shape}\")\nprint(f\"Sample Submission Shape: {sample_submission.shape}\")\n\n\n# カラム一覧を先頭50個だけ表示（匿名カラムの形式を特定）\nprint(train.columns[:50])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T12:23:42.683766Z","iopub.execute_input":"2025-06-28T12:23:42.684146Z","iopub.status.idle":"2025-06-28T12:24:35.241542Z","shell.execute_reply.started":"2025-06-28T12:23:42.684123Z","shell.execute_reply":"2025-06-28T12:24:35.240641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 匿名特徴量（X1〜X890相当）の抽出\nanon_cols = [col for col in train.columns if col.startswith('X')]\n\n# 相関係数を算出\ncorrelations = train[anon_cols].corrwith(train['label'])\n\n# 相関の絶対値が高い上位50個を取得\ntop_corr = correlations.abs().sort_values(ascending=False).head(50)\nprint(\"Top 50 Features by Absolute Correlation:\")\nprint(top_corr)\n\n# 公開特徴量\npublic_cols = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\n\n# 特徴量候補：公開5 + 匿名特徴量上位50\nselected_features = public_cols + top_corr.index.tolist()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T12:24:35.242531Z","iopub.execute_input":"2025-06-28T12:24:35.242816Z","iopub.status.idle":"2025-06-28T12:24:47.277433Z","shell.execute_reply.started":"2025-06-28T12:24:35.242794Z","shell.execute_reply":"2025-06-28T12:24:47.276503Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"モデル構築。\n\n相関上位20で実施。\n\n交差検証は行わない。　\n多重共線性を排除する、特徴量選択を行う。\n\nただし、 公開6特徴量は学習にかならず使用する。","metadata":{}},{"cell_type":"markdown","source":"前処理：\n\n❌ 全カラムを一気に処理 → ✅ 対象列のみ処理\n\n❌ inplaceコピー大量使用 → ✅ 参照元は変えず必要列のみ一時処理\n\n✅ 公開 + 相関上位50列のみを対象に inf → NaN → 0 補完（最小限の処理）","metadata":{}},{"cell_type":"code","source":"import numpy as np\n\n# === 公開特徴量（常に使う） ===\npublic_cols = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\n\n# === 派生特徴量（公開系） ===\ntrain['bid_ask_ratio'] = train['bid_qty'] / (train['ask_qty'] + 1e-6)\ntrain['buy_sell_ratio'] = train['buy_qty'] / (train['sell_qty'] + 1e-6)\ntrain['buy_volume_ratio'] = train['buy_qty'] / (train['volume'] + 1e-6)\ntrain['sell_volume_ratio'] = train['sell_qty'] / (train['volume'] + 1e-6)\n\n# 派生特徴量の名前を保持\nderived_features = [\n    'bid_ask_ratio', 'buy_sell_ratio', 'buy_volume_ratio', 'sell_volume_ratio'\n]\n\n# === 匿名特徴量の上位50（すでに前セルで抽出済みの top_corr）===\n# 再計算せず再利用\ntop_50_anon = top_corr.index.tolist()\n\n# === 最小限の欠損/無限大処理：対象列のみ（匿名 + 公開 + 派生） ===\ntarget_cols = public_cols + top_50_anon + derived_features\nfor col in target_cols:\n    train[col] = train[col].replace([np.inf, -np.inf], np.nan).fillna(0)\n\n# === 相関係数を用いた共線性除去（匿名特徴量のみ） ===\ndef remove_highly_correlated_features(df, features, threshold=0.95):\n    corr_matrix = df[features].corr().abs()\n    upper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool))\n    drop_cols = [column for column in upper.columns if any(upper[column] > threshold)]\n    return [f for f in features if f not in drop_cols]\n\nfiltered_anon = remove_highly_correlated_features(train, top_50_anon, threshold=0.95)\n\n# === 最終特徴量セット（公開 + 派生 + 共線性除去済み匿名） ===\nfinal_features = public_cols + derived_features + filtered_anon\nprint(f\"最終使用特徴量数: {len(final_features)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T12:24:47.278442Z","iopub.execute_input":"2025-06-28T12:24:47.278722Z","iopub.status.idle":"2025-06-28T12:24:51.888774Z","shell.execute_reply.started":"2025-06-28T12:24:47.278689Z","shell.execute_reply":"2025-06-28T12:24:51.887630Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"スライスは：\n\n\n   'first_50pct': (0, int(n_total * 0.5)),         # 最初の50%\n    'middle_30pct': (int(n_total * 0.5), int(n_total * 0.8)),  # 次の30%\n    'last_20pct': (int(n_total * 0.8), n_total)     # 最後の20%\n\n\nモデルは：\n\n\nXGBRegressor（early stopping付き）\n","metadata":{}},{"cell_type":"code","source":"from scipy.stats import pearsonr\nfrom sklearn.model_selection import train_test_split\nfrom xgboost import XGBRegressor\nimport joblib\nimport numpy as np\n\n# ✅ 全体データ数\nn_total = len(train)\n\n# ✅ スライス設定（時間順）\nslices = {\n    'first_50pct': (0, int(n_total * 0.5)),\n    'middle_30pct': (int(n_total * 0.5), int(n_total * 0.8)),\n    'last_20pct': (int(n_total * 0.8), n_total)\n}\n\n# === スコア格納 ===\nval_scores = []\nval_lens = []\nval_preds = []\nval_true = None\n\nfor slice_name, (start_idx, end_idx) in slices.items():\n    print(f\"\\n[Slice: {slice_name}] Rows {start_idx} to {end_idx}\")\n\n    # スライス抽出\n    subset = train.iloc[start_idx:end_idx].copy()\n\n    # === 派生特徴量の再計算（データが分割されているため） ===\n    subset['bid_ask_ratio'] = subset['bid_qty'] / (subset['ask_qty'] + 1e-6)\n    subset['buy_sell_ratio'] = subset['buy_qty'] / (subset['sell_qty'] + 1e-6)\n    subset['buy_volume_ratio'] = subset['buy_qty'] / (subset['volume'] + 1e-6)\n    subset['sell_volume_ratio'] = subset['sell_qty'] / (subset['volume'] + 1e-6)\n\n    # === 欠損補完（学習で使用する全特徴量に対して） ===\n    for col in final_features:\n        if col in subset.columns:\n            subset[col] = subset[col].replace([np.inf, -np.inf], np.nan).fillna(0)\n        else:\n            print(f\"Warning: {col} not found in subset. Filling with 0.\")\n            subset[col] = 0\n\n    # 特徴量と目的変数を抽出\n    X = subset[final_features]\n    y = subset['label']\n\n    # 最初の1回のみ検証用ラベルを保存（可視化や解析用途）\n    if val_true is None:\n        val_true = y.iloc[int(len(y) * 0.8):].reset_index(drop=True)\n\n    # 時系列分割（shuffle=False）\n    X_train, X_val, y_train, y_val = train_test_split(\n        X, y, test_size=0.2, shuffle=False\n    )\n\n    print(\" - Training XGB\")\n    model = XGBRegressor(\n        n_estimators=100,\n        learning_rate=0.1,\n        max_depth=5,\n        random_state=42,\n        tree_method='hist',\n        enable_categorical=False,\n        verbosity=0,\n        early_stopping_rounds=30  # ✅ constructor側に設定（警告対策）\n    )\n\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_val, y_val)],\n        verbose=False\n    )\n\n    # ✅ モデル保存（特徴量とともに）\n    model_path = f\"/kaggle/working/xgb_model_{slice_name}.pkl\"\n    joblib.dump((model, final_features), model_path)\n    print(f\"   → Model + features saved to {model_path}\")\n\n    # === 検証スコア計算 ===\n    y_pred = model.predict(X_val)\n    score = pearsonr(y_val, y_pred)[0]\n    val_preds.append(y_pred)\n    val_lens.append(len(y_val))\n    val_scores.append(score)\n\n    print(f\"   → Pearson: {score:.5f} (len: {len(y_val)})\")\n\n# === スコア集計表示 ===\navg_score = np.mean(val_scores)\nweighted_score = np.average(val_scores, weights=val_lens)\n\nprint(f\"\\n✅ Average Pearson (equal):   {avg_score:.5f}\")\nprint(f\"✅ Average Pearson (weighted by size): {weighted_score:.5f}\")\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T12:24:51.890211Z","iopub.execute_input":"2025-06-28T12:24:51.890533Z","iopub.status.idle":"2025-06-28T12:25:04.738562Z","shell.execute_reply.started":"2025-06-28T12:24:51.890509Z","shell.execute_reply":"2025-06-28T12:25:04.737853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport joblib  # モデル読み込み用\n\n# === テストデータの読み込み ===\ntest = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\n\n# === 派生特徴量の追加（学習と同様）===\ntest['bid_ask_ratio'] = test['bid_qty'] / (test['ask_qty'] + 1e-6)\ntest['buy_sell_ratio'] = test['buy_qty'] / (test['sell_qty'] + 1e-6)\ntest['buy_volume_ratio'] = test['buy_qty'] / (test['volume'] + 1e-6)\ntest['sell_volume_ratio'] = test['sell_qty'] / (test['volume'] + 1e-6)\n\n# === スライス名と重み設定（後半データ重視） ===\nslices = ['first_50pct', 'middle_30pct', 'last_20pct']\nweights = {'first_50pct': 0.2, 'middle_30pct': 0.3, 'last_20pct': 0.5}\n\n# === 予測結果初期化 ===\nweighted_preds = np.zeros(len(test))\n\n# === 各スライスモデルで推論して加重平均 ===\nfor slice_name in slices:\n    model_path = f'/kaggle/working/xgb_model_{slice_name}.pkl'\n    print(f\"Loading model: {model_path}\")\n    \n    # ✅ モデルと使用特徴量を読み込み\n    model, final_features = joblib.load(model_path)\n    \n    # 特徴量がテストデータに存在するか確認・補完\n    for col in final_features:\n        if col not in test.columns:\n            print(f\"Warning: {col} not in test set. Filling with 0.\")\n            test[col] = 0\n        else:\n            test[col] = test[col].replace([np.inf, -np.inf], np.nan).fillna(0)\n\n    # 特徴量抽出と予測\n    X_test = test[final_features]\n    y_pred = model.predict(X_test)\n\n    # 重み付きで加算\n    weighted_preds += weights[slice_name] * y_pred\n\n# === 提出ファイルの作成 ===\nsubmission = pd.DataFrame({\n    'ID': test['row_id'] if 'row_id' in test.columns else test.index,\n    'prediction': weighted_preds\n})\n\n# 保存\nsubmission_path = '/kaggle/working/submission.csv'\nsubmission.to_csv(submission_path, index=False)\n\n# === 提出ファイルの確認 ===\nprint(\"First 10 rows of the submission file:\")\nprint(submission.head(10))\n\nprint(\"\\nSubmission file stats:\")\nprint(submission.describe())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T12:25:04.739813Z","iopub.execute_input":"2025-06-28T12:25:04.740154Z","iopub.status.idle":"2025-06-28T12:25:14.981532Z","shell.execute_reply.started":"2025-06-28T12:25:04.740123Z","shell.execute_reply":"2025-06-28T12:25:14.980557Z"}},"outputs":[],"execution_count":null}]}