{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-13T12:42:49.579997Z","iopub.execute_input":"2025-06-13T12:42:49.580942Z","iopub.status.idle":"2025-06-13T12:42:50.952921Z","shell.execute_reply.started":"2025-06-13T12:42:49.580908Z","shell.execute_reply":"2025-06-13T12:42:50.952126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# データの読み込み\ntrain = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\ntest = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\nsample_submission = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\n\n# データの確認\nprint(f\"Train Data Shape: {train.shape}\")\nprint(f\"Test Data Shape: {test.shape}\")\nprint(f\"Sample Submission Shape: {sample_submission.shape}\")\n\n# 最初の数行を確認\nprint(train.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T12:42:50.954374Z","iopub.execute_input":"2025-06-13T12:42:50.954785Z","iopub.status.idle":"2025-06-13T12:43:41.663542Z","shell.execute_reply.started":"2025-06-13T12:42:50.954755Z","shell.execute_reply":"2025-06-13T12:43:41.661939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# データの列名を確認\nprint(train.columns)\n\n# timestampをIDとして残し、その他の数値データのみを抽出\nnumerical_features = train.select_dtypes(include=[np.number])\n\n# 目的変数（label）の取得\ntarget = train['label']\n\n# 特徴量と目的変数の相関を算出\ncorrelations = numerical_features.corrwith(target)\n\n# 相関係数を降順に並べ替え\nsorted_correlations = correlations.sort_values(ascending=False)\n\n# 相関係数の絶対値を取って、上位15の特徴量を選択\ntop_15_features = sorted_correlations.abs().sort_values(ascending=False).head(15)\nprint(\"Top 15 Features Based on Absolute Correlation with Target:\")\nprint(top_15_features)\n\n# 目的変数との相関が上位15の特徴量を選択\nselected_features = top_15_features.index.tolist()\n\n# 交差特徴量の作成：bid_qty / ask_qty, buy_qty / sell_qty\ntrain['bid_qty_to_ask_qty'] = train['bid_qty'] / train['ask_qty']\ntrain['buy_qty_to_sell_qty'] = train['buy_qty'] / train['sell_qty']\n\n# 新たに作成した交差特徴量をcorrelationsに追加\nadditional_correlations = train[['bid_qty_to_ask_qty', 'buy_qty_to_sell_qty']].corrwith(target)\n\n# 交差特徴量の相関をcorrelationsに手動で追加\ncorrelations = pd.concat([correlations, additional_correlations])\n\n# 交差特徴量をfinal_featuresに追加\nfinal_features = selected_features + ['bid_qty_to_ask_qty', 'buy_qty_to_sell_qty']\n\n# 最終的に使用する特徴量の相関を表示\nfinal_correlations = correlations[final_features]\nprint(\"\\nCorrelations of Selected Features with Target:\")\nprint(final_correlations)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T13:29:14.572559Z","iopub.execute_input":"2025-06-13T13:29:14.573060Z","iopub.status.idle":"2025-06-13T13:29:45.327840Z","shell.execute_reply.started":"2025-06-13T13:29:14.573037Z","shell.execute_reply":"2025-06-13T13:29:45.326733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\n# 使用する特徴量（final_features）を抽出\nX = train[final_features]\n\n# StandardScalerを使用してデータを標準化（スケーリング）\nscaler = StandardScaler()\n\n# スケーリングされた特徴量\nX_scaled = scaler.fit_transform(X)\n\n# スケーリング後のデータをDataFrameに戻す\nX_scaled_df = pd.DataFrame(X_scaled, columns=final_features)\n\n# 最初の数行を表示\nprint(X_scaled_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T13:32:13.265814Z","iopub.execute_input":"2025-06-13T13:32:13.266235Z","iopub.status.idle":"2025-06-13T13:32:13.738941Z","shell.execute_reply.started":"2025-06-13T13:32:13.266205Z","shell.execute_reply":"2025-06-13T13:32:13.737943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\nfrom sklearn.model_selection import train_test_split, KFold, cross_val_score\nfrom sklearn.metrics import mean_squared_error\n\n# 訓練データと特徴量を分割\nX = train[final_features]\ny = target\n\n# 分割検証のためのKFoldを定義\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\n\n# XGBoostモデルの設定\nmodel = xgb.XGBRegressor(objective='reg:squarederror', random_state=42)\n\n# クロスバリデーションでモデルを評価\ncv_scores = cross_val_score(model, X, y, cv=kf, scoring='neg_mean_squared_error')\n\n# 評価結果（平均二乗誤差）を表示\nprint(f\"Cross-Validation MSE scores: {-cv_scores}\")\nprint(f\"Mean CV MSE: {-cv_scores.mean()}\")\n\n# 最終的に全データでモデルを学習\nmodel.fit(X, y)\n\n# 学習済みモデルを使って予測を実行（訓練データで予測）\ny_pred = model.predict(X)\n\n# 平均二乗誤差（MSE）を計算\nmse = mean_squared_error(y, y_pred)\nprint(f\"Mean Squared Error on Training Data: {mse}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T13:32:35.290453Z","iopub.execute_input":"2025-06-13T13:32:35.290777Z","iopub.status.idle":"2025-06-13T13:32:57.674960Z","shell.execute_reply.started":"2025-06-13T13:32:35.290754Z","shell.execute_reply":"2025-06-13T13:32:57.673924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# 正式なテストデータの読み込み\ntest = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\n\n# 交差特徴量の作成：テストデータにも追加\ntest['bid_qty_to_ask_qty'] = test['bid_qty'] / test['ask_qty']\ntest['buy_qty_to_sell_qty'] = test['buy_qty'] / test['sell_qty']\n\n# テストデータから、学習で使用した特徴量（final_features）を抽出\nX_test = test[final_features]\n\n# 学習したXGBoostモデルで予測を実施\ny_test_pred = model.predict(X_test)\n\n# 提出用ファイルの作成\nsubmission = pd.DataFrame({\n    'ID': test.index,  # testデータのインデックスをID列として使用\n    'prediction': y_test_pred  # 予測結果をprediction列として保存\n})\n\n# 提出用ファイルをCSV形式で保存\nsubmission_path = '/kaggle/working/submission.csv'\nsubmission.to_csv(submission_path, index=False)\n\n# 提出用ファイルの最初の30行を表示\nprint(\"First 30 rows of the submission file:\")\nprint(submission.head(30))\n\n# 提出用ファイルの記述統計を表示\nprint(\"\\nDescriptive statistics of the submission file:\")\nprint(submission.describe())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T13:35:00.468109Z","iopub.execute_input":"2025-06-13T13:35:00.468500Z","iopub.status.idle":"2025-06-13T13:35:10.023058Z","shell.execute_reply.started":"2025-06-13T13:35:00.468474Z","shell.execute_reply":"2025-06-13T13:35:10.022125Z"}},"outputs":[],"execution_count":null}]}