{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:32:21.373301Z","iopub.execute_input":"2025-07-22T12:32:21.374340Z","iopub.status.idle":"2025-07-22T12:32:21.921061Z","shell.execute_reply.started":"2025-07-22T12:32:21.374273Z","shell.execute_reply":"2025-07-22T12:32:21.919804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:32:21.922432Z","iopub.execute_input":"2025-07-22T12:32:21.922923Z","iopub.status.idle":"2025-07-22T12:32:55.735518Z","shell.execute_reply.started":"2025-07-22T12:32:21.922894Z","shell.execute_reply":"2025-07-22T12:32:55.734521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df.drop('label', axis=1)\ny = df['label']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:32:55.736674Z","iopub.execute_input":"2025-07-22T12:32:55.736979Z","iopub.status.idle":"2025-07-22T12:32:58.181507Z","shell.execute_reply.started":"2025-07-22T12:32:55.736954Z","shell.execute_reply":"2025-07-22T12:32:58.180435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.3, random_state = 0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:32:58.183552Z","iopub.execute_input":"2025-07-22T12:32:58.183855Z","iopub.status.idle":"2025-07-22T12:33:04.931979Z","shell.execute_reply.started":"2025-07-22T12:32:58.183817Z","shell.execute_reply":"2025-07-22T12:33:04.930975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBRegressor\n\n# 모델 선언 예시\nmodel = XGBRegressor(n_estimators=500, learning_rate=0.2, max_depth=4, random_state =0)\nmodel.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:33:04.932924Z","iopub.execute_input":"2025-07-22T12:33:04.933547Z","iopub.status.idle":"2025-07-22T12:39:54.732610Z","shell.execute_reply.started":"2025-07-22T12:33:04.933516Z","shell.execute_reply":"2025-07-22T12:39:54.731265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fscore = model.get_booster().get_fscore()\nsorted_dic = dict(sorted(fscore.items(), key=lambda item: item[1], reverse=True))\nsorted_dic","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:39:54.733842Z","iopub.execute_input":"2025-07-22T12:39:54.734162Z","iopub.status.idle":"2025-07-22T12:39:54.752413Z","shell.execute_reply.started":"2025-07-22T12:39:54.734133Z","shell.execute_reply":"2025-07-22T12:39:54.751000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = [k for k, v in sorted_dic.items() if v >= 20]\nfeatures","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:39:54.753418Z","iopub.execute_input":"2025-07-22T12:39:54.753687Z","iopub.status.idle":"2025-07-22T12:39:54.764275Z","shell.execute_reply.started":"2025-07-22T12:39:54.753665Z","shell.execute_reply":"2025-07-22T12:39:54.763294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_X = X[features]\nnew_X","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:39:54.765395Z","iopub.execute_input":"2025-07-22T12:39:54.765707Z","iopub.status.idle":"2025-07-22T12:39:55.183806Z","shell.execute_reply.started":"2025-07-22T12:39:54.765680Z","shell.execute_reply":"2025-07-22T12:39:55.182642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(new_X, y, test_size = 0.3, random_state = 0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:39:55.185227Z","iopub.execute_input":"2025-07-22T12:39:55.185608Z","iopub.status.idle":"2025-07-22T12:39:56.189883Z","shell.execute_reply.started":"2025-07-22T12:39:55.185571Z","shell.execute_reply":"2025-07-22T12:39:56.188905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBRegressor\n\n# 모델 선언 예시\nmodel = XGBRegressor(n_estimators=500, learning_rate=0.2, max_depth=4, random_state =0)\nmodel.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:39:56.193150Z","iopub.execute_input":"2025-07-22T12:39:56.193517Z","iopub.status.idle":"2025-07-22T12:40:32.292124Z","shell.execute_reply.started":"2025-07-22T12:39:56.193489Z","shell.execute_reply":"2025-07-22T12:40:32.290989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import r2_score\n\ny_pred = model.predict(X_test)\nprint(f'mse:{mean_squared_error(y_test, y_pred)}')\nprint(f'r2_score:{r2_score(y_test, y_pred)}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:40:32.294445Z","iopub.execute_input":"2025-07-22T12:40:32.294784Z","iopub.status.idle":"2025-07-22T12:40:32.992656Z","shell.execute_reply.started":"2025-07-22T12:40:32.294755Z","shell.execute_reply":"2025-07-22T12:40:32.991558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_pred = model.predict(X)\n# pd.DataFrame(train_pred).to_csv('train_pred.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:40:32.993814Z","iopub.execute_input":"2025-07-22T12:40:32.994739Z","iopub.status.idle":"2025-07-22T12:40:32.998903Z","shell.execute_reply.started":"2025-07-22T12:40:32.994699Z","shell.execute_reply":"2025-07-22T12:40:32.997684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:40:32.999826Z","iopub.execute_input":"2025-07-22T12:40:33.000089Z","iopub.status.idle":"2025-07-22T12:40:58.325832Z","shell.execute_reply.started":"2025-07-22T12:40:33.000068Z","shell.execute_reply":"2025-07-22T12:40:58.324804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = test.drop('label',axis=1)[features]\ny = test['label']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:55:28.943855Z","iopub.execute_input":"2025-07-22T12:55:28.944187Z","iopub.status.idle":"2025-07-22T12:55:31.035442Z","shell.execute_reply.started":"2025-07-22T12:55:28.944166Z","shell.execute_reply":"2025-07-22T12:55:31.034207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"prediction = model.predict(X)\nprediction = pd.DataFrame(prediction)\nprediction['ID'] = range(1,len(prediction)+1)\nprediction = prediction.rename(columns={0:'prediction'})[['ID','prediction']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:41:00.554360Z","iopub.execute_input":"2025-07-22T12:41:00.554646Z","iopub.status.idle":"2025-07-22T12:41:03.814439Z","shell.execute_reply.started":"2025-07-22T12:41:00.554623Z","shell.execute_reply":"2025-07-22T12:41:03.813299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"prediction[['ID','prediction']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:41:03.815521Z","iopub.execute_input":"2025-07-22T12:41:03.815810Z","iopub.status.idle":"2025-07-22T12:41:03.831398Z","shell.execute_reply.started":"2025-07-22T12:41:03.815786Z","shell.execute_reply":"2025-07-22T12:41:03.830449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"prediction.to_csv('prediction.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:41:03.832295Z","iopub.execute_input":"2025-07-22T12:41:03.832562Z","iopub.status.idle":"2025-07-22T12:41:04.923589Z","shell.execute_reply.started":"2025-07-22T12:41:03.832535Z","shell.execute_reply":"2025-07-22T12:41:04.922600Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom xgboost import XGBRegressor\nfrom sklearn.model_selection import TimeSeriesSplit\nfrom sklearn.metrics import mean_squared_error\n\n# X, y는 주어진 노트북 기준\n# 예시: X = new_X, y = y\n# 아래 라인은 생략 가능\nX = df.drop('label', axis=1)[features]\ny = df['label']\n\n# 자동 n_splits 설정\nmin_val_size = 20  # 한 validation set의 최소 크기\nn_samples = len(X)\nmax_splits = n_samples // min_val_size\n\nif max_splits < 2:\n    raise ValueError(\"데이터가 너무 적어 교차검증을 할 수 없습니다.\")\n\nn_splits = min(5, max_splits)  # 최대 5개 폴드까지만 사용\ntscv = TimeSeriesSplit(n_splits=n_splits)\n\nprint(f\"총 샘플 수: {n_samples}, 사용 폴드 수: {n_splits}\\n\")\n\n# 결과 저장용\nrmse_list = []\n\nfor fold, (train_idx, val_idx) in enumerate(tscv.split(X), 1):\n    if max(val_idx) >= len(X):\n        print(f\"⚠️ Fold {fold} skipped: val_idx out of bounds\")\n        continue\n\n    X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n    y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n    print(f\"Fold {fold} | Train: {train_idx[0]}~{train_idx[-1]} | Val: {val_idx[0]}~{val_idx[-1]}\")\n\n    # 모델 훈련\n    model = XGBRegressor(\n        n_estimators=100,\n        learning_rate=0.1,\n        max_depth=3,\n        random_state=42,\n        n_jobs=-1\n    )\n\n    model.fit(X_train, y_train)\n\n    # 예측 및 평가\n    y_pred = model.predict(X_val)\n    rmse = mean_squared_error(y_val, y_pred, squared=False)\n    rmse_list.append(rmse)\n\n    print(f\"  📊 Fold {fold} RMSE: {rmse:.4f}\\n\")\n\n# 전체 평균 RMSE\nprint(f\"✅ 평균 RMSE: {np.mean(rmse_list):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:57:48.657555Z","iopub.execute_input":"2025-07-22T12:57:48.657949Z","iopub.status.idle":"2025-07-22T12:58:22.462075Z","shell.execute_reply.started":"2025-07-22T12:57:48.657915Z","shell.execute_reply":"2025-07-22T12:58:22.461298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = test.drop('label',axis=1)[features]\n\nprediction = model.predict(X)\nprediction = pd.DataFrame(prediction)\nprediction['ID'] = range(1,len(prediction)+1)\nprediction = prediction.rename(columns={0:'prediction'})[['ID','prediction']]\n\nprediction[['ID','prediction']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:58:55.558646Z","iopub.execute_input":"2025-07-22T12:58:55.559012Z","iopub.status.idle":"2025-07-22T12:58:57.483156Z","shell.execute_reply.started":"2025-07-22T12:58:55.558984Z","shell.execute_reply":"2025-07-22T12:58:57.482003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"prediction.to_csv('fold_prediction.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T12:59:16.282803Z","iopub.execute_input":"2025-07-22T12:59:16.283144Z","iopub.status.idle":"2025-07-22T12:59:17.418391Z","shell.execute_reply.started":"2025-07-22T12:59:16.283121Z","shell.execute_reply":"2025-07-22T12:59:17.417149Z"}},"outputs":[],"execution_count":null}]}