{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":11305158,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport polars as pl\nfrom matplotlib import pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-20T02:45:28.260411Z","iopub.execute_input":"2025-09-20T02:45:28.261333Z","iopub.status.idle":"2025-09-20T02:45:30.258726Z","shell.execute_reply.started":"2025-09-20T02:45:28.261289Z","shell.execute_reply":"2025-09-20T02:45:30.257362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ROOT_DIR = \"/kaggle/input/jane-street-real-time-market-data-forecasting\"\n\n# === 경로 ===\ntrain_path = os.path.join(ROOT_DIR, \"train.parquet\")\n\n# === 메타데이터 ===\nlags = os.path.join(ROOT_DIR, \"lags.parquet\")\ntrain_path = os.path.join(ROOT_DIR, \"train.parquet\")\nlags_path  = os.path.join(ROOT_DIR, \"lags.parquet\")\nfeatures   = pl.read_csv(os.path.join(ROOT_DIR, \"features.csv\"))\nresponders = pl.read_csv(os.path.join(ROOT_DIR, \"responders.csv\"))\nsample_sub = pl.read_csv(os.path.join(ROOT_DIR, \"sample_submission.csv\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T02:46:06.163774Z","iopub.execute_input":"2025-09-20T02:46:06.164177Z","iopub.status.idle":"2025-09-20T02:46:06.190179Z","shell.execute_reply.started":"2025-09-20T02:46:06.164149Z","shell.execute_reply":"2025-09-20T02:46:06.188811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_path = os.path.join(ROOT_DIR, \"train.parquet\", \"partition_id=0\", \"part-0.parquet\")\ndata_0 = pl.read_parquet(file_path)\n\n# responder 계열 컬럼 확인\nresp_cols = [c for c in data_0.columns if c.startswith(\"responder_\")]\n\n# responder_6만 남기고 나머지 drop\nresp_drop = [c for c in resp_cols if c != \"responder_6\"]\ndata_0 = data_0.drop(resp_drop)\n\n\n# 삭제할 feature 컬럼들 : 결측치 많은 애들\ndrop_features = [\"feature_00\", \"feature_01\", \"feature_02\", \"feature_03\",\"feature_04\", \"feature_26\", \"feature_27\", \"feature_31\"]\n\ndata_0 = data_0.drop(drop_features)\n\n# 버전1 데이터셋 구성 미리보기 : x 데이터에 feature만 남기기\ndata_0_y = data_0[\"responder_6\"]\ndata_0_X = data_0.drop(\"responder_6\",\"date_id\",\"time_id\",\"symbol_id\", \"weight\" )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T02:46:14.403808Z","iopub.execute_input":"2025-09-20T02:46:14.404162Z","iopub.status.idle":"2025-09-20T02:46:19.171020Z","shell.execute_reply.started":"2025-09-20T02:46:14.404137Z","shell.execute_reply":"2025-09-20T02:46:19.169508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_0_X","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T02:47:00.249829Z","iopub.execute_input":"2025-09-20T02:47:00.251078Z","iopub.status.idle":"2025-09-20T02:47:00.282729Z","shell.execute_reply.started":"2025-09-20T02:47:00.250991Z","shell.execute_reply":"2025-09-20T02:47:00.281295Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **(part0만 해봄)**","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\nimport pandas as pd\n\n# Polars → Pandas\nX = data_0_X.to_pandas()\ny = data_0_y.to_pandas()\n\n# LightGBM 모델 (회귀 예시)\nmodel = lgb.LGBMRegressor(\n    n_estimators=500,\n    learning_rate=0.05,\n    subsample=0.8,\n    colsample_bytree=0.8,\n    random_state=42,\n    n_jobs=-1\n)\n\n# 학습 (train만)\nmodel.fit(X, y)\n\n# 피처 중요도 추출\nbooster = model.booster_\nimp_df = pd.DataFrame({\n    \"feature\": model.feature_name_,\n    \"gain\": booster.feature_importance(importance_type=\"gain\"),\n    \"split\": booster.feature_importance(importance_type=\"split\"),\n})\nimp_df[\"gain_norm\"] = imp_df[\"gain\"] / (imp_df[\"gain\"].sum() + 1e-12)\nimp_df = imp_df.sort_values(\"gain\", ascending=False).reset_index(drop=True)\n\nprint(imp_df.head(30))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T02:47:26.387752Z","iopub.execute_input":"2025-09-20T02:47:26.388200Z","iopub.status.idle":"2025-09-20T02:48:55.078910Z","shell.execute_reply.started":"2025-09-20T02:47:26.388168Z","shell.execute_reply":"2025-09-20T02:48:55.077711Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **for문으로 part0~6까지 돌리는 코드**","metadata":{}},{"cell_type":"code","source":"# 첫 번째 코드에서 정의한 drop_features 재사용\ndrop_features = [\"feature_00\", \"feature_01\", \"feature_02\", \"feature_03\",\n                 \"feature_04\", \"feature_26\", \"feature_27\", \"feature_31\"]\n\n# X에서 제외할 메타 컬럼 (첫 번째 코드박스 기준)\nmeta_cols = [\"date_id\", \"time_id\", \"symbol_id\", \"weight\"]\n\n# 결과 저장 dict\nhigh_features_dict = {} \n\nfor pid in range(7):  # partition_id=0~6\n    print(f\"=== Partition {pid} ===\")\n\n    # --------------------\n    # 1) 데이터 로드\n    # --------------------\n    f = os.path.join(ROOT_DIR, \"train.parquet\", f\"partition_id={pid}\", \"part-0.parquet\")\n    df = pl.read_parquet(f)\n\n    # responder 처리 (responder_6만 남기기)\n    df = df.drop([c for c in df.columns if c.startswith(\"responder_\") and c != \"responder_6\"])\n\n    # 결측치 많은 feature drop\n    df = df.drop(drop_features)\n\n    # --------------------\n    # 2) X / y 준비 (첫 번째 코드 로직 재사용)\n    # --------------------\n    df_y = df[\"responder_6\"]\n    df_X = df.drop([\"responder_6\"] + [c for c in meta_cols if c in df.columns])\n\n    X = df_X.to_pandas()\n    y = df_y.to_pandas()\n\n    # --------------------\n    # 3) LightGBM 학습 (partition별)\n    # --------------------\n    model = lgb.LGBMRegressor(\n        n_estimators=500,\n        learning_rate=0.05,\n        subsample=0.8,\n        colsample_bytree=0.8,\n        random_state=42,\n        n_jobs=-1\n    )\n    model.fit(X, y)\n\n    booster = model.booster_\n    imp_df = pd.DataFrame({\n        \"feature\": model.feature_name_,\n        \"gain\": booster.feature_importance(importance_type=\"gain\"),\n    })\n    imp_df[\"gain_norm\"] = imp_df[\"gain\"] / (imp_df[\"gain\"].sum() + 1e-12)\n    imp_df = imp_df.sort_values(\"gain\", ascending=False).reset_index(drop=True)\n\n    # --------------------\n    # 4) 상위 30개 feature 추출 (리스트만 저장)\n    TOP_N = 30\n    high_features = imp_df[\"feature\"].iloc[:TOP_N].tolist()\n\n    # 파티션별로 리스트 저장\n    high_features_dict[f\"high_features_{pid}\"] = high_features\n\n    print(f\"→ high_features_{pid}: {len(high_features)} features\")\n\n    # 이제 high_features_dict[\"high_features_0\"], high_features_dict[\"high_features_1\"], ... 접근 가능","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T02:49:06.470063Z","iopub.execute_input":"2025-09-20T02:49:06.470936Z","iopub.status.idle":"2025-09-20T03:10:22.545008Z","shell.execute_reply.started":"2025-09-20T02:49:06.470904Z","shell.execute_reply":"2025-09-20T03:10:22.543077Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **뽑은 피처들 교집합 구하기**","metadata":{}},{"cell_type":"code","source":"print(high_features_dict.keys())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T03:10:30.434534Z","iopub.execute_input":"2025-09-20T03:10:30.435009Z","iopub.status.idle":"2025-09-20T03:10:30.444873Z","shell.execute_reply.started":"2025-09-20T03:10:30.434970Z","shell.execute_reply":"2025-09-20T03:10:30.443382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 모든 high_features 리스트의 교집합 구하기\ncommon_features = set(high_features_dict[\"high_features_0\"])\n\nfor pid in range(1, 7):  # 1~6까지\n    common_features &= set(high_features_dict[f\"high_features_{pid}\"])\n\ncommon_features = list(common_features)\n\nprint(f\"모든 파티션에 공통으로 등장한 feature 개수: {len(common_features)}\")\nprint(\"공통 feature 목록:\", common_features)\n\nfrom collections import Counter\n\n# 모든 피처를 하나의 리스트로 합치기\nall_features = []\nfor pid in range(7):\n    all_features.extend(high_features_dict[f\"high_features_{pid}\"])\n\n# 등장 횟수 세기\nfeature_counts = Counter(all_features)\n\n# 3개 이상 리스트에서 등장한 피처만 추출\ncommon_3plus_features = [f for f, cnt in feature_counts.items() if cnt >= 3]\n\nprint(f\"3개 이상 리스트에 나타난 feature 개수: {len(common_3plus_features)}\")\nprint(\"feature 목록:\", common_3plus_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T03:10:32.138984Z","iopub.execute_input":"2025-09-20T03:10:32.139527Z","iopub.status.idle":"2025-09-20T03:10:32.155157Z","shell.execute_reply.started":"2025-09-20T03:10:32.139419Z","shell.execute_reply":"2025-09-20T03:10:32.153728Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **버전1 : 공통으로 등장한 feature 20개**","metadata":{}},{"cell_type":"code","source":"# ===============================\n# A) 결측치 확인 (NaN Audit)\n# - 단위: partition_id = 0 ~ 6 (각 파티션별로 확인)\n# - 대상: FEATURES (학습에 쓸 피처 리스트)\n# - 지표: 피처별 NaN 비율의 '평균(mean)'을 주로 본다\n# ===============================\n\nimport os\nimport polars as pl\nimport pandas as pd\n\nROOT_DIR = \"/kaggle/input/jane-street-real-time-market-data-forecasting\"\n\n# ✅ 학습에 쓸 피처 리스트만 넣어주면 됨 (Top20 또는 교집합20 중 택1)\n# 예) FEATURES = top20_features  또는  FEATURES = intersect20_features\nFEATURES = [\n    'feature_61', 'feature_58', 'feature_60', 'feature_15', 'feature_08', 'feature_24', 'feature_30', \n    'feature_47', 'feature_07', 'feature_29', 'feature_38', 'feature_25', 'feature_62', \n    'feature_06', 'feature_22', 'feature_05', 'feature_23', 'feature_20', 'feature_37', 'feature_28'\n]\n\ndef na_audit(features, root_dir=ROOT_DIR, n_parts=7):\n    \"\"\"\n    아주 쉽게 설명:\n    - 각 파티션 파일을 하나씩 연다.\n    - 해당 파티션에서 'features' 각 컬럼의 NaN(빈칸) 비율을 계산한다.\n    - 모든 파티션 결과를 한 표로 모아서, 피처별 '평균 NaN 비율'을 추가로 계산한다.\n    - 평균 NaN 비율이 큰 순서대로 정렬해서 돌려준다.\n    \"\"\"\n    part_ratio = {}   # 예: {\"partition_0\": Series(피처별 NaN비율), ...}\n    for pid in range(n_parts):\n        f = os.path.join(root_dir, \"train.parquet\", f\"partition_id={pid}\", \"part-0.parquet\")\n        # 필요한 컬럼만 읽으면 빠르고 메모리 절약\n        df_pl = pl.read_parquet(f, columns=features)\n        df = df_pl.to_pandas()\n\n        # 컬럼별 NaN 비율 계산 (0.0 ~ 1.0)\n        part_ratio[f\"partition_{pid}\"] = df.isna().mean()\n\n        print(f\"=== Partition {pid} === 완료\")  # 진행상황 출력\n\n    # 파티션별 NaN 비율을 하나의 DataFrame으로 합치기 (행=피처, 열=partition_x)\n    na_df = pd.DataFrame(part_ratio)\n\n    # 피처별 평균 NaN 비율 추가(모든 파티션을 동일 가중으로 평균)\n    na_df[\"mean\"] = na_df.mean(axis=1)\n\n    # 평균 NaN 비율이 큰 순서대로 정렬 (어떤 피처가 더 비어있는지 한눈에)\n    na_df = na_df.sort_values(\"mean\", ascending=False)\n\n    return na_df\n\n# ===== 실행 =====\ncol_name = \"feature\"  # 사람이 읽기 좋게 컬럼명 표기용 문자열\nprint(f\"{col_name} (점검 대상) 개수: {len(FEATURES)}\")\nprint(f\"{col_name} (점검 대상) 목록: {sorted(FEATURES)}\\n\")\n\nna_report = na_audit(FEATURES, ROOT_DIR, n_parts=7)\n\nprint(\"\\n🔎 피처별 NaN 비율 요약 (열=각 파티션, mean=평균 NaN 비율)\")\nprint(na_report)\n\n# (선택) CSV로 저장해서 엑셀/구글시트로 확인하고 싶을 때\n# na_report.to_csv(\"/kaggle/working/na_audit_report.csv\")\n# print(\"저장: /kaggle/working/na_audit_report.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T03:44:25.168752Z","iopub.execute_input":"2025-09-20T03:44:25.169074Z","iopub.status.idle":"2025-09-20T03:44:38.702370Z","shell.execute_reply.started":"2025-09-20T03:44:25.169049Z","shell.execute_reply":"2025-09-20T03:44:38.701378Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"여기선 Nan 따로 처리하지 않는게 나을듯\n\n----------","metadata":{}},{"cell_type":"markdown","source":"# **학습**","metadata":{}},{"cell_type":"code","source":"# =========================================\n# C) LightGBM 학습 (고정 플로우)\n# - 입력: FEATURES 리스트(Top20 또는 교집합20 중 택1)\n# - 결측치: 입력 피처 NaN은 \"그대로\" 둔다 (LightGBM이 처리)\n# - 타깃 결측: responder_6 결측 행만 제거\n# - 검증: 5-Fold KFold (shuffle, random_state 고정)\n# - 보고: MSE / wMSE 의 평균 ± 표준편차\n# =========================================\n\nimport os\nimport polars as pl\nimport pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_error\n\n# 1) 데이터 로드 (필요 컬럼만) -------------------------------------------------\ncols = FEATURES + [\"responder_6\", \"weight\"]  # 학습에 필요한 컬럼만 읽는다\ndfs = []\nfor pid in range(7):  # partition_id=0~6 모두 순회\n    f = os.path.join(ROOT_DIR, \"train.parquet\", f\"partition_id={pid}\", \"part-0.parquet\")\n    df_pl = pl.read_parquet(f, columns=cols)\n    dfs.append(df_pl)\n\n# 모든 파티션을 세로로 이어붙여 하나의 표로 만든다\ndf_all = pl.concat(dfs, how=\"vertical_relaxed\").to_pandas()\n\n# 2) 타깃 결측 제거 -----------------------------------------------------------\n# - 입력 피처 NaN은 놔두고, 타깃이 비어있는 행만 제거(학습 불가)\ndf_all = df_all.dropna(subset=[\"responder_6\"])\n\n# 3) 학습용 X, y, w 준비 ------------------------------------------------------\n# - X: 선택한 FEATURES만 사용\n# - y: responder_6\n# - w: weight(샘플 가중치)\nX = df_all[FEATURES]\ny = df_all[\"responder_6\"]\nw = df_all[\"weight\"]\n\n# 4) 모델/검증 설정 -----------------------------------------------------------\n# - 하이퍼파라미터는 고정해 재현성 유지\nparams = dict(\n    n_estimators=1000,\n    learning_rate=0.05,\n    subsample=0.8,\n    colsample_bytree=0.8,\n    random_state=42,\n    n_jobs=-1,\n)\n\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\n\n# 5) K-Fold 루프 (MSE / wMSE 계산) -------------------------------------------\nmses, wmses = [], []\n\nprint(\"🚀 LightGBM 5-Fold 학습 시작\")\nfor fold, (tr, va) in enumerate(kf.split(X), 1):\n    # 학습/검증 분할\n    X_tr, X_va = X.iloc[tr], X.iloc[va]\n    y_tr, y_va = y.iloc[tr], y.iloc[va]\n    w_tr, w_va = w.iloc[tr], w.iloc[va]\n\n    # 모델 생성 (LightGBM은 입력 NaN을 자동 처리)\n    model = lgb.LGBMRegressor(**params)\n\n    # 학습 (샘플 가중치 반영)\n    model.fit(X_tr, y_tr, sample_weight=w_tr)\n\n    # 검증 예측\n    pred = model.predict(X_va)\n\n    # 성능 계산: MSE(작을수록 좋음), 가중 MSE(참고용)\n    mse = mean_squared_error(y_va, pred)\n    mse_w = mean_squared_error(y_va, pred, sample_weight=w_va)\n\n    mses.append(mse)\n    wmses.append(mse_w)\n\n    print(f\"  - Fold {fold}: MSE={mse:.6f} | wMSE={mse_w:.6f}\")\n\n# 6) 결과 요약 출력 -----------------------------------------------------------\nmse_mean, mse_std = np.mean(mses), np.std(mses)\nwmse_mean, wmse_std = np.mean(wmses), np.std(wmses)\n\nprint(\"\\n📊 CV 요약\")\nprint(f\"  MSE  : {mse_mean:.6f} ± {mse_std:.6f}\")\nprint(f\"  wMSE : {wmse_mean:.6f} ± {wmse_std:.6f}\")\n\n# (선택) 7) 전체 데이터로 최종 모델 한 번 더 적합 -----------------------------\n# - CV가 끝난 뒤, 실사용/추론 대비로 전체 데이터를 사용해 최종 모델을 만든다\nfinal_model = lgb.LGBMRegressor(**params)\nfinal_model.fit(X, y, sample_weight=w)\n# → 필요하면 pickle로 저장 가능:\n# import joblib; joblib.dump(final_model, \"/kaggle/working/lgbm_final.pkl\")\n# print(\"저장: /kaggle/working/lgbm_final.pkl\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T03:44:45.672673Z","iopub.execute_input":"2025-09-20T03:44:45.673439Z","iopub.status.idle":"2025-09-20T04:53:47.455211Z","shell.execute_reply.started":"2025-09-20T03:44:45.673401Z","shell.execute_reply":"2025-09-20T04:53:47.453913Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **버전2: 교집합 30개**","metadata":{}},{"cell_type":"code","source":"# ===============================\n# A) 결측치 확인 (버전2: 교집합 20)\n# ===============================\nimport os, polars as pl, pandas as pd\n\nROOT_DIR = \"/kaggle/input/jane-street-real-time-market-data-forecasting\"\n\n# 교집합 20개 (네가 뽑은 리스트 그대로 넣어)\nintersect20_features = [\n    'feature_06', 'feature_20', 'feature_22', 'feature_07', 'feature_28', 'feature_25', 'feature_30', 'feature_24', \n    'feature_29', 'feature_62', 'feature_61', 'feature_23', 'feature_47', 'feature_38', 'feature_15', 'feature_60', \n    'feature_64', 'feature_58', 'feature_37', 'feature_08', 'feature_59', 'feature_05', 'feature_49', 'feature_69', 'feature_56', \n    'feature_17', 'feature_33', 'feature_74', 'feature_73', 'feature_36', 'feature_21', 'feature_50', 'feature_72'\n]\n\ndef na_audit(features, root_dir=ROOT_DIR, n_parts=7):\n    \"\"\"\n    - 각 파티션 파일을 연다.\n    - 선택한 features의 NaN(빈칸) 비율을 계산한다.\n    - 파티션별 결과를 모아 피처별 '평균 NaN 비율'을 구한다.\n    - 평균 NaN 비율이 큰 순으로 정렬해 반환한다.\n    \"\"\"\n    part_ratio = {}  # {\"partition_0\": Series(피처별 NaN비율), ...}\n    for pid in range(n_parts):\n        f = os.path.join(root_dir, \"train.parquet\", f\"partition_id={pid}\", \"part-0.parquet\")\n        df_pl = pl.read_parquet(f, columns=features)  # 필요한 컬럼만\n        df = df_pl.to_pandas()\n        part_ratio[f\"partition_{pid}\"] = df.isna().mean()  # 컬럼별 NaN 비율\n        print(f\"=== Partition {pid} === 완료\")\n\n    na_df = pd.DataFrame(part_ratio)   # 행=피처, 열=partition_x\n    na_df[\"mean\"] = na_df.mean(axis=1) # 피처별 평균 NaN 비율\n    na_df = na_df.sort_values(\"mean\", ascending=False)\n    return na_df\n\n# ===== 실행 =====\ncol_name = \"feature\"\nprint(f\"{col_name} (버전2-교집합20) 개수: {len(intersect20_features)}\")\nprint(f\"{col_name} (버전2-교집합20) 목록: {sorted(intersect20_features)}\\n\")\n\nna_report = na_audit(intersect20_features, ROOT_DIR, n_parts=7)\n\nprint(\"\\n🔎 피처별 NaN 비율 요약 (열=각 파티션, mean=평균 NaN 비율)\")\nprint(na_report)\n\n# (선택) CSV 저장\n# na_report.to_csv(\"/kaggle/working/na_audit_report_v2.csv\", index=True)\n# print(\"저장: /kaggle/working/na_audit_report_v2.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T05:13:34.951292Z","iopub.execute_input":"2025-09-20T05:13:34.952341Z","iopub.status.idle":"2025-09-20T05:13:55.198389Z","shell.execute_reply.started":"2025-09-20T05:13:34.952305Z","shell.execute_reply":"2025-09-20T05:13:55.197544Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"여기도 Nan 따로 처리하지 않는게 나을듯","metadata":{}},{"cell_type":"code","source":"# =========================================\n# C) LightGBM 학습 — Version 2 (교집합 20)\n# =========================================\nimport os, polars as pl, pandas as pd, numpy as np\nimport lightgbm as lgb\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_error\n\nROOT_DIR = \"/kaggle/input/jane-street-real-time-market-data-forecasting\"\n\n# 교집합 20개\nintersect20_features = [\n    'feature_61','feature_58','feature_60','feature_15','feature_08','feature_24','feature_30',\n    'feature_47','feature_07','feature_29','feature_38','feature_25','feature_62',\n    'feature_06','feature_22','feature_05','feature_23','feature_20','feature_37','feature_28'\n]\nFEATURES = intersect20_features\n\n# 데이터 로드 (필요 컬럼만)\nneed_cols = FEATURES + [\"responder_6\",\"weight\"]\nframes = []\nfor pid in range(7):\n    f = os.path.join(ROOT_DIR,\"train.parquet\",f\"partition_id={pid}\",\"part-0.parquet\")\n    frames.append(pl.read_parquet(f, columns=need_cols))\ndf_all = pl.concat(frames, how=\"vertical_relaxed\").to_pandas()\n\n# 타깃 결측만 제거 (피처 NaN은 그대로)\ndf_all = df_all.dropna(subset=[\"responder_6\"])\nX = df_all[FEATURES]\ny = df_all[\"responder_6\"]\nw = df_all[\"weight\"]\n\n# 모델/검증 설정\nparams = dict(\n    n_estimators=1000, learning_rate=0.05,\n    subsample=0.8, colsample_bytree=0.8,\n    random_state=42, n_jobs=-1,\n    force_col_wise=True,  # 로그가 권장: 멀티스레드 오버헤드 감소\n)\n\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\n\n# 5-Fold CV\nmses, wmses = [], []\nprint(\"🚀 LightGBM 5-Fold 학습 시작 (Version2: 교집합20)\")\nfor fold, (tr, va) in enumerate(kf.split(X), 1):\n    model = lgb.LGBMRegressor(**params)\n    model.fit(X.iloc[tr], y.iloc[tr], sample_weight=w.iloc[tr])\n    pred = model.predict(X.iloc[va])\n    mse = mean_squared_error(y.iloc[va], pred)\n    wmse = mean_squared_error(y.iloc[va], pred, sample_weight=w.iloc[va])\n    mses.append(mse); wmses.append(wmse)\n    print(f\"  - Fold {fold}: MSE={mse:.6f} | wMSE={wmse:.6f}\")\n\nprint(\"\\n📊 CV 요약 (Version2)\")\nprint(f\"  MSE  : {np.mean(mses):.6f} ± {np.std(mses):.6f}\")\nprint(f\"  wMSE : {np.mean(wmses):.6f} ± {np.std(wmses):.6f}\")\n\n# (선택) 전체 데이터로 최종 적합\nfinal_model_v2 = lgb.LGBMRegressor(**params)\nfinal_model_v2.fit(X, y, sample_weight=w)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}