{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":11305158,"sourceType":"competition"},{"sourceId":13931036,"sourceType":"datasetVersion","datasetId":8862728}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install /kaggle/input/janestreet2025-code/janestreet-0.1-py3-none-any.whl --force-reinstall --no-deps","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T03:24:52.983207Z","iopub.execute_input":"2025-12-01T03:24:52.983406Z","iopub.status.idle":"2025-12-01T03:24:55.619268Z","shell.execute_reply.started":"2025-12-01T03:24:52.983369Z","shell.execute_reply":"2025-12-01T03:24:55.618456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nimport lightgbm as lgb\nimport polars as pl\nimport numpy as np\nimport os\nimport joblib\nimport gc\nimport time\n\nfrom janestreet.data_processor import DataProcessor\nfrom janestreet.config import PATH_MODELS\nfrom janestreet.utils import create_folder\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-01T03:24:55.621694Z","iopub.execute_input":"2025-12-01T03:24:55.622033Z","iopub.status.idle":"2025-12-01T03:25:03.772661Z","shell.execute_reply.started":"2025-12-01T03:24:55.621984Z","shell.execute_reply":"2025-12-01T03:25:03.771857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"MODEL_NAME = \"lgbm_full_v1\"\nTARGET_COL = \"responder_6\"\nSKIP_DAYS = 677\n\nLGB_PARAMS = {\n    \"objective\": \"regression_l2\",\n    \"metric\": \"rmse\",\n    \"learning_rate\": 0.05,\n    \"num_leaves\": 62,\n    \"feature_fraction\": 0.8,\n    \"bagging_fraction\": 0.8,\n    \"bagging_freq\": 5,\n    \"lambda_l1\": 1.0,\n    \"lambda_l2\": 1.0,\n    \"verbosity\": -1,\n    \"seed\": 42,\n    \"n_jobs\": -1,\n    'device' : 'gpu',\n    'gpu_use_dp': True,\n}\n\ndef r2_weighted(y_true, y_pred, weights):\n    if len(y_true) == 0:\n        return np.nan\n    weights = np.array(weights)\n    y_true = np.array(y_true)\n    y_pred = np.array(y_pred)\n\n    ss_res = np.sum(weights * (y_true - y_pred) ** 2)\n    weighted_mean = np.average(y_true, weights=weights)\n    ss_tot = np.sum(weights * (y_true - weighted_mean) ** 2)\n\n    return 1 - ss_res / (ss_tot + 1e-38)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T03:25:03.773663Z","iopub.execute_input":"2025-12-01T03:25:03.774233Z","iopub.status.idle":"2025-12-01T03:25:03.779513Z","shell.execute_reply.started":"2025-12-01T03:25:03.774210Z","shell.execute_reply":"2025-12-01T03:25:03.778778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"t0 = time.time()\nprint(\"加载 partition 0–8 训练数据...\")\n\nprocessor = DataProcessor(\n    name=MODEL_NAME,\n    skip_days=SKIP_DAYS,\n    use_aux_targets=False\n)\n\ndf_all = processor.get_train_valid_data()\nfeatures = processor.features\n\nprint(f\"训练数据行数: {df_all.height}\")\nprint(f\"特征数量: {len(features)}\")\nprint(f\"数据加载时间: {time.time() - t0:.2f} sec\")\n\n\ndf_all = df_all.with_columns([pl.col(c).cast(pl.Float32) for c in features])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T03:25:03.780332Z","iopub.execute_input":"2025-12-01T03:25:03.780645Z","iopub.status.idle":"2025-12-01T03:26:23.260098Z","shell.execute_reply.started":"2025-12-01T03:25:03.780620Z","shell.execute_reply":"2025-12-01T03:26:23.259419Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"t1 = time.time()\nprint(\"\\n提取训练矩阵...\")\n\nX_train = df_all.select(features).to_numpy()\ny_train = df_all.select(TARGET_COL).to_numpy().flatten()\nw_train = df_all.select(\"weight\").to_numpy().flatten()\n\nprint(f\"矩阵构建时间: {time.time() - t1:.2f} sec\")\n\ntrain_set = lgb.Dataset(\n    X_train,\n    label=y_train,\n    weight=w_train,\n    feature_name=features\n)\n\ndel df_all\ngc.collect()\n\n\n# ====================== 3. FULL 训练 ======================\nt2 = time.time()\nprint(\"\\n开始 FULL LightGBM 训练...\")\n\nmodel = lgb.train(\n    params=LGB_PARAMS,\n    train_set=train_set,\n    num_boost_round=300,\n    valid_sets=[train_set],\n    valid_names=[\"train\"],\n    callbacks=[lgb.log_evaluation(period=50)]\n)\n\nprint(f\"\\n训练完成，耗时: {time.time() - t2:.2f} sec\")\n\ncreate_folder(PATH_MODELS)\n\nsave_txt = os.path.join(PATH_MODELS, f\"{MODEL_NAME}.txt\")\nsave_joblib = os.path.join(PATH_MODELS, f\"{MODEL_NAME}.joblib\")\n\nmodel.save_model(save_txt)\njoblib.dump(model, save_joblib)\n\nprint(f\"模型已保存到: {save_txt}\")\nprint(f\"Joblib 已保存到: {save_joblib}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T03:26:23.260957Z","iopub.execute_input":"2025-12-01T03:26:23.261278Z","iopub.status.idle":"2025-12-01T03:38:41.600046Z","shell.execute_reply.started":"2025-12-01T03:26:23.261258Z","shell.execute_reply":"2025-12-01T03:38:41.599430Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del X_train, y_train, w_train, train_set \ngc.collect() \nprint(\"已释放训练数据内存.\")\n\ndf_test = processor.get_test_data()\n\ndf_test = df_test.with_columns([pl.col(c).cast(pl.Float32) for c in features])\n\nprint(f\"LB 测试集行数: {df_test.height}\")\n\nX_test = df_test.select(features).to_numpy()\ny_test = df_test.select(TARGET_COL).to_numpy().flatten()\nw_test = df_test.select(\"weight\").to_numpy().flatten()\n\nprint(\"开始预测 partition 9 ...\")\ny_pred_test = model.predict(X_test)\n\nlb_score = r2_weighted(y_test, y_pred_test, w_test)\nprint(f\"\\n===== LB Score (Weighted R²) = {lb_score:.6f} =====\")\n\ndel df_test, X_test, y_test, w_test\ngc.collect()\n\nprint(f\"\\n总耗时: {time.time() - t0:.2f} sec\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T03:38:41.600741Z","iopub.execute_input":"2025-12-01T03:38:41.600941Z","iopub.status.idle":"2025-12-01T03:40:15.510665Z","shell.execute_reply.started":"2025-12-01T03:38:41.600924Z","shell.execute_reply":"2025-12-01T03:40:15.510000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n\nimportance_gain = model.feature_importance(importance_type='gain')\nimportance_split = model.feature_importance(importance_type='split')\n\ndf_importance = pd.DataFrame({\n    'feature': features,\n    'gain': importance_gain,\n    'split': importance_split\n})\n\n# 按 gain 排序\ndf_importance = df_importance.sort_values('gain', ascending=False)\n\nprint(\"\\nTop 30 Important Features (by Gain):\")\nprint(df_importance.head(30))\n\n# ========== 可视化 ==========\nplt.figure(figsize=(10, 12))\nplt.barh(df_importance.head(30)['feature'], df_importance.head(30)['gain'])\nplt.gca().invert_yaxis()\nplt.title(\"Top 30 Feature Importance (Gain)\")\nplt.xlabel(\"Gain Importance\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T04:10:03.950254Z","iopub.execute_input":"2025-12-01T04:10:03.950948Z","iopub.status.idle":"2025-12-01T04:10:04.442366Z","shell.execute_reply.started":"2025-12-01T04:10:03.950921Z","shell.execute_reply":"2025-12-01T04:10:04.441656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_importance['gain_norm'] = df_importance['gain'] / df_importance['gain'].sum()\ndf_importance['gain_cumsum'] = df_importance['gain_norm'].cumsum()\n\nprint(df_importance[['feature','gain_cumsum']].head(40))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T04:11:05.593136Z","iopub.execute_input":"2025-12-01T04:11:05.593950Z","iopub.status.idle":"2025-12-01T04:11:05.607768Z","shell.execute_reply.started":"2025-12-01T04:11:05.593919Z","shell.execute_reply":"2025-12-01T04:11:05.606965Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"top50_df = df_importance[df_importance['gain_cumsum'] <= 0.50]\nprint(top50_df['feature'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T04:16:01.256341Z","iopub.execute_input":"2025-12-01T04:16:01.257042Z","iopub.status.idle":"2025-12-01T04:16:01.262715Z","shell.execute_reply.started":"2025-12-01T04:16:01.257016Z","shell.execute_reply":"2025-12-01T04:16:01.261770Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_importance.tail()['feature']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T04:24:27.569925Z","iopub.execute_input":"2025-12-01T04:24:27.570465Z","iopub.status.idle":"2025-12-01T04:24:27.576004Z","shell.execute_reply.started":"2025-12-01T04:24:27.570440Z","shell.execute_reply":"2025-12-01T04:24:27.575352Z"}},"outputs":[],"execution_count":null}]}