{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Created by <a href=\"https://github.com/yunsuxiaozi/\">yunsuxiaozi </a>  2024/11/02\n\nDue to the lack of offline cross validation in <a href=\"https://www.kaggle.com/code/yunsuxiaozi/js2024-starter\">JS2024 Starter</a> notebook, this notebook was created.What I am using here is <a href=\"https://www.kaggle.com/code/marketneutral/purged-time-series-cv-xgboost-optuna\">Purged Time Series CV</a>.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"# <span><h1 style = \"font-family: garamond; font-size: 40px; font-style: normal; letter-spcaing: 3px; background-color: #f6f5f5; color :#fe346e; border-radius: 100px 100px; text-align:center\">Import Libraries</h1></span>","metadata":{}},{"cell_type":"code","source":"import polars as pl#similar to pandas, but with better performance when dealing with large datasets.\nimport pandas as pd#read csv,parquet\nimport numpy as np#for scientific computation of matrices\n#model\nfrom  lightgbm import LGBMRegressor\nimport gc#rubbish collection\nimport os#interact with operation system\n#environment provided by competition hoster\nimport kaggle_evaluation.jane_street_inference_server\nimport warnings#avoid some negligible errors\n#The filterwarnings () method is used to set warning filters, which can control the output method and level of warning information.\nwarnings.filterwarnings('ignore')\n\nimport random#provide some function to generate random_seed.\n#set random seed,to make sure model can be recurrented.\ndef seed_everything(seed):\n    np.random.seed(seed)#numpy's random seed\n    random.seed(seed)#python built-in random seed\nseed_everything(seed=2024)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span><h1 style = \"font-family: garamond; font-size: 40px; font-style: normal; letter-spcaing: 3px; background-color: #f6f5f5; color :#fe346e; border-radius: 100px 100px; text-align:center\">Origin data</h1></span>\n\nConsidering the memory issue, only the data from three files will be used here. Although it is possible to use function 'reduce_mem_usage', I found that it cannot be opened again after being saved as a parquet file.\n","metadata":{}},{"cell_type":"code","source":"all_datas=[]\nfor idx in [7,8,9]:\n    train=pl.read_parquet(f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={idx}/part-0.parquet\")\n    train=train.to_pandas()\n    train=train[train['date_id']>=1320]\n    print(f\"train.shape:{train.shape}\")\n    all_datas.append(train)\nall_datas=pd.concat(all_datas)\nprint(f\"len(all_datas):{len(all_datas)}\")\nall_datas.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span><h1 style = \"font-family: garamond; font-size: 40px; font-style: normal; letter-spcaing: 3px; background-color: #f6f5f5; color :#fe346e; border-radius: 100px 100px; text-align:center\">train_test_split</h1></span>\n\nBecause the CV method already found, the test set is only used to test whether the program runs smoothly, and most of the data here is used for training.\n","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.max_columns',100)\n\ntrain_cols=['date_id', 'time_id', 'symbol_id', 'weight', 'feature_00', 'feature_01', 'feature_02', 'feature_03', 'feature_04', 'feature_05', 'feature_06', 'feature_07', 'feature_08', 'feature_09', 'feature_10', 'feature_11', 'feature_12', 'feature_13', 'feature_14', 'feature_15', 'feature_16', 'feature_17', 'feature_18', 'feature_19', 'feature_20', 'feature_21', 'feature_22', 'feature_23', 'feature_24', 'feature_25', 'feature_26', 'feature_27', 'feature_28', 'feature_29', 'feature_30', 'feature_31', 'feature_32', 'feature_33', 'feature_34', 'feature_35', 'feature_36', 'feature_37', 'feature_38', 'feature_39', 'feature_40', 'feature_41', 'feature_42', 'feature_43', 'feature_44', 'feature_45', 'feature_46', 'feature_47', 'feature_48', 'feature_49', 'feature_50', 'feature_51', 'feature_52', 'feature_53', 'feature_54', 'feature_55', 'feature_56', 'feature_57', 'feature_58', 'feature_59', 'feature_60', 'feature_61', 'feature_62', 'feature_63', 'feature_64', 'feature_65', 'feature_66', 'feature_67', 'feature_68', 'feature_69', 'feature_70', 'feature_71', 'feature_72', 'feature_73', 'feature_74', 'feature_75', 'feature_76', 'feature_77', 'feature_78', 'responder_0', 'responder_1', 'responder_2', 'responder_3', 'responder_4', 'responder_5', 'responder_6', 'responder_7', 'responder_8']\ntest_cols=['row_id', 'date_id','time_id', 'symbol_id', 'weight', 'is_scored', 'feature_00', 'feature_01', 'feature_02', 'feature_03', 'feature_04', 'feature_05', 'feature_06', 'feature_07', 'feature_08', 'feature_09', 'feature_10', 'feature_11', 'feature_12', 'feature_13', 'feature_14', 'feature_15', 'feature_16', 'feature_17', 'feature_18', 'feature_19', 'feature_20', 'feature_21', 'feature_22', 'feature_23', 'feature_24', 'feature_25', 'feature_26', 'feature_27', 'feature_28', 'feature_29', 'feature_30', 'feature_31', 'feature_32', 'feature_33', 'feature_34', 'feature_35', 'feature_36', 'feature_37', 'feature_38', 'feature_39', 'feature_40', 'feature_41', 'feature_42', 'feature_43', 'feature_44', 'feature_45', 'feature_46', 'feature_47', 'feature_48', 'feature_49', 'feature_50', 'feature_51', 'feature_52', 'feature_53', 'feature_54', 'feature_55', 'feature_56', 'feature_57', 'feature_58', 'feature_59', 'feature_60', 'feature_61', 'feature_62', 'feature_63', 'feature_64', 'feature_65', 'feature_66', 'feature_67', 'feature_68', 'feature_69', 'feature_70', 'feature_71', 'feature_72', 'feature_73', 'feature_74', 'feature_75', 'feature_76', 'feature_77', 'feature_78']\nlags_cols=['date_id', 'time_id','symbol_id', 'responder_0_lag_1', 'responder_1_lag_1', 'responder_2_lag_1', 'responder_3_lag_1', 'responder_4_lag_1', 'responder_5_lag_1', 'responder_6_lag_1', 'responder_7_lag_1', 'responder_8_lag_1']\nsplit=1695\ntest=all_datas[all_datas['date_id']>=split]\ntrain=all_datas[all_datas['date_id']<split]\ndel all_datas\ngc.collect()\n\nprint(\"< train_lags.parquet >\")\nfor i in range(9):\n    train[f'responder_{i}_lag_1']=train[f'responder_{i}']\ntrain_lags=train[lags_cols].reset_index(drop=True)\ntrain_lags['date_id']=train_lags['date_id']-train_lags['date_id'].min()+1\ntrain_lags=train_lags[train_lags['date_id']>=1]\n\nprint(\"< train.parquet >\")\ntrain=train[train_cols].reset_index(drop=True)\ntrain['date_id']=train['date_id']-train['date_id'].min()\ntrain=train[train['date_id']>=1]\n\nprint(\"< lags.parquet >\")\nfor i in range(9):\n    test[f'responder_{i}_lag_1']=test[f'responder_{i}']\nlags=test[lags_cols].reset_index(drop=True)\nlags['date_id']=lags['date_id']-split\nlags=lags[lags['date_id']>=1]\n\nprint(\"< test.parquet >\")\ntest['is_scored']=True\ntest['row_id']=np.arange(len(test))\ntest['date_id']=test['date_id']-split-1\ntest=test[test['date_id']>=1]\ntest=test[test_cols].reset_index(drop=True)\n\nprint(f\"len(train):{len(train)},len(train_lags):{len(train_lags)},len(test):{len(test)},len(lags):{len(lags)}\")\ntest.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span><h1 style = \"font-family: garamond; font-size: 40px; font-style: normal; letter-spcaing: 3px; background-color: #f6f5f5; color :#fe346e; border-radius: 100px 100px; text-align:center\">Save kfold data</h1></span>","metadata":{}},{"cell_type":"code","source":"os.makedirs('synthetic_train.parquet', exist_ok=True) \nos.makedirs('synthetic_valid.parquet', exist_ok=True) \n\nnum_folds=3\ntrain_gap_time=21#3weeks\ngap_time=15#half of the month\ntrain_time=183#half of the year\n\nassert (num_folds-1)*train_gap_time+train_time+gap_time+123<=train['date_id'].max()-train['date_id'].min()\nassert num_folds*train_gap_time+train_time+gap_time<=train['date_id'].max()-train['date_id'].min()\n\nfor fold in range(num_folds):\n    \n    os.makedirs(f'synthetic_train.parquet/fold={fold}/', exist_ok=True) \n    os.makedirs(f'synthetic_train_lags.parquet/fold={fold}/', exist_ok=True) \n    os.makedirs(f'synthetic_valid.parquet/fold={fold}/', exist_ok=True) \n    os.makedirs(f'synthetic_valid_lags.parquet/fold={fold}/', exist_ok=True)\n    \n    train_start_time=train['date_id'].min()+fold*train_gap_time\n    train_end_time=train_start_time+train_time\n    valid_start_time=train_end_time+gap_time\n    valid_end_time=valid_start_time+122#around 450 test datas\n    print(f\"train_start_time:{train_start_time},train_end_time:{train_end_time}\")\n    print(f\"valid_start_time:{valid_start_time},valid_end_time:{valid_end_time}\")\n\n    print(f\"< fold{fold+1} train data >\")\n    train_fold=train[(train['date_id']>=train_start_time)&(train['date_id']<train_end_time)].copy()\n    train_lags_fold=train_lags[(train_lags['date_id']>=train_start_time)&(train_lags['date_id']<train_end_time)].copy()\n    train_fold.to_parquet(f'synthetic_train.parquet/fold={fold}/part-0.parquet',index=None)\n    train_lags_fold.to_parquet(f'synthetic_train_lags.parquet/fold={fold}/part-0.parquet',index=None)\n    \n    print(f\"< fold{fold+1} valid data >\")\n    valid_fold=train[(train['date_id']>=valid_start_time)&(train['date_id']<valid_end_time)].copy()\n    valid_lags_fold=train_lags[(train_lags['date_id']>=valid_start_time)&(train_lags['date_id']<valid_end_time)].copy()\n    valid_fold.to_parquet(f'synthetic_valid.parquet/fold={fold}/part-0.parquet',index=None)\n    valid_lags_fold.to_parquet(f'synthetic_valid_lags.parquet/fold={fold}/part-0.parquet',index=None)\n    \n    print(f\"train_fold.shape:{train_fold.shape},train_lags_fold.shape:{train_lags_fold.shape}\")\n    print(f\"valid_fold.shape:{valid_fold.shape},valid_lags_fold.shape:{valid_lags_fold.shape}\")\n    print(\"-\"*30)\n    del train_fold,train_lags_fold,valid_fold,valid_lags_fold\n    gc.collect()\n\nprint(\"< final train data >\")\ntrain_start_time=train['date_id'].max()-gap_time-train_time\ntrain_end_time=train_start_time+train_time\nprint(f\"train_start_time:{train_start_time}\")\nprint(f\"train_end_time:{train_end_time}\")\ntrain_final=train[(train['date_id']>=train_start_time)&(train['date_id']<train_end_time)].copy()\ntrain_lags_final=train_lags[(train_lags['date_id']>=train_start_time)&(train_lags['date_id']<train_end_time)].copy()\ntrain_final.to_parquet(f'synthetic_train.parquet/part-0.parquet',index=None)\ntrain_lags_final.to_parquet(f'synthetic_train_lags.parquet/part-0.parquet',index=None)\ndel train_final,train_lags_final,train\ngc.collect()\n\nprint(\"< sample test data >\")\nos.makedirs('synthetic_test.parquet', exist_ok=True) \nos.makedirs('synthetic_lags.parquet', exist_ok=True) \nfor d in range(1,int(test['date_id'].max()+1)):\n    os.makedirs(f'synthetic_test.parquet/date_id={d-1}', exist_ok=True) \n    os.makedirs(f'synthetic_lags.parquet/date_id={d-1}', exist_ok=True) \n    d_test,d_lags=test[test['date_id']==d].copy(),lags[lags['date_id']==d].copy()\n    d_test.to_parquet(f'synthetic_test.parquet/date_id={d-1}/part-0.parquet',index=None)\n    d_lags.to_parquet(f'synthetic_lags.parquet/date_id={d-1}/part-0.parquet',index=None)\n    print(f\"len(d_test):{len(d_test)},len(d_lags):{len(d_lags)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span><h1 style = \"font-family: garamond; font-size: 40px; font-style: normal; letter-spcaing: 3px; background-color: #f6f5f5; color :#fe346e; border-radius: 100px 100px; text-align:center\">Purgedkfold CV</h1></span>\n\nHere we use origin features as simple baseline and fill nan value with -1.\n","metadata":{}},{"cell_type":"code","source":"lgb_params={\"boosting_type\": \"gbdt\",\"metric\": 'rmse',\n            'random_state': 2025,  \"max_depth\": 10,\"learning_rate\": 0.1,\n            \"n_estimators\": 125,\"colsample_bytree\": 0.6,\"colsample_bynode\": 0.6,\"verbose\": -1,\"reg_alpha\": 0.2,\n            \"reg_lambda\": 5,\"extra_trees\":True,'num_leaves':64,\"max_bin\":255,\n            'device':'gpu','gpu_use_dp':True,\n            }\ndef custom_metric(y_true,y_pred,weight):\n    weighted_r2=1-(np.sum(weight*(y_true-y_pred)**2)/np.sum(weight*y_true**2))\n    return weighted_r2\norigin_feats=[f'feature_0{i}' if i<10 else f'feature_{i}' for i in range(79)]\nnum_folds_score=[]\nfor fold in range(num_folds):\n    print(f\"fold:{fold}\")\n    train=pl.read_parquet(f\"synthetic_train.parquet/fold={fold}/part-0.parquet\")\n    train=train.to_pandas()\n    print(f\"train.shape:{train.shape}\")\n    \n    # train_lags=pl.read_parquet(f\"synthetic_train_lags.parquet/fold={fold}/part-0.parquet\")\n    # train_lags=train_lags.to_pandas()\n    # print(f\"train_lags.shape:{train_lags.shape}\")\n\n    valid=pl.read_parquet(f\"synthetic_valid.parquet/fold={fold}/part-0.parquet\")\n    valid=valid.to_pandas()\n    print(f\"valid.shape:{valid.shape}\")\n    \n    # valid_lags=pl.read_parquet(f\"synthetic_valid_lags.parquet/fold={fold}/part-0.parquet\")\n    # valid_lags=valid_lags.to_pandas()\n    # print(f\"valid_lags.shape:{valid_lags.shape}\")\n    #xgboost -1\n    train,valid=train[origin_feats+['responder_6']].fillna(-1),valid[origin_feats+['responder_6','weight']].fillna(-1)\n    X=train[origin_feats].values\n    y=train['responder_6'].values\n    valid_X=valid[origin_feats].values\n    valid_y=valid['responder_6'].values\n    valid_weight=valid['weight'].values\n    model= LGBMRegressor(**lgb_params)\n    model.fit(X,y)\n    valid_preds=model.predict(valid_X)\n    weighted_r2=custom_metric(valid_y,valid_preds,valid_weight)\n    print(f\"weighted_r2:{weighted_r2}\")\n    num_folds_score.append(weighted_r2)\n    del train,valid,X,y,valid_X,valid_y,valid_weight,model\n    gc.collect()\nprint(f\"mean {num_folds} folds CV_score:{np.mean(num_folds_score)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span><h1 style = \"font-family: garamond; font-size: 40px; font-style: normal; letter-spcaing: 3px; background-color: #f6f5f5; color :#fe346e; border-radius: 100px 100px; text-align:center\">Final Submission</h1></span>","metadata":{}},{"cell_type":"code","source":"train=pl.read_parquet('synthetic_train.parquet/part-0.parquet')\ntrain=train.to_pandas()\nprint(f\"train.shape:{train.shape}\")\ntrain_lags=pl.read_parquet(\"synthetic_train_lags.parquet/part-0.parquet\")\ntrain_lags=train_lags.to_pandas()\nprint(f\"train_lags.shape:{train_lags.shape}\")\ntrain=train[origin_feats+['responder_6']].fillna(-1)\nX=train[origin_feats].values\ny=train['responder_6'].values\nlgb= LGBMRegressor(**lgb_params)\nlgb.fit(X,y)\ndel train,train_lags,X,y\ngc.collect()\n\ndef predict(test,lags):\n    global lgb\n    \n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    test=test.to_pandas()\n    test=test.fillna(-1)\n    test=test[origin_feats]\n    eps=1e-10\n    test_preds=np.clip(lgb.predict(test),-5+eps,5-eps)\n    predictions = predictions.with_columns(pl.Series('responder_6', test_preds.ravel()))\n    return predictions\n\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            'synthetic_test.parquet/',\n            'synthetic_lags.parquet/',\n        )\n    )","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}