{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":4.669361,"end_time":"2024-10-10T13:05:46.686069","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-10-10T13:05:42.016708","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport pandas as pd\nimport lightgbm as lgb\n\nimport os\n\nimport kaggle_evaluation.jane_street_inference_server\n\nimport joblib\nfrom sklearn.model_selection import train_test_split\nimport xgboost as xgb\nimport matplotlib.pyplot as plt\nfrom sklearn.linear_model import Ridge\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom  lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor\nfrom xgboost import XGBRegressor\nimport gc\nimport pickle\n\n\ndef reduce_mem_usage(df:pd.DataFrame, float16_as32:bool=True)->pd.DataFrame:\n    #memory_usage()是df每列的内存使用量,sum是对它们求和, B->KB->MB\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    for col in df.columns:#遍历每列的列名\n        col_type = df[col].dtype#列名的type\n        if col_type != object and str(col_type)!='category':#不是object也就是说这里处理的是数值类型的变量\n            c_min,c_max = df[col].min(),df[col].max() #求出这列的最大值和最小值\n            if str(col_type)[:3] == 'int':#如果是int类型的变量,不管是int8,int16,int32还是int64\n                #如果这列的取值范围是在int8的取值范围内,那就对类型进行转换 (-128 到 127)\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                #如果这列的取值范围是在int16的取值范围内,那就对类型进行转换(-32,768 到 32,767)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                #如果这列的取值范围是在int32的取值范围内,那就对类型进行转换(-2,147,483,648到2,147,483,647)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                #如果这列的取值范围是在int64的取值范围内,那就对类型进行转换(-9,223,372,036,854,775,808到9,223,372,036,854,775,807)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:#如果是浮点数类型.\n                #如果数值在float16的取值范围内,如果觉得需要更高精度可以考虑float32\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    if float16_as32:#如果数据需要更高的精度可以选择float32\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        df[col] = df[col].astype(np.float16)  \n                #如果数值在float32的取值范围内，对它进行类型转换\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                #如果数值在float64的取值范围内，对它进行类型转换\n                else:\n                    df[col] = df[col].astype(np.float64)\n    #calculate memory after optimization\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    return df","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":1.223703,"end_time":"2024-10-10T13:05:45.825911","exception":false,"start_time":"2024-10-10T13:05:44.602208","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T02:49:59.688342Z","iopub.execute_input":"2024-11-26T02:49:59.688762Z","iopub.status.idle":"2024-11-26T02:50:06.433314Z","shell.execute_reply.started":"2024-11-26T02:49:59.688714Z","shell.execute_reply":"2024-11-26T02:50:06.432630Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CONFIG:\n    target_col = \"responder_6\"\n    lag_cols_original = [\"date_id\", \"symbol_id\"] + [f\"responder_{idx}\" for idx in range(9)]\n    lag_cols_rename = { f\"responder_{idx}\" : f\"responder_{idx}_lag_1\" for idx in range(9)}\n    valid_ratio = 0.05\n    start_dt = 600\n\nfeatures=['symbol_id','sin_time_id','cos_time_id','sin_time_id_halfday','cos_time_id_halfday'] \\\n    +[f\"feature_{idx:02d}\" for idx in range(79)] \\\n    # +[f\"responder_{idx}_lag_1\" for idx in range(9)]\n\n\ndata=[]\nfor i in [6,7,8,9]:\n    train=pl.read_parquet(f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}/part-0.parquet\")\n    train=train.to_pandas()\n    train['sin_time_id']=np.sin(2*np.pi*train['time_id']/967)\n    train['cos_time_id']=np.cos(2*np.pi*train['time_id']/967)\n    train['sin_time_id_halfday']=np.sin(2*np.pi*train['time_id']/483)\n    train['cos_time_id_halfday']=np.cos(2*np.pi*train['time_id']/483)\n    train=train.fillna(-1)\n    train=reduce_mem_usage(train,float16_as32=False)\n    data.append(train)\ntrain=pd.concat(data)\nprint(f\"train.shape:{train.shape}\")\ndel data\ngc.collect()\n\n\nfor i in range(9):\n    train[f'responder_{i}_lag_1']=train[f'responder_{i}'].shift(1)\n    \ntrain=train[['responder_6','weight']+features]\n\ntrain=train.fillna(-1)\ngc.collect()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T02:50:06.435036Z","iopub.execute_input":"2024-11-26T02:50:06.435962Z","iopub.status.idle":"2024-11-26T02:51:32.910915Z","shell.execute_reply.started":"2024-11-26T02:50:06.435918Z","shell.execute_reply":"2024-11-26T02:51:32.910106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# X=train[features].fillna(3).values\n# y=train['responder_6'].values\n# weights=train['weight'].values\n\n# T=len(X)\n# split_ratio=0.95\n# X_train, X_test, y_train, y_test,weight_train,weight_test = X[:int(T*split_ratio)],X[int(T*split_ratio):],y[:int(T*split_ratio)],y[int(T*split_ratio):],weights[:int(T*split_ratio)],weights[int(T*split_ratio):],\n# # X_train, X_test, y_train, y_test,weight_train,weight_test = X,X[int(T*split_ratio):],y,y[int(T*split_ratio):],weights,weights[int(T*split_ratio):],\n\n# del X,y,train","metadata":{"execution":{"iopub.status.busy":"2024-11-26T02:51:32.912327Z","iopub.execute_input":"2024-11-26T02:51:32.912674Z","iopub.status.idle":"2024-11-26T02:51:32.916669Z","shell.execute_reply.started":"2024-11-26T02:51:32.912636Z","shell.execute_reply":"2024-11-26T02:51:32.915866Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_params={\"boosting_type\": \"gbdt\",\"metric\": 'rmse',\n            'random_state': 2025,  \"max_depth\": 6,\"learning_rate\": 0.05,\n            \"n_estimators\": 120,\"colsample_bytree\": 0.6,\"colsample_bynode\": 0.6,\"verbose\": 1,\"reg_alpha\": 0.1,\n            \"reg_lambda\": 5,\"extra_trees\":True,'num_leaves':64,\"max_bin\":255,\n            'device':'gpu','gpu_use_dp':True,\n            }\n\ncat_params={'task_type':'GPU',\n           'random_state':2025,\n           'eval_metric'         : 'RMSE',\n           'bagging_temperature' : 0.50,\n           'iterations'          : 200,\n           'learning_rate'       : 0.1,\n           'max_depth'           : 12,\n           'l2_leaf_reg'         : 1.25,\n           'min_data_in_leaf'    : 24,\n           'random_strength'     : 0.25, \n           'verbose'             : 0,\n          }\nxgb_params={'random_state': 2025, 'n_estimators': 100, \n            'learning_rate': 0.1, 'max_depth': 10,\n            'reg_alpha': 0.08, 'reg_lambda': 0.8, \n            'subsample': 0.95, 'colsample_bytree': 0.6, \n            'min_child_weight': 3,\n            'tree_method':'gpu_hist',\n           }\n# print(\"lgb\")\n# lgb=LGBMRegressor(**lgb_params)\n# lgb.fit(train[features].values,train['responder_6'].values)\nprint(\"cat\")\ncat=CatBoostRegressor(**cat_params)\ncat.fit(train[features].values,train['responder_6'].values)\nprint(\"xgb\")\nxgb=XGBRegressor(**xgb_params)\nxgb.fit(train[features].values,train['responder_6'].values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T02:51:32.917563Z","iopub.execute_input":"2024-11-26T02:51:32.917775Z","iopub.status.idle":"2024-11-26T02:57:28.320479Z","shell.execute_reply.started":"2024-11-26T02:51:32.917752Z","shell.execute_reply":"2024-11-26T02:57:28.319556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def custom_metric(y_true,y_pred,weight):\n    y_true=np.float128(y_true)\n    y_pred=np.float128(y_pred)\n    weight=np.float128(weight)\n    weighted_r2=1-(np.sum(weight*(y_true-y_pred)**2)/np.sum(weight*y_true**2))\n    return weighted_r2\n\n# def predict_models(test,cols):\neps=1e-10\ntest=train.iloc[::10]\ncols=features\n# lgb_res=lgb.predict(test[cols].values)\ncat_res=cat.predict(test[cols].values)\nxgb_res=xgb.predict(test[cols].values)\n# lgb_res=np.clip(lgb_res,-5+eps,5-eps)\ncat_res=np.clip(cat_res,-5+eps,5-eps)\nxgb_res=np.clip(xgb_res,-5+eps,5-eps)\n\ntest_preds=(cat_res+xgb_res)/2\ny_true=test['responder_6'].values\nweight=test['weight'].values\n# print(f\"lgb loss: {custom_metric(y_true,lgb_res,weight)}\")\nprint(f\"cat loss: {custom_metric(y_true,cat_res,weight)}\")\nprint(f\"xgb loss: {custom_metric(y_true,xgb_res,weight)}\")\nprint(f\"overall loss: {custom_metric(y_true,test_preds,weight)}\")\n\n# predict_models(train.iloc[::10],features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T02:57:28.322948Z","iopub.execute_input":"2024-11-26T02:57:28.323359Z","iopub.status.idle":"2024-11-26T02:57:36.510617Z","shell.execute_reply.started":"2024-11-26T02:57:28.323314Z","shell.execute_reply":"2024-11-26T02:57:36.509591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict_no_lags(test,lags):\n    cols=[\"symbol_id\", \"weight\"] \\\n    +[f\"feature_{idx:02d}\" for idx in range(79)] \\\n    +[f\"responder_{idx}_lag_1\" for idx in range(9)]\n    \n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    test = test.with_columns(\n            ( pl.lit(0.0).alias(f'responder_{idx}_lag_1') for idx in range(9))\n        )\n    test_preds=model.predict(test[cols].to_pandas().fillna(3).values)\n    predictions = predictions.with_columns(pl.Series('responder_6', test_preds.ravel()))\n    return predictions\n\ndef predict(test,lags):\n    cols=['symbol_id','sin_time_id','cos_time_id','sin_time_id_halfday','cos_time_id_halfday'] \\\n    +[f\"feature_{idx:02d}\" for idx in range(79)] \\\n    # +[f\"responder_{idx}_lag_1\" for idx in range(9)]\n\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n\n    if not lags is None:\n        lags = lags.group_by([\"date_id\", \"symbol_id\"], maintain_order=True).last() # pick up last record of previous date\n        test = test.join(lags, on=[\"date_id\", \"symbol_id\"],  how=\"left\")\n    else:\n        test = test.with_columns(\n            ( pl.lit(0.0).alias(f'responder_{idx}_lag_1') for idx in range(9))\n        )\n    test=test.to_pandas()\n    test['sin_time_id']=np.sin(2*np.pi*test['time_id']/967)\n    test['cos_time_id']=np.cos(2*np.pi*test['time_id']/967)\n    test['sin_time_id_halfday']=np.sin(2*np.pi*test['time_id']/483)\n    test['cos_time_id_halfday']=np.cos(2*np.pi*test['time_id']/483)\n    test=test.fillna(-1)\n    # print(test)\n    eps=1e-10\n    test_preds=(cat.predict(test[cols].values)+xgb.predict(test[cols].values))/2\n    test_preds=np.clip(test_preds,-5+eps,5-eps)\n    predictions = predictions.with_columns(pl.Series('responder_6', test_preds.ravel()))\n    \n    return predictions\n\n# function confirmation\ntest=pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet/date_id=0/part-0.parquet\")\nlags=pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet/date_id=0/part-0.parquet\")\npredict(test,lags)","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.015917,"end_time":"2024-10-10T13:05:45.848958","exception":false,"start_time":"2024-10-10T13:05:45.833041","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T02:57:36.511635Z","iopub.execute_input":"2024-11-26T02:57:36.511919Z","iopub.status.idle":"2024-11-26T02:57:36.676324Z","shell.execute_reply.started":"2024-11-26T02:57:36.511892Z","shell.execute_reply":"2024-11-26T02:57:36.675433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.308219,"end_time":"2024-10-10T13:05:46.163573","exception":false,"start_time":"2024-10-10T13:05:45.855354","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2024-11-26T02:57:36.677348Z","iopub.execute_input":"2024-11-26T02:57:36.677615Z","iopub.status.idle":"2024-11-26T02:57:36.883116Z","shell.execute_reply.started":"2024-11-26T02:57:36.677589Z","shell.execute_reply":"2024-11-26T02:57:36.882333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}