{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":214257523,"sourceType":"kernelVersion"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Created by <a href=\"https://github.com/yunsuxiaozi/\">yunsuxiaozi </a>  2024/11/12\n\n### Here we will use lightgbm、xgboost、catboost and origin features to create a simple baseline.\n\n- version1:LB:0.0043\n\n- version2:LB:0.0045\n\n- version3:LB:0.0045\n\n- version4:LB:0.0045\n\n- version5:LB:0.0045\n\n- version6:failed\n\n- version7:LB:0.0051\n\n- version8:LB:0.0050\n\n- version9:Failed","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"# <span><h1 style = \"font-family: garamond; font-size: 40px; font-style: normal; letter-spcaing: 3px; background-color: #f6f5f5; color :#fe346e; border-radius: 100px 100px; text-align:center\">Import Libraries</h1></span>","metadata":{}},{"cell_type":"code","source":"source_file_path = '/kaggle/input/yunbase/Yunbase/baseline.py'\ntarget_file_path = '/kaggle/working/baseline.py'\nwith open(source_file_path, 'r', encoding='utf-8') as file:\n    content = file.read()\nwith open(target_file_path, 'w', encoding='utf-8') as file:\n    file.write(content)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:42:05.576245Z","iopub.execute_input":"2024-12-27T18:42:05.577053Z","iopub.status.idle":"2024-12-27T18:42:05.583486Z","shell.execute_reply.started":"2024-12-27T18:42:05.577002Z","shell.execute_reply":"2024-12-27T18:42:05.582671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q --requirement /kaggle/input/yunbase/Yunbase/requirements.txt  \\\n--no-index --find-links file:/kaggle/input/yunbase/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:42:09.531346Z","iopub.execute_input":"2024-12-27T18:42:09.532163Z","iopub.status.idle":"2024-12-27T18:42:25.503969Z","shell.execute_reply.started":"2024-12-27T18:42:09.532128Z","shell.execute_reply":"2024-12-27T18:42:25.503115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from baseline import Yunbase\nimport polars as pl#similar to pandas, but with better performance when dealing with large datasets.\nimport pandas as pd#read csv,parquet\nimport numpy as np#for scientific computation of matrices\n#model\nfrom catboost import CatBoostRegressor\nimport os#Libraries that interact with the operating system\nimport gc#rubbish collection\n#environment provided by competition hoster\nimport kaggle_evaluation.jane_street_inference_server\n\nimport random#provide some function to generate random_seed.\n#set random seed,to make sure model can be recurrented.\ndef seed_everything(seed):\n    np.random.seed(seed)#numpy's random seed\n    random.seed(seed)#python built-in random seed\nseed_everything(seed=2025)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:42:29.335817Z","iopub.execute_input":"2024-12-27T18:42:29.336604Z","iopub.status.idle":"2024-12-27T18:42:46.860987Z","shell.execute_reply.started":"2024-12-27T18:42:29.336571Z","shell.execute_reply":"2024-12-27T18:42:46.860247Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span><h1 style = \"font-family: garamond; font-size: 40px; font-style: normal; letter-spcaing: 3px; background-color: #f6f5f5; color :#fe346e; border-radius: 100px 100px; text-align:center\">Load Train Data</h1></span>\n\nI accidentally made a mistake by not filling in missing values in the training data and filling in -1 in the test data, resulting in a good LB (Those who know the reason can leave a message in the discussion forum)","metadata":{}},{"cell_type":"code","source":"yunbase=Yunbase()\ndata=[]\nvalid_data=[]\n\nfor i in [6,7,8,9]:\n    #Read the parquet\n    train_df=pl.read_parquet(f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}/part-0.parquet\")\n    \n    #Split the training into training and validation\n    train = train_df.sample(fraction=0.8, seed=42)  # Set a seed for reproducibility\n    #valid = train_df.filter(~pl.col(\"*\").is_in(train.select(pl.all()).to_series()))\n    #train_indices = train.with_row_count(\"idx\").select(\"idx\").to_series()\n    train_indices = train.with_row_index().select(\"index\").to_series()\n    # Use row indices to filter validation set\n    valid = train_df.with_row_index().filter(~pl.col(\"index\").is_in(train_indices)).drop(\"index\")\n\n\n    #Add features and fill NAs with medians for both training and validation data\n    #TRAINING DATA\n    train=train.to_pandas()\n    train['sin_time_id']=np.sin(2*np.pi*train['time_id']/967)\n    train['cos_time_id']=np.cos(2*np.pi*train['time_id']/967)\n    train['sin_time_id_halfday']=np.sin(2*np.pi*train['time_id']/483)\n    train['cos_time_id_halfday']=np.cos(2*np.pi*train['time_id']/483)\n    \n\n    # Compute medians for each column and fill NaNs\n    train_medians = train.median(axis=0, skipna=True)\n    train.fillna(train_medians, inplace=True)\n    \n    # Convert infinite values to 0, if any\n    train.replace([np.inf, -np.inf], 0, inplace=True)\n    train.fillna(-1, inplace=True)  # Ensure no remaining NaNs\n    \n    train=yunbase.reduce_mem_usage(train,float16_as32=False)\n    data.append(train)\n\n    #VALIDATION DATA\n    valid=valid.to_pandas()\n    valid['sin_time_id']=np.sin(2*np.pi*valid['time_id']/967)\n    valid['cos_time_id']=np.cos(2*np.pi*valid['time_id']/967)\n    valid['sin_time_id_halfday']=np.sin(2*np.pi*valid['time_id']/483)\n    valid['cos_time_id_halfday']=np.cos(2*np.pi*valid['time_id']/483)\n    \n    # Compute medians for each column and fill NaNs\n    valid_medians = valid.median(axis=0, skipna=True)\n    valid.fillna(valid_medians, inplace=True)\n    \n    # Convert infinite values to 0, if any\n    valid.replace([np.inf, -np.inf], 0, inplace=True)\n    valid.fillna(-1, inplace=True)  # Ensure no remaining NaNs\n    \n    valid=yunbase.reduce_mem_usage(valid,float16_as32=False)\n    valid_data.append(valid)\n    \ntrain=pd.concat(data)\nvalid=pd.concat(valid_data)\n\nprint(f\"train.shape:{train.shape}\")\nprint(f\"valid.shape:{valid.shape}\")\ndel data\ndel valid_data\ngc.collect()\nfinal_feature=['symbol_id','sin_time_id','cos_time_id','sin_time_id_halfday','cos_time_id_halfday']+[f'feature_0{i}' if i<10 else f'feature_{i}' for i in range(79)]\ntrain=train[['responder_6']+final_feature]\nvalid=valid[['responder_6']+['weight']+final_feature]\ntrain.head()\nvalid.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:43:11.553693Z","iopub.execute_input":"2024-12-27T18:43:11.555002Z","iopub.status.idle":"2024-12-27T18:44:45.676560Z","shell.execute_reply.started":"2024-12-27T18:43:11.554962Z","shell.execute_reply":"2024-12-27T18:44:45.675629Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span><h1 style = \"font-family: garamond; font-size: 40px; font-style: normal; letter-spcaing: 3px; background-color: #f6f5f5; color :#fe346e; border-radius: 100px 100px; text-align:center\">Model training</h1></span>","metadata":{}},{"cell_type":"code","source":"# lgb_params={\"boosting_type\": \"gbdt\",\"metric\": 'rmse',\n#             'random_state': 2025,  \"max_depth\": 10,\"learning_rate\": 0.1,\n#             \"n_estimators\": 120,\"colsample_bytree\": 0.6,\"colsample_bynode\": 0.6,\"verbose\": -1,\"reg_alpha\": 0.2,\n#             \"reg_lambda\": 5,\"extra_trees\":True,'num_leaves':64,\"max_bin\":255,\n#             'device':'gpu','gpu_use_dp':True,\n#             }\n\nfirst_cat_params={'task_type':'GPU',\n           'random_state':2025,\n           'eval_metric'         : 'RMSE',\n           'bagging_temperature' : 0.50,\n           'iterations'          : 200,\n           'learning_rate'       : 0.1,\n           'max_depth'           : 12,\n           'l2_leaf_reg'         : 1.25,\n           'min_data_in_leaf'    : 24,\n           'random_strength'     : 0.25, \n           'verbose'             : 0,\n          }\nsecond_cat_params={'task_type':'GPU',\n           'random_state':2026,\n           'eval_metric'         : 'RMSE',\n           'bagging_temperature' : 0.50,\n           'iterations'          : 200,\n           'learning_rate'       : 0.1,\n           'max_depth'           : 12,\n           'l2_leaf_reg'         : 1.25,\n           'min_data_in_leaf'    : 24,\n           'random_strength'     : 0.25, \n           'verbose'             : 0,\n          }\nthird_cat_params={'task_type':'GPU',\n           'random_state':2027,\n           'eval_metric'         : 'RMSE',\n           'bagging_temperature' : 0.50,\n           'iterations'          : 200,\n           'learning_rate'       : 0.1,\n           'max_depth'           : 12,\n           'l2_leaf_reg'         : 1.25,\n           'min_data_in_leaf'    : 24,\n           'random_strength'     : 0.25, \n           'verbose'             : 0,\n          }\n# xgb_params={'random_state': 2025, 'n_estimators': 125, \n#             'learning_rate': 0.1, 'max_depth': 10,\n#             'reg_alpha': 0.08, 'reg_lambda': 0.8, \n#             'subsample': 0.95, 'colsample_bytree': 0.6, \n#             'min_child_weight': 3,\n#             'tree_method':'gpu_hist',\n#            }\n# print(\"lgb\")\n# lgb=LGBMRegressor(**lgb_params)\n# lgb.fit(train[final_feature].values,train['responder_6'].values)\nprint(\"first cat\")\nfirst_cat=CatBoostRegressor(**first_cat_params)\nfirst_cat.fit(train[final_feature].values,train['responder_6'].values)\n\nprint(\"second cat\")\nsecond_cat=CatBoostRegressor(**second_cat_params)\nsecond_cat.fit(train[final_feature].values,train['responder_6'].values)\n\nprint(\"third cat\")\nthird_cat=CatBoostRegressor(**third_cat_params)\nthird_cat.fit(train[final_feature].values,train['responder_6'].values)\n# print(\"xgb\")\n# xgb=XGBRegressor(**xgb_params)\n# xgb.fit(train[final_feature].values,train['responder_6'].values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:49:17.155915Z","iopub.execute_input":"2024-12-27T18:49:17.156611Z","iopub.status.idle":"2024-12-27T18:52:09.925708Z","shell.execute_reply.started":"2024-12-27T18:49:17.156572Z","shell.execute_reply":"2024-12-27T18:52:09.924270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Handle NaN values appropriately\nvalid_cleaned = valid.dropna(subset=['responder_6', 'weight'])  # Drop rows with NaNs in target or weight\ny_true_values = valid_cleaned['responder_6'].values\nweight_values = valid_cleaned['weight'].values\nweight_values = np.clip(weight_values, 1e-6, None)  # Avoid extremely small weights\n\n\n# Predictions from models\nfirst_cat_pred = first_cat.predict(valid_cleaned[final_feature].values)\nsecond_cat_pred = second_cat.predict(valid_cleaned[final_feature].values)\nthird_cat_pred = third_cat.predict(valid_cleaned[final_feature].values)\n\n# Custom R2 metric for validation\ndef r2_val(y_true, y_pred, sample_weight):\n    numerator = np.average((y_pred - y_true) ** 2, weights=sample_weight)\n    denominator = np.average((y_pred * 0 - y_true) ** 2, weights=sample_weight) + 1e-38\n    r2 = 1 - numerator / denominator\n    print(f\"Numerator: {numerator}, Denominator: {denominator}, R2: {r2}\")\n    return r2\n\n\n# Initialize search\neps = 1e-10\nmax_r2 = -float('inf')\nbest_alpha, best_beta, best_gamma = 0, 0, 0\n\n# y_true_values = valid['responder_6'].fillna(-1).values\n# weight_values = valid['weight'].fillna(-1).values\n\n# Grid search for alpha, beta, gamma\nfor alpha in np.arange(0, 1.01, 0.01):\n    for beta in np.arange(0, 1.01 - alpha, 0.01):\n        gamma = 1 - alpha - beta\n        if gamma < 0:  # Skip invalid combinations\n            continue\n        \n        # Blend predictions\n        test_preds = alpha * first_cat_pred + beta * second_cat_pred + gamma * third_cat_pred\n        test_preds = np.clip(test_preds, -5 + eps, 5 - eps)\n\n        #All 3 Always Return False:\n        # print(\"NaNs in y_true_values:\", np.isnan(y_true_values).any())\n        # print(\"NaNs in weight_values:\", np.isnan(weight_values).any())\n        # print(\"NaNs in test_preds:\", np.isnan(test_preds).any())\n\n        # if np.isnan(test_preds).any():\n        #     print(f\"NaN found in test_preds for alpha={alpha}, beta={beta}, gamma={gamma}\")\n        #     continue\n            \n        # Calculate weighted R2\n        try:\n            weighted_r2 = r2_val(y_true_values, test_preds, weight_values)\n            if weighted_r2 > max_r2:\n                max_r2 = weighted_r2\n                best_alpha, best_beta, best_gamma = alpha, beta, gamma\n        except Exception as e:\n            print(f\"Error in R2 calculation: {e}, for alpha={alpha}, beta={beta}, gamma={gamma}\")\n\n# Output the best alpha, beta, gamma\nprint(\"Best Alpha:\", best_alpha)\nprint(\"Best Beta:\", best_beta)\nprint(\"Best Gamma:\", best_gamma)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span><h1 style = \"font-family: garamond; font-size: 40px; font-style: normal; letter-spcaing: 3px; background-color: #f6f5f5; color :#fe346e; border-radius: 100px 100px; text-align:center\">Model Inference</h1></span>\n\n#### The following code is used in the training set to load more data for training, and not in the testing set to save time. I tried calling this function 100000 times and it would take about an hour.\n\n```python\ntest=yunbase.reduce_mem_usage(test,float16_as32=False)\n```","metadata":{}},{"cell_type":"code","source":"def predict(test,lags):\n    global first_cat,second_cat,third_cat,best_alpha,best_beta,best_gamma\n    \n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    test=test.to_pandas()\n    test['sin_time_id']=np.sin(2*np.pi*test['time_id']/967)\n    test['cos_time_id']=np.cos(2*np.pi*test['time_id']/967)\n    test['sin_time_id_halfday']=np.sin(2*np.pi*test['time_id']/483)\n    test['cos_time_id_halfday']=np.cos(2*np.pi*test['time_id']/483)\n    \n    # Compute medians for each column and fill NaNs\n    test_medians = test.median(axis=0, skipna=True)\n    test.fillna(test_medians, inplace=True)\n    \n    # Convert infinite values to 0, if any\n    test.replace([np.inf, -np.inf], 0, inplace=True)\n    test.fillna(-1, inplace=True)  # Ensure no remaining NaNs\n    \n    #test=test.fillna(-1)\n    #Use the medians instead\n    #test_np = test.to_numpy()\n    #test_medians = np.nanmedian(test_np, axis=0)\n    #test_np = np.where(np.isnan(test_np), test_medians, test_np)\n    #test_np = np.nan_to_num(test_np, nan=0.0, posinf=0.0, neginf=0.0)\n    #test = pd.DataFrame(test_np, columns=test.columns)\n    \n    \n    test=test[final_feature]\n    eps=1e-10\n    #test_preds=0.55*lgb.predict(test)+0.2*cat.predict(test)+0.25*xgb.predict(test)\n    #test_preds=alpha*lgb.predict(test)+beta*cat.predict(test)+gamma*xgb.predict(test)\n    test_preds = best_alpha * first_cat.predict(test) + best_beta * second_cat.predict(test) + best_gamma * third_cat.predict(test)\n    test_preds = np.clip(test_preds,-5+eps,5-eps)\n    predictions = predictions.with_columns(pl.Series('responder_6', test_preds.ravel()))\n    return predictions\n\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}