{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":9925596,"sourceType":"datasetVersion","datasetId":6100619},{"sourceId":9930699,"sourceType":"datasetVersion","datasetId":6104284},{"sourceId":192414,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":164062,"modelId":186409}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport polars as pl\nimport matplotlib.pyplot as plt\nimport os\nimport kaggle_evaluation.jane_street_inference_server\nimport pickle\nfrom sklearn.feature_selection import f_regression,mutual_info_regression,SelectKBest\nfrom sklearn.model_selection import train_test_split","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:16:50.260961Z","iopub.execute_input":"2024-12-08T09:16:50.261365Z","iopub.status.idle":"2024-12-08T09:16:54.028458Z","shell.execute_reply.started":"2024-12-08T09:16:50.261329Z","shell.execute_reply":"2024-12-08T09:16:54.027140Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dir = \"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet\"\nparquet_paths = [f\"{data_dir}/{files}\" for files in os.listdir(data_dir)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:50.230597Z","iopub.execute_input":"2024-12-07T09:17:50.231686Z","iopub.status.idle":"2024-12-07T09:17:50.240131Z","shell.execute_reply.started":"2024-12-07T09:17:50.231620Z","shell.execute_reply":"2024-12-07T09:17:50.238996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_parquet_data(paths):\n    \"\"\"Returns the concatenated dataset\"\"\"\n    \n    frames = []\n    for path in paths:\n        df = pl.read_parquet(path)\n        #columns to cast\n        columns_cast = [cols for cols,dtype in df.schema.items() if dtype == pl.Float32]\n        \n        # casting higher precision data types to lower precision data types\n        df = df.with_columns([pl.col(col).cast(pl.Int16,strict=False) for col in columns_cast])\n        frames.append(df)\n    \n    df_combined = pl.concat(frames)\n    return df_combined\n    \n        ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:50.241663Z","iopub.execute_input":"2024-12-07T09:17:50.242069Z","iopub.status.idle":"2024-12-07T09:17:50.254107Z","shell.execute_reply.started":"2024-12-07T09:17:50.242036Z","shell.execute_reply":"2024-12-07T09:17:50.252816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train_data = load_parquet_data(parquet_paths)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:50.255590Z","iopub.execute_input":"2024-12-07T09:17:50.255963Z","iopub.status.idle":"2024-12-07T09:17:50.267112Z","shell.execute_reply.started":"2024-12-07T09:17:50.255931Z","shell.execute_reply":"2024-12-07T09:17:50.265985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#lags_data = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet/date_id=0\")\n#lags_data.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:50.270884Z","iopub.execute_input":"2024-12-07T09:17:50.271292Z","iopub.status.idle":"2024-12-07T09:17:50.281902Z","shell.execute_reply.started":"2024-12-07T09:17:50.271259Z","shell.execute_reply":"2024-12-07T09:17:50.280628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# understanding the response variable distribution\ndef hist_plots(response_set,df,nrows,ncols):\n    \"\"\"Returns the histograms for the response variable\"\"\"\n    fig,ax = plt.subplots(nrows=nrows,ncols=ncols,figsize=(10,10))\n    ax = ax.flatten()\n    for i,col in enumerate(response_set):\n        ax[i].hist(df[col],bins=10,edgecolor='black')\n        ax[i].set_xlabel('Bins')\n        ax[i].set_ylabel('Count')\n        ax[i].grid()\n        ax[i].set_title(f\"Distribution of {col}\")\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:50.283362Z","iopub.execute_input":"2024-12-07T09:17:50.283870Z","iopub.status.idle":"2024-12-07T09:17:50.298696Z","shell.execute_reply.started":"2024-12-07T09:17:50.283821Z","shell.execute_reply":"2024-12-07T09:17:50.297546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#response_set = [col for col in train_data.columns if col.startswith('responder')]\n#hist_plots(response_set,train_data,3,3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:50.300238Z","iopub.execute_input":"2024-12-07T09:17:50.301177Z","iopub.status.idle":"2024-12-07T09:17:50.316171Z","shell.execute_reply.started":"2024-12-07T09:17:50.301138Z","shell.execute_reply":"2024-12-07T09:17:50.315086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#plt.figure(figsize=(18,6))\n#plt.plot(train_data.null_count().columns,train_data.null_count().row(0),marker='o')\n#plt.grid()\n#plt.title(\"Null values in Data\")\n#plt.xlabel('Features')\n#plt.xticks(rotation=90)\n#plt.ylabel('Null Counts')\n#plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:50.317630Z","iopub.execute_input":"2024-12-07T09:17:50.318065Z","iopub.status.idle":"2024-12-07T09:17:50.330394Z","shell.execute_reply.started":"2024-12-07T09:17:50.318028Z","shell.execute_reply":"2024-12-07T09:17:50.329108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# dropping features with higher null counts\n#drop_features = ['feature_00','feature_01','feature_02','feature_03','feature_04','feature_08',\n#                 'feature_15','feature_21','feature_26','feature_27','feature_31','feature_39',\n#                 'feature_41','feature_42','feature_43','feature_44','feature_52','feature_55',\n#                 'feature_45','feature_46','feature_50','feature_53','feature_62','feature_58',\n#                 'feature_62','feature_63','feature_64','feature_66','feature_73','feature_74',\n#                 'responder_0','responder_1','responder_2','responder_3','responder_4','responder_5',\n#                 'responder_7','responder_8']\n#selected_features = [col for col in train_data.columns if col not in drop_features]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:50.331741Z","iopub.execute_input":"2024-12-07T09:17:50.332119Z","iopub.status.idle":"2024-12-07T09:17:50.344313Z","shell.execute_reply.started":"2024-12-07T09:17:50.332076Z","shell.execute_reply":"2024-12-07T09:17:50.343018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#selected_data = train_data[selected_features]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:50.345740Z","iopub.execute_input":"2024-12-07T09:17:50.346136Z","iopub.status.idle":"2024-12-07T09:17:50.358159Z","shell.execute_reply.started":"2024-12-07T09:17:50.346102Z","shell.execute_reply":"2024-12-07T09:17:50.356929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#lags_data = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet/date_id=0\")\n#lags_data.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:50.359662Z","iopub.execute_input":"2024-12-07T09:17:50.360063Z","iopub.status.idle":"2024-12-07T09:17:50.370059Z","shell.execute_reply.started":"2024-12-07T09:17:50.360028Z","shell.execute_reply":"2024-12-07T09:17:50.368844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train_lag_data = selected_data.sort(by='date_id').with_columns([pl.col('responder_6').shift(39).alias('responder_6_lags')])\n#train_lag_data.tail(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:50.371608Z","iopub.execute_input":"2024-12-07T09:17:50.372094Z","iopub.status.idle":"2024-12-07T09:17:50.382052Z","shell.execute_reply.started":"2024-12-07T09:17:50.372046Z","shell.execute_reply":"2024-12-07T09:17:50.380926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#plt.scatter(selected_data['feature_05'],selected_data['responder_6'],s=2,c='r')\n#plt.grid()\n#plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:50.383685Z","iopub.execute_input":"2024-12-07T09:17:50.384097Z","iopub.status.idle":"2024-12-07T09:17:50.395270Z","shell.execute_reply.started":"2024-12-07T09:17:50.384063Z","shell.execute_reply":"2024-12-07T09:17:50.394165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#plt.hist(selected_data['feature_09'],bins=15)\n#plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:50.401060Z","iopub.execute_input":"2024-12-07T09:17:50.401450Z","iopub.status.idle":"2024-12-07T09:17:50.412291Z","shell.execute_reply.started":"2024-12-07T09:17:50.401417Z","shell.execute_reply":"2024-12-07T09:17:50.411162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train_lag_data.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:50.413919Z","iopub.execute_input":"2024-12-07T09:17:50.414476Z","iopub.status.idle":"2024-12-07T09:17:50.424185Z","shell.execute_reply.started":"2024-12-07T09:17:50.414427Z","shell.execute_reply":"2024-12-07T09:17:50.422998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# updating the values from lag tables with the lag values\n#train_lag_data = train_lag_data.update(\n#    lags_data.rename({'responder_6_lag_1':'responder_6_lags'}),\n#    on = ['date_id','symbol_id'],\n#    how = 'left'\n#)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:50.425556Z","iopub.execute_input":"2024-12-07T09:17:50.426064Z","iopub.status.idle":"2024-12-07T09:17:50.439125Z","shell.execute_reply.started":"2024-12-07T09:17:50.426013Z","shell.execute_reply":"2024-12-07T09:17:50.437829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train_lag_data.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:50.440520Z","iopub.execute_input":"2024-12-07T09:17:50.441033Z","iopub.status.idle":"2024-12-07T09:17:50.450167Z","shell.execute_reply.started":"2024-12-07T09:17:50.440983Z","shell.execute_reply":"2024-12-07T09:17:50.449200Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train_lag_data.write_parquet(\"final_data.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:50.451448Z","iopub.execute_input":"2024-12-07T09:17:50.451833Z","iopub.status.idle":"2024-12-07T09:17:50.461193Z","shell.execute_reply.started":"2024-12-07T09:17:50.451765Z","shell.execute_reply":"2024-12-07T09:17:50.460129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_data = pl.read_parquet('/kaggle/input/final-data')\nnew_data.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:16:54.030724Z","iopub.execute_input":"2024-12-08T09:16:54.031493Z","iopub.status.idle":"2024-12-08T09:17:01.368845Z","shell.execute_reply.started":"2024-12-08T09:16:54.031436Z","shell.execute_reply":"2024-12-08T09:17:01.367622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_col = [col for col in new_data.columns if col.startswith('f')]\nnew_data[features_col].var()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:18:08.873450Z","iopub.execute_input":"2024-12-08T09:18:08.873892Z","iopub.status.idle":"2024-12-08T09:18:12.814614Z","shell.execute_reply.started":"2024-12-08T09:18:08.873858Z","shell.execute_reply":"2024-12-08T09:18:12.813286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# selecting features with values close to 1 and higher\n\nselected_features = ['feature_09','feature_10','feature_11','feature_24','feature_47','feature_34','weight','responder_6']\nselected_data = new_data[selected_features].drop_nulls()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:34:27.170072Z","iopub.execute_input":"2024-12-08T09:34:27.170518Z","iopub.status.idle":"2024-12-08T09:34:27.196194Z","shell.execute_reply.started":"2024-12-08T09:34:27.170482Z","shell.execute_reply":"2024-12-08T09:34:27.194612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_data.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:27:24.066911Z","iopub.execute_input":"2024-12-08T09:27:24.067450Z","iopub.status.idle":"2024-12-08T09:27:24.079942Z","shell.execute_reply.started":"2024-12-08T09:27:24.067399Z","shell.execute_reply":"2024-12-08T09:27:24.077024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#lag_values = selected_data['responder_6'].shift(39)\n#selected_data.select(pl.lit(0,0).alias('responder_6_lags'))\n#new_data = selected_data.with_columns(pl.Series('responder_6_lags',lag_values))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:58.965300Z","iopub.execute_input":"2024-12-07T09:17:58.965925Z","iopub.status.idle":"2024-12-07T09:17:58.977614Z","shell.execute_reply.started":"2024-12-07T09:17:58.965859Z","shell.execute_reply":"2024-12-07T09:17:58.976146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#lag_series = lags_data.select('date_id','symbol_id','responder_6_lag_1')\n#lag_series.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:58.979659Z","iopub.execute_input":"2024-12-07T09:17:58.980206Z","iopub.status.idle":"2024-12-07T09:17:58.994241Z","shell.execute_reply.started":"2024-12-07T09:17:58.980156Z","shell.execute_reply":"2024-12-07T09:17:58.992868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#for i,row in enumerate(lag_series.iter_rows()):\n#    if new_data['date_id'] == 0:\n#        if new_data['symbol_id'] == lag_series['symbol_id']:\n#            new_data['responder_6_lags'][i] = lag_series['responder_6_lags_1'[i]]\n\n#new_data.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:58.995673Z","iopub.execute_input":"2024-12-07T09:17:58.996078Z","iopub.status.idle":"2024-12-07T09:17:59.013292Z","shell.execute_reply.started":"2024-12-07T09:17:58.996041Z","shell.execute_reply":"2024-12-07T09:17:59.011749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.spatial.distance import correlation,cosine\n\ndef distance_corr(dataset,feature_set,target):\n    \"\"\"Returns distance coefficient for the features and target\"\"\"\n\n    selected_set = [col for col in feature_set if dataset[col].null_count() == 0]\n    for col in selected_set:\n        distance_r = cosine(dataset[col],dataset[target],dataset['weight'])\n        print(\"Correlation between %s and %s is %.2f\"%(col,target,distance_r))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:59.014681Z","iopub.execute_input":"2024-12-07T09:17:59.015089Z","iopub.status.idle":"2024-12-07T09:17:59.027843Z","shell.execute_reply.started":"2024-12-07T09:17:59.015052Z","shell.execute_reply":"2024-12-07T09:17:59.026570Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#distance_corr(selected_data.drop_nulls(),selected_features,'responder_6')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:17:59.029168Z","iopub.execute_input":"2024-12-07T09:17:59.029528Z","iopub.status.idle":"2024-12-07T09:17:59.041549Z","shell.execute_reply.started":"2024-12-07T09:17:59.029495Z","shell.execute_reply":"2024-12-07T09:17:59.040136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#means = new_data.select([pl.col(col).mean().alias(f\"{col}_mean\") for col in new_data.columns])\n#stddev = new_data.select([pl.col(col).std().alias(f\"{col}_std\") for col in new_data.columns])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:18:02.692358Z","iopub.execute_input":"2024-12-07T09:18:02.692966Z","iopub.status.idle":"2024-12-07T09:18:02.698322Z","shell.execute_reply.started":"2024-12-07T09:18:02.692920Z","shell.execute_reply":"2024-12-07T09:18:02.697086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#std_data = new_data.with_columns([((pl.col(col) - means[f\"{col}_mean\"])/stddev[f\"{col}_std\"]).cast(pl.Int16,strict=False).alias(col) for col in new_data.columns])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:18:02.699653Z","iopub.execute_input":"2024-12-07T09:18:02.700088Z","iopub.status.idle":"2024-12-07T09:18:02.712433Z","shell.execute_reply.started":"2024-12-07T09:18:02.700054Z","shell.execute_reply":"2024-12-07T09:18:02.711094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"split_size = int(0.70 * selected_data.shape[0])\nxtrain = selected_data.drop(['responder_6','weight'])[:split_size]\nytrain = selected_data['responder_6'][:split_size]\n\nxtest = selected_data.drop(['responder_6','weight'])[split_size:]\nytest = selected_data['responder_6'][split_size:]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:34:51.146437Z","iopub.execute_input":"2024-12-08T09:34:51.146847Z","iopub.status.idle":"2024-12-08T09:34:51.154429Z","shell.execute_reply.started":"2024-12-08T09:34:51.146812Z","shell.execute_reply":"2024-12-08T09:34:51.153120Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xtrain.null_count()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:34:54.543692Z","iopub.execute_input":"2024-12-08T09:34:54.544070Z","iopub.status.idle":"2024-12-08T09:34:54.553355Z","shell.execute_reply.started":"2024-12-08T09:34:54.544038Z","shell.execute_reply":"2024-12-08T09:34:54.552296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#kbest_f2 = SelectKBest(score_func=mutual_info_regression,k=10)\n#X_train = kbest_f2.fit_transform(xtrain,ytrain)\n#X_test = kbest_f2.transform(xtest)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:18:02.753664Z","iopub.execute_input":"2024-12-07T09:18:02.754665Z","iopub.status.idle":"2024-12-07T09:18:02.762029Z","shell.execute_reply.started":"2024-12-07T09:18:02.754611Z","shell.execute_reply":"2024-12-07T09:18:02.760849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#kbest_f2.get_feature_names_out()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:18:02.763675Z","iopub.execute_input":"2024-12-07T09:18:02.764167Z","iopub.status.idle":"2024-12-07T09:18:02.774590Z","shell.execute_reply.started":"2024-12-07T09:18:02.764119Z","shell.execute_reply":"2024-12-07T09:18:02.773030Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import r2_score,mean_squared_error\nfrom lightgbm import LGBMRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:51:27.108488Z","iopub.execute_input":"2024-12-08T09:51:27.109716Z","iopub.status.idle":"2024-12-08T09:51:28.579203Z","shell.execute_reply.started":"2024-12-08T09:51:27.109671Z","shell.execute_reply":"2024-12-08T09:51:28.577970Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluation_metric(ytrue,ypred,weight_train):\n    r2 = r2_score(ytrue,ypred,sample_weight=weight_train)\n    mse = mean_squared_error(ytrue,ypred,sample_weight=weight_train)\n    print(\"R2 score train set is:\",r2)\n    print(\"MSE score on train set is:\",mse)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:51:28.581080Z","iopub.execute_input":"2024-12-08T09:51:28.581672Z","iopub.status.idle":"2024-12-08T09:51:28.587575Z","shell.execute_reply.started":"2024-12-08T09:51:28.581634Z","shell.execute_reply":"2024-12-08T09:51:28.586391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\nfrom xgboost import XGBRegressor\nfrom sklearn.feature_selection import SequentialFeatureSelector","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:51:28.589220Z","iopub.execute_input":"2024-12-08T09:51:28.589688Z","iopub.status.idle":"2024-12-08T09:51:28.798692Z","shell.execute_reply.started":"2024-12-08T09:51:28.589641Z","shell.execute_reply":"2024-12-08T09:51:28.797240Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#dtrain = xgb.DMatrix(X_train,label=ytrain,nthread=-1)\n#xgbmodel = xgb.train(params={'max_depth':7,'objective':'reg:squarederror','num_parallel_tree':4},\n#                   dtrain=dtrain,num_boost_round=4)\n\n\n#ypred = xgbmodel.predict(xgb.DMatrix(X_train))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:18:05.393453Z","iopub.execute_input":"2024-12-07T09:18:05.394000Z","iopub.status.idle":"2024-12-07T09:18:05.400066Z","shell.execute_reply.started":"2024-12-07T09:18:05.393953Z","shell.execute_reply":"2024-12-07T09:18:05.398845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#weight_train = selected_data['weight'][:split_size]\n#evaluation_metric(ytrain,ypred,weight_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T09:18:05.401699Z","iopub.execute_input":"2024-12-07T09:18:05.402121Z","iopub.status.idle":"2024-12-07T09:18:05.413824Z","shell.execute_reply.started":"2024-12-07T09:18:05.402086Z","shell.execute_reply":"2024-12-07T09:18:05.412446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#xgb = XGBRegressor(max_depth=7,num_parallel_tree=4,eta=0.3,min_child_weight=2)\n#xgb.fit(xtrain,ytrain)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T09:51:32.534120Z","iopub.execute_input":"2024-12-08T09:51:32.535252Z","iopub.status.idle":"2024-12-08T10:05:01.026052Z","shell.execute_reply.started":"2024-12-08T09:51:32.535171Z","shell.execute_reply":"2024-12-08T10:05:01.024895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\n#with open(\"xgb_model\",'wb') as f:\n#    pickle.dump(xgb,f)\n\nwith open(\"/kaggle/input/xgb/scikitlearn/default/1/xgb_model\",\"rb\") as f:\n    model = pickle.load(f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:14:08.588107Z","iopub.execute_input":"2024-12-08T10:14:08.589025Z","iopub.status.idle":"2024-12-08T10:14:08.655514Z","shell.execute_reply.started":"2024-12-08T10:14:08.588983Z","shell.execute_reply":"2024-12-08T10:14:08.654618Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#lgbm_model = LGBMRegressor(boosting_type='gbdt',max_depth=15,learning_rate=0.01,reg_lambda=0.025,num_leaves=64,reg_alpha=0.3)\n#lgbm_model.fit(xtrain,ytrain)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#ypred = lgbm_model.predict(xtrain)\n#evaluation_metric(ytrain,ypred,weight_train)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#xgbreg = XGBRegressor(n_estimators=250,max_depth=15,eta=0.15,gamma=0.1)\n#xgbreg.fit(X_train,ytrain)\n\n\n#ypred_train_xgb = xgbreg.predict(X_train)\n#ypred_test_xgb = xgbreg.predict(X_test)\n\n#weight_test = fill_data['weight'][split_size:]\n#print(\"Evaluation for train set is\")\n#evaluation_metric(ytrain,ypred_train_xgb,weight_train)\n#print(\"\\n\")\n#print(\"Evaluation for test set is\")\n#evaluation_metric(ytest,ypred_test_xgb,weight_test)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet/date_id=0\")\ntest_data.head(5)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#test_data.join(\n#    lags_data.rename({'responder_6_lag_1':'responder_6_lags'}),\n#    on = ['date_id','symbol_id'],\n#    how='left'\n#)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data['date_id'].unique()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict(test: pl.DataFrame,lags: pl.DataFrame) -> pl.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    # All the responders from the previous day are passed in at time_id == 0. We save them in a global variable for access at every time_id.\n    # Use them as extra features, if you like.\n    cols = xtrain.columns\n    #new_test = test.join(\n    #    lags.rename({'responder_6_lag_1':'responder_6_lags'}),\n     #   on = ['date_id','symbol_id'],\n    #  how = 'left'\n    #)\n    # Use them as extra features, if you like.\n    #global lags_\n    #if lags is not None:\n    #    lags_ = lags\n        \n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    test_pred = model.predict(test[cols])\n    predictions = predictions.with_columns(pl.Series('responder_6',test_pred.ravel()))\n    # The predict function must return a DataFrame\n    assert isinstance(predictions, pl.DataFrame)\n    # with columns 'row_id', 'responer_6'\n    assert predictions.columns == ['row_id', 'responder_6']\n    # and as many rows as the test data.\n    assert len(predictions) == len(test)\n    return predictions\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:14:25.800305Z","iopub.execute_input":"2024-12-08T10:14:25.800679Z","iopub.status.idle":"2024-12-08T10:14:25.808001Z","shell.execute_reply.started":"2024-12-08T10:14:25.800648Z","shell.execute_reply":"2024-12-08T10:14:25.806849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#predict(test_data,lags_data)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet'\n        )\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T10:14:29.363405Z","iopub.execute_input":"2024-12-08T10:14:29.363806Z","iopub.status.idle":"2024-12-08T10:14:29.612515Z","shell.execute_reply.started":"2024-12-08T10:14:29.363769Z","shell.execute_reply":"2024-12-08T10:14:29.611417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}