{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport plotly_express as px\nimport plotly.graph_objects as go\nimport os\nimport polars as pl\nfrom plotly.offline import init_notebook_mode, iplot\ninit_notebook_mode(connected=True)\nfrom lightgbm import LGBMRegressor\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import r2_score,mean_squared_error","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-25T15:35:29.161974Z","iopub.execute_input":"2024-10-25T15:35:29.162507Z","iopub.status.idle":"2024-10-25T15:35:35.531339Z","shell.execute_reply.started":"2024-10-25T15:35:29.162453Z","shell.execute_reply":"2024-10-25T15:35:35.529946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Understanding Data","metadata":{}},{"cell_type":"code","source":"# loading all the files\n\ndf0 = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=0\")\ndf1 = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=1\")\ndf2 = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=2\")\ndf3 = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=3\")\ndf4 = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=4\")\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:35:43.792531Z","iopub.execute_input":"2024-10-25T15:35:43.793282Z","iopub.status.idle":"2024-10-25T15:36:04.151075Z","shell.execute_reply.started":"2024-10-25T15:35:43.793217Z","shell.execute_reply":"2024-10-25T15:36:04.149770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Partition","metadata":{}},{"cell_type":"code","source":"# dropping features with high percentage of null values\n#from plotly.subplots import make_subplots\n#def distribution_plot(dataset,feature_set):\n#    \"\"\"Returns distribution of features set\"\"\"\n\n#    fig = []\n#    for col in feature_set:\n#        if dataset[col].null_count() == 0:\n#            plots = px.histogram(dataset,x = col)\n#            fig.append(plots)\n\n#    num_columns = 2\n#    num_rows = (len(fig) - num_columns + 1) // num_columns \n\n#    figures = make_subplots(rows = len(fig),cols=2)\n\n    # adding each histogram\n#    for i,figure in enumerate(fig):\n#        row = (i // num_columns) + 1\n#        cols = (i % num_columns) + 1\n#        for trace in figure['data']:\n#            figures.add_trace(trace,row,cols)\n\n#    figures.update_layout(height=150*num_rows,width=500,title='Distribution of Features')\n#    iplot(figures)\n\n\n#distribution_plot(df0,features_set)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:23:57.178560Z","iopub.execute_input":"2024-10-25T15:23:57.179739Z","iopub.status.idle":"2024-10-25T15:23:57.185660Z","shell.execute_reply.started":"2024-10-25T15:23:57.179668Z","shell.execute_reply":"2024-10-25T15:23:57.184320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# performing statistical test to understand whether data is drawn from normal population\nfrom scipy.stats import kstest\nfrom scipy import stats\n\ndef normality_test(dataset,features):\n    \"\"\"Returns whether the features belong to normal distribution\"\"\"\n\n    for col in features:\n        if dataset[col].null_count() == 0:\n            ks_test = kstest(rvs=dataset[col],cdf=stats.norm.cdf,alternative='two-sided')\n            print(f\"Results for {col}\\n:{ks_test}\")\n\n\n    return\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:36:37.962776Z","iopub.execute_input":"2024-10-25T15:36:37.963290Z","iopub.status.idle":"2024-10-25T15:36:37.972936Z","shell.execute_reply.started":"2024-10-25T15:36:37.963243Z","shell.execute_reply":"2024-10-25T15:36:37.971190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# majority of the features set differ from normality.\n# understanding relations of features with target variable\n\ndef relationship_plot(dataset,features_set,target,nrows,ncols):\n    \"\"\"Returns plots for the feature target relationship\"\"\"\n\n    selected_features = [col for col in features_set if dataset[col].null_count() == 0]\n    nrows = nrows\n    ncols = ncols\n\n    fig,ax = plt.subplots(nrows=nrows,ncols=ncols,figsize=(20,20))\n    ax = ax.flatten()\n    for i,col in enumerate(selected_features):\n        ax[i].scatter(dataset[col],dataset[target],alpha=0.5)\n        ax[i].set_title(f\"{col} VS {target}\")\n        ax[i].set_xlabel(f\"{col}\")\n        ax[i].set_ylabel(target)\n\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:36:38.411392Z","iopub.execute_input":"2024-10-25T15:36:38.412021Z","iopub.status.idle":"2024-10-25T15:36:38.425626Z","shell.execute_reply.started":"2024-10-25T15:36:38.411962Z","shell.execute_reply":"2024-10-25T15:36:38.423847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df1 = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=1\")\n#df2 = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=2\")\n#df3 = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=3\")\n#df4 = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=4\")\n#df5 = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=5\")\n#df6 = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=6\")\n#df7 = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=7\")\n#df8 = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=8\")\n#df9 = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=9\")","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:36:39.054854Z","iopub.execute_input":"2024-10-25T15:36:39.056866Z","iopub.status.idle":"2024-10-25T15:36:39.063913Z","shell.execute_reply.started":"2024-10-25T15:36:39.056667Z","shell.execute_reply":"2024-10-25T15:36:39.061511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# understanding correlation of features set with target data\nfrom scipy.spatial.distance import correlation\n\ndef distance_corr(dataset,feature_set,target):\n    \"\"\"Returns distance coefficient for the features and target\"\"\"\n\n    selected_set = [col for col in feature_set if dataset[col].null_count() == 0]\n    for col in selected_set:\n        distance_r = correlation(dataset[col],dataset[target])\n        print(\"Correlation between %s and %s is %.2f\"%(col,target,distance_r))\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:36:39.493758Z","iopub.execute_input":"2024-10-25T15:36:39.494325Z","iopub.status.idle":"2024-10-25T15:36:39.503422Z","shell.execute_reply.started":"2024-10-25T15:36:39.494269Z","shell.execute_reply":"2024-10-25T15:36:39.501899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_part1 = pl.concat([df0,df1,df2,df3,df4],how='vertical')\ndata_part1.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:36:39.940203Z","iopub.execute_input":"2024-10-25T15:36:39.940821Z","iopub.status.idle":"2024-10-25T15:36:39.968111Z","shell.execute_reply.started":"2024-10-25T15:36:39.940749Z","shell.execute_reply":"2024-10-25T15:36:39.966607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = data_part1.drop_nulls()","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:36:40.558683Z","iopub.execute_input":"2024-10-25T15:36:40.559170Z","iopub.status.idle":"2024-10-25T15:36:42.189814Z","shell.execute_reply.started":"2024-10-25T15:36:40.559124Z","shell.execute_reply":"2024-10-25T15:36:42.188158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.null_count()","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:36:47.585521Z","iopub.execute_input":"2024-10-25T15:36:47.586148Z","iopub.status.idle":"2024-10-25T15:36:47.607076Z","shell.execute_reply.started":"2024-10-25T15:36:47.586099Z","shell.execute_reply":"2024-10-25T15:36:47.605665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_set = [col for col in train_data.columns if col not in ('date_id','symbol_id','time_id','responder_6')]","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:36:50.375328Z","iopub.execute_input":"2024-10-25T15:36:50.376024Z","iopub.status.idle":"2024-10-25T15:36:50.383254Z","shell.execute_reply.started":"2024-10-25T15:36:50.375961Z","shell.execute_reply":"2024-10-25T15:36:50.381564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# understanding the responder by using histograms\nplt.hist(df0['responder_6'],bins=20,edgecolor='black')\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:36:51.530389Z","iopub.execute_input":"2024-10-25T15:36:51.531286Z","iopub.status.idle":"2024-10-25T15:36:51.929462Z","shell.execute_reply.started":"2024-10-25T15:36:51.531235Z","shell.execute_reply":"2024-10-25T15:36:51.928104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# relationship of responder set with target set\nresponder_distance = distance_corr(train_data,feature_set,'responder_6')\nresponder_distance","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:36:52.433215Z","iopub.execute_input":"2024-10-25T15:36:52.433661Z","iopub.status.idle":"2024-10-25T15:37:01.128166Z","shell.execute_reply.started":"2024-10-25T15:36:52.433614Z","shell.execute_reply":"2024-10-25T15:37:01.126809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# applying hyperbolic sine transformation on data\n\nfor col in feature_set:\n    data_part1.with_columns(pl.col(col).sinh())\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:37:01.130428Z","iopub.execute_input":"2024-10-25T15:37:01.130884Z","iopub.status.idle":"2024-10-25T15:37:18.744121Z","shell.execute_reply.started":"2024-10-25T15:37:01.130840Z","shell.execute_reply":"2024-10-25T15:37:18.742783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# implement PCA for maximum data capture\nfrom sklearn.decomposition import PCA\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\n\nX = train_data.select(feature_set)\nY = train_data.select('responder_6')\n\n# dataset splitting into train and val set\nxtrain,xval,ytrain,yval = train_test_split(X,Y,test_size=0.30,random_state=10)\n\n# implementing pca on\npca = PCA(n_components=0.99)\nX_train = pca.fit_transform(xtrain)\nX_test = pca.transform(xval)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:37:18.745528Z","iopub.execute_input":"2024-10-25T15:37:18.745953Z","iopub.status.idle":"2024-10-25T15:38:37.455113Z","shell.execute_reply.started":"2024-10-25T15:37:18.745913Z","shell.execute_reply":"2024-10-25T15:38:37.453341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting pca data \ncum_variance = np.cumsum(pca.explained_variance_ratio_)\nn_components = np.arange(1,3,1)\nplt.plot(n_components,cum_variance,label='Cumsum Variance')\nplt.xlabel('N Components')\nplt.ylabel(\"% Cummulative variance\")\nplt.axhline(y=0.99,color='yellow',linestyle='-')\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:39:12.181848Z","iopub.execute_input":"2024-10-25T15:39:12.182334Z","iopub.status.idle":"2024-10-25T15:39:12.487153Z","shell.execute_reply.started":"2024-10-25T15:39:12.182289Z","shell.execute_reply":"2024-10-25T15:39:12.485596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_1 = LGBMRegressor(boosting='gbdt',max_depth=11,learning_rate=0.01,n_estimators=200,random_state=12)\nlgbm_1.fit(X_train,ytrain)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:39:18.206094Z","iopub.execute_input":"2024-10-25T15:39:18.206632Z","iopub.status.idle":"2024-10-25T15:39:55.832378Z","shell.execute_reply.started":"2024-10-25T15:39:18.206546Z","shell.execute_reply":"2024-10-25T15:39:55.831059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = lgbm_1.predict(X_train)\nyval_pred = lgbm_1.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:39:55.834931Z","iopub.execute_input":"2024-10-25T15:39:55.835380Z","iopub.status.idle":"2024-10-25T15:41:05.031550Z","shell.execute_reply.started":"2024-10-25T15:39:55.835333Z","shell.execute_reply":"2024-10-25T15:41:05.030265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weights_train = np.array(xtrain.select('weight'))\nweights_test = np.array(xval.select('weight'))","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:42:28.953618Z","iopub.execute_input":"2024-10-25T15:42:28.954621Z","iopub.status.idle":"2024-10-25T15:42:28.968488Z","shell.execute_reply.started":"2024-10-25T15:42:28.954540Z","shell.execute_reply":"2024-10-25T15:42:28.966817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluation_metric(ytrue,ypred,weight_train,ytest,ypred_test,weight_test):\n    r2 = r2_score(ytrue,ypred,sample_weight=weight_train)\n    mse = mean_squared_error(ytrue,ypred,sample_weight=weight_train)\n    print(\"R2 score train set is:\",r2)\n    print(\"MSE score on train set is:\",mse)\n\n    # test set\n    r2_test = r2_score(ytest,ypred_test,sample_weight=weight_test)\n    mse_test = mean_squared_error(ytest,ypred_test,sample_weight=weight_test)\n    print(\"R2 score on test set is:\",r2_test)\n    print(\"MSE score on test set is:\",mse_test)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:44:10.173672Z","iopub.execute_input":"2024-10-25T15:44:10.174130Z","iopub.status.idle":"2024-10-25T15:44:10.183360Z","shell.execute_reply.started":"2024-10-25T15:44:10.174091Z","shell.execute_reply":"2024-10-25T15:44:10.181671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluation_metric(ytrain,y_pred,weights_train,yval,yval_pred,weights_test)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:44:11.075288Z","iopub.execute_input":"2024-10-25T15:44:11.075790Z","iopub.status.idle":"2024-10-25T15:44:11.258691Z","shell.execute_reply.started":"2024-10-25T15:44:11.075741Z","shell.execute_reply":"2024-10-25T15:44:11.257290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using ISOMAP dimensionality reduction\n#from sklearn.manifold import Isomap\n\n#scaler = StandardScaler()\n#X_train_std = scaler.fit_transform(xtrain)\n#X_test_std = scaler.transform(xval)\n\n# fitting Isomap\n#isomap = Isomap(n_neighbors=30,n_components=5)\n#X_train_map = isomap.fit_transform(X_train_std)\n#X_test_map = isomap.transform(X_test_std)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T15:44:28.040287Z","iopub.execute_input":"2024-10-24T15:44:28.040878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_2 = LGBMRegressor(boosting='gbdt',max_depth=17,learning_rate=0.01,n_estimators=200,random_state=13)\nlgbm_2.fit(X_train,ytrain)\n\n# predicting value\ny_pred_train = lgbm_2.predict(X_train)\ny_pred_test = lgbm_2.predict(X_test)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T15:48:37.717157Z","iopub.execute_input":"2024-10-25T15:48:37.717762Z","iopub.status.idle":"2024-10-25T16:01:53.814914Z","shell.execute_reply.started":"2024-10-25T15:48:37.717706Z","shell.execute_reply":"2024-10-25T16:01:53.813092Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluation_metric(ytrain,y_pred_train,weights_train,yval,y_pred_test,weights_test)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T16:03:03.325962Z","iopub.execute_input":"2024-10-25T16:03:03.326403Z","iopub.status.idle":"2024-10-25T16:03:03.503589Z","shell.execute_reply.started":"2024-10-25T16:03:03.326365Z","shell.execute_reply":"2024-10-25T16:03:03.501953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}