{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-24T11:39:03.070060Z","iopub.execute_input":"2024-10-24T11:39:03.070949Z","iopub.status.idle":"2024-10-24T11:39:03.128259Z","shell.execute_reply.started":"2024-10-24T11:39:03.070892Z","shell.execute_reply":"2024-10-24T11:39:03.126817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#importing libraries neccesary for data preprocessing and vissualization\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom mlxtend.plotting import scatterplotmatrix","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:31:46.901563Z","iopub.execute_input":"2024-10-24T11:31:46.902420Z","iopub.status.idle":"2024-10-24T11:31:51.505246Z","shell.execute_reply.started":"2024-10-24T11:31:46.902364Z","shell.execute_reply":"2024-10-24T11:31:51.503839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.svm import SVR\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor\nfrom sklearn.metrics import mean_absolute_error, r2_score\nfrom sklearn.impute import SimpleImputer\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:33:44.234490Z","iopub.execute_input":"2024-10-24T11:33:44.235171Z","iopub.status.idle":"2024-10-24T11:33:45.494444Z","shell.execute_reply.started":"2024-10-24T11:33:44.235126Z","shell.execute_reply":"2024-10-24T11:33:45.492992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#due to space connstraints we will train our model on one parquet file lets take partition_id=4\ntrain=pd.read_parquet('/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=4/part-0.parquet')","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:33:48.723668Z","iopub.execute_input":"2024-10-24T11:33:48.724275Z","iopub.status.idle":"2024-10-24T11:33:57.734619Z","shell.execute_reply.started":"2024-10-24T11:33:48.724231Z","shell.execute_reply":"2024-10-24T11:33:57.733124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:33:57.738014Z","iopub.execute_input":"2024-10-24T11:33:57.739156Z","iopub.status.idle":"2024-10-24T11:33:57.751215Z","shell.execute_reply.started":"2024-10-24T11:33:57.739077Z","shell.execute_reply":"2024-10-24T11:33:57.749734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:34:01.083889Z","iopub.execute_input":"2024-10-24T11:34:01.084418Z","iopub.status.idle":"2024-10-24T11:34:01.145503Z","shell.execute_reply.started":"2024-10-24T11:34:01.084373Z","shell.execute_reply":"2024-10-24T11:34:01.144162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns=list(train.columns)\ncolumns","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:34:02.618888Z","iopub.execute_input":"2024-10-24T11:34:02.619310Z","iopub.status.idle":"2024-10-24T11:34:02.629778Z","shell.execute_reply.started":"2024-10-24T11:34:02.619272Z","shell.execute_reply":"2024-10-24T11:34:02.628427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''def plotting_on_scattermatrix(s):\n    scatterplotmatrix(train[s].values, figsize=(16, 14), names=s, alpha=0.5)\n    plt.tight_layout()\n    plt.show()'''\n\n    ","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:34:07.814890Z","iopub.execute_input":"2024-10-24T11:34:07.816055Z","iopub.status.idle":"2024-10-24T11:34:07.825050Z","shell.execute_reply.started":"2024-10-24T11:34:07.816001Z","shell.execute_reply":"2024-10-24T11:34:07.823482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#visualizing the first nine distribution to see how are some of the colums related ,fro this \n#we can see that feature 3 an 00 are some how lenearly depoenadent\n#cols=['date_id','time_id','symbol_id','weight','feature_00','feature_01','feature_02','feature_03','responder_6']\n#plotting_on_scattermatrix(cols)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:34:08.505895Z","iopub.execute_input":"2024-10-24T11:34:08.506333Z","iopub.status.idle":"2024-10-24T11:34:08.512438Z","shell.execute_reply.started":"2024-10-24T11:34:08.506292Z","shell.execute_reply":"2024-10-24T11:34:08.510948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from the below distribution we can see that most of the features are not much more correlated to each other\n#ols1=['feature_04','feature_05','feature_06','feature_07','feature_08','feature_09','feature_10','responder_6']\n#plotting_on_scattermatrix(cols1)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:34:09.339908Z","iopub.execute_input":"2024-10-24T11:34:09.340323Z","iopub.status.idle":"2024-10-24T11:34:09.346276Z","shell.execute_reply.started":"2024-10-24T11:34:09.340285Z","shell.execute_reply":"2024-10-24T11:34:09.344734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#it clearly seems that the respondents are also clearly uncorrelated\n#cols2=['responder_1','responder_2','responder_3','responder_4','responder_5','responder_6','responder_7','responder_8']\n#plotting_on_scattermatrix(cols2)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:34:12.227110Z","iopub.execute_input":"2024-10-24T11:34:12.227648Z","iopub.status.idle":"2024-10-24T11:34:12.233568Z","shell.execute_reply.started":"2024-10-24T11:34:12.227515Z","shell.execute_reply":"2024-10-24T11:34:12.232080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:34:14.602686Z","iopub.execute_input":"2024-10-24T11:34:14.603164Z","iopub.status.idle":"2024-10-24T11:34:14.641476Z","shell.execute_reply.started":"2024-10-24T11:34:14.603116Z","shell.execute_reply":"2024-10-24T11:34:14.639861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"pd.set_option('display.max_columns',100)\ntrain.describe()","metadata":{"execution":{"iopub.status.busy":"2024-10-18T08:41:11.614987Z","iopub.execute_input":"2024-10-18T08:41:11.615530Z","iopub.status.idle":"2024-10-18T08:41:29.065028Z","shell.execute_reply.started":"2024-10-18T08:41:11.615484Z","shell.execute_reply":"2024-10-18T08:41:29.063702Z"}}},{"cell_type":"code","source":"#lets check for the presence of null values \nmissing_values = train.isnull().sum().reset_index()\nmissing_values.columns = ['Column', 'Missing Values']\n\n# Display columns with missing values only\nmissing_values[missing_values['Missing Values'] > 0]","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:34:16.602781Z","iopub.execute_input":"2024-10-24T11:34:16.603722Z","iopub.status.idle":"2024-10-24T11:34:17.250156Z","shell.execute_reply.started":"2024-10-24T11:34:16.603647Z","shell.execute_reply":"2024-10-24T11:34:17.248984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(missing_values)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:34:21.978221Z","iopub.execute_input":"2024-10-24T11:34:21.978692Z","iopub.status.idle":"2024-10-24T11:34:21.988100Z","shell.execute_reply.started":"2024-10-24T11:34:21.978641Z","shell.execute_reply":"2024-10-24T11:34:21.986562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"column_list=[]\nfor values in missing_values['Column']:\n    column_list.append(values)\n    \nprint(column_list)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:34:24.786924Z","iopub.execute_input":"2024-10-24T11:34:24.787382Z","iopub.status.idle":"2024-10-24T11:34:24.795813Z","shell.execute_reply.started":"2024-10-24T11:34:24.787339Z","shell.execute_reply":"2024-10-24T11:34:24.794465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#the best possible way to deal with this null valued columns is to impoute them using mean due to our data distribution which \n#seem to be much more centered\n\nimr = SimpleImputer(missing_values=np.nan, strategy='mean')\nimr = imr.fit(train[column_list])\ntrain[column_list] = imr.transform(train[column_list])","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:34:28.362631Z","iopub.execute_input":"2024-10-24T11:34:28.363119Z","iopub.status.idle":"2024-10-24T11:34:38.787334Z","shell.execute_reply.started":"2024-10-24T11:34:28.363073Z","shell.execute_reply":"2024-10-24T11:34:38.786070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_values = train.isnull().sum().reset_index()\nmissing_values.columns = ['Column', 'Missing Values']\n\n# Display columns with missing values only\nmissing_values[missing_values['Missing Values'] > 0]","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:34:43.748468Z","iopub.execute_input":"2024-10-24T11:34:43.748909Z","iopub.status.idle":"2024-10-24T11:34:44.492824Z","shell.execute_reply.started":"2024-10-24T11:34:43.748868Z","shell.execute_reply":"2024-10-24T11:34:44.491383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:34:47.123064Z","iopub.execute_input":"2024-10-24T11:34:47.123520Z","iopub.status.idle":"2024-10-24T11:34:47.163286Z","shell.execute_reply.started":"2024-10-24T11:34:47.123473Z","shell.execute_reply":"2024-10-24T11:34:47.161625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train.drop(['responder_6'],axis=1)\ny = train['responder_6']","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:34:50.146385Z","iopub.execute_input":"2024-10-24T11:34:50.146857Z","iopub.status.idle":"2024-10-24T11:34:51.825678Z","shell.execute_reply.started":"2024-10-24T11:34:50.146806Z","shell.execute_reply":"2024-10-24T11:34:51.824135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X,y,test_size=.2,random_state=0)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:34:55.426226Z","iopub.execute_input":"2024-10-24T11:34:55.426861Z","iopub.status.idle":"2024-10-24T11:35:06.058114Z","shell.execute_reply.started":"2024-10-24T11:34:55.426814Z","shell.execute_reply":"2024-10-24T11:35:06.056816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"linear_reg = LinearRegression()\n#svr_reg = SVR(kernel='sigmoid') # You can choose different kernels if needed\nxgb_reg = XGBRegressor(random_state=42)\n#rf_reg = RandomForestRegressor(random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:35:19.987817Z","iopub.execute_input":"2024-10-24T11:35:19.988271Z","iopub.status.idle":"2024-10-24T11:35:19.994842Z","shell.execute_reply.started":"2024-10-24T11:35:19.988231Z","shell.execute_reply":"2024-10-24T11:35:19.993449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"slr = LinearRegression()\nslr.fit(X_train, y_train)\ny_train_pred = slr.predict(X_train)\ny_test_pred = slr.predict(X_test)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:35:32.093175Z","iopub.execute_input":"2024-10-24T11:35:32.094547Z","iopub.status.idle":"2024-10-24T11:36:12.298325Z","shell.execute_reply.started":"2024-10-24T11:35:32.094463Z","shell.execute_reply":"2024-10-24T11:36:12.295120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" from sklearn.metrics import r2_score\nprint('R^2 train: %.3f, test: %.3f' % (r2_score(y_train, y_train_pred), r2_score(y_test, y_test_pred)))","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:36:37.939637Z","iopub.execute_input":"2024-10-24T11:36:37.940200Z","iopub.status.idle":"2024-10-24T11:36:37.992804Z","shell.execute_reply.started":"2024-10-24T11:36:37.940146Z","shell.execute_reply":"2024-10-24T11:36:37.991351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\nprint('MSE train: %.3f, test: %.3f' % ( mean_squared_error(y_train, y_train_pred), mean_squared_error(y_test, y_test_pred)))","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:36:41.195473Z","iopub.execute_input":"2024-10-24T11:36:41.196062Z","iopub.status.idle":"2024-10-24T11:36:41.223086Z","shell.execute_reply.started":"2024-10-24T11:36:41.196007Z","shell.execute_reply":"2024-10-24T11:36:41.221157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_reg.fit(X_train, y_train)\ny_train_pred = xgb_reg.predict(X_train)\ny_test_pred = xgb_reg.predict(X_test)\nprint('R^2 train: %.3f, test: %.3f' % (r2_score(y_train, y_train_pred), r2_score(y_test, y_test_pred)))\nprint('MSE train: %.3f, test: %.3f' % ( mean_squared_error(y_train, y_train_pred), mean_squared_error(y_test, y_test_pred)))","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:36:44.075600Z","iopub.execute_input":"2024-10-24T11:36:44.077041Z","iopub.status.idle":"2024-10-24T11:38:50.074936Z","shell.execute_reply.started":"2024-10-24T11:36:44.076981Z","shell.execute_reply":"2024-10-24T11:38:50.073133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef predict(test: pd.DataFrame, lags: pd.DataFrame | None = None) -> pd.DataFrame:\n    \"\"\"Make a prediction using XGBoost.\"\"\"\n    global lags_\n    \n    # Store the lags if provided\n    if lags is not None:\n        lags_ = lags\n\n    # Use the model to make predictions\n    predictions = xgb_reg.predict(test)\n\n    # Create a DataFrame with predictions\n    result = pd.DataFrame({\n        'row_id': test['row_id'],\n        'responder_6': predictions  # Assuming 'responder_6' is the target column\n    })\n\n    # Ensure the DataFrame structure is correct\n    assert isinstance(result, pd.DataFrame)\n    assert result.columns.tolist() == ['row_id', 'responder_6']\n    assert len(result) == len(test)\n\n    return result","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:39:41.389980Z","iopub.execute_input":"2024-10-24T11:39:41.390504Z","iopub.status.idle":"2024-10-24T11:39:41.400637Z","shell.execute_reply.started":"2024-10-24T11:39:41.390458Z","shell.execute_reply":"2024-10-24T11:39:41.399172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from kaggle_evaluation import jane_street_inference_server\ninference_server = jane_street_inference_server.JSInferenceServer(predict)\n\n# Run the inference server to submit predictions\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()  # For actual submission during competition rerun","metadata":{"execution":{"iopub.status.busy":"2024-10-24T11:41:44.534978Z","iopub.execute_input":"2024-10-24T11:41:44.535466Z","iopub.status.idle":"2024-10-24T11:41:44.543692Z","shell.execute_reply.started":"2024-10-24T11:41:44.535423Z","shell.execute_reply":"2024-10-24T11:41:44.542044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}