{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-30T09:43:01.049618Z","iopub.execute_input":"2024-10-30T09:43:01.050113Z","iopub.status.idle":"2024-10-30T09:43:02.48169Z","shell.execute_reply.started":"2024-10-30T09:43:01.050068Z","shell.execute_reply":"2024-10-30T09:43:02.480278Z"},"trusted":true},"execution_count":1,"outputs":[{"name":"stdout","text":"/kaggle/input/jane-street-real-time-market-data-forecasting/responders.csv\n/kaggle/input/jane-street-real-time-market-data-forecasting/sample_submission.csv\n/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv\n/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=4/part-0.parquet\n/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=5/part-0.parquet\n/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=6/part-0.parquet\n/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=3/part-0.parquet\n/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=1/part-0.parquet\n/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=8/part-0.parquet\n/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=2/part-0.parquet\n/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=0/part-0.parquet\n/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=7/part-0.parquet\n/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=9/part-0.parquet\n/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet/date_id=0/part-0.parquet\n/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet/date_id=0/part-0.parquet\n/kaggle/input/jane-street-real-time-market-data-forecasting/kaggle_evaluation/jane_street_gateway.py\n/kaggle/input/jane-street-real-time-market-data-forecasting/kaggle_evaluation/jane_street_inference_server.py\n/kaggle/input/jane-street-real-time-market-data-forecasting/kaggle_evaluation/__init__.py\n/kaggle/input/jane-street-real-time-market-data-forecasting/kaggle_evaluation/core/templates.py\n/kaggle/input/jane-street-real-time-market-data-forecasting/kaggle_evaluation/core/base_gateway.py\n/kaggle/input/jane-street-real-time-market-data-forecasting/kaggle_evaluation/core/relay.py\n/kaggle/input/jane-street-real-time-market-data-forecasting/kaggle_evaluation/core/kaggle_evaluation.proto\n/kaggle/input/jane-street-real-time-market-data-forecasting/kaggle_evaluation/core/__init__.py\n/kaggle/input/jane-street-real-time-market-data-forecasting/kaggle_evaluation/core/generated/kaggle_evaluation_pb2.py\n/kaggle/input/jane-street-real-time-market-data-forecasting/kaggle_evaluation/core/generated/kaggle_evaluation_pb2_grpc.py\n/kaggle/input/jane-street-real-time-market-data-forecasting/kaggle_evaluation/core/generated/__init__.py\n","output_type":"stream"}]},{"cell_type":"code","source":"csvdfFe = pd.read_csv(\"/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv\")\ndfFe.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-21T08:48:55.322711Z","iopub.execute_input":"2024-10-21T08:48:55.323921Z","iopub.status.idle":"2024-10-21T08:48:55.351023Z","shell.execute_reply.started":"2024-10-21T08:48:55.323869Z","shell.execute_reply":"2024-10-21T08:48:55.349839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# How many responder have all tags true","metadata":{}},{"cell_type":"code","source":"count = 0\nfor i in range(len(dfFe[\"tag_0\"])):\n    if dfFe[\"tag_0\"].iloc[i] == True and dfFe[\"tag_1\"].iloc[i] == True and dfFe[\"tag_2\"].iloc[i] == True and dfFe[\"tag_3\"].iloc[i] == True and dfFe[\"tag_4\"].iloc[i] == True:\n        count += 1\nprint(count)","metadata":{"execution":{"iopub.status.busy":"2024-10-21T08:43:16.030131Z","iopub.execute_input":"2024-10-21T08:43:16.03059Z","iopub.status.idle":"2024-10-21T08:43:16.038712Z","shell.execute_reply.started":"2024-10-21T08:43:16.030549Z","shell.execute_reply":"2024-10-21T08:43:16.037446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So no reponder have all true values","metadata":{}},{"cell_type":"markdown","source":"# Value counts of Tag_0","metadata":{}},{"cell_type":"code","source":"vCOT0 = dfFe[\"tag_0\"].value_counts()\nprint(vCOT0)","metadata":{"execution":{"iopub.status.busy":"2024-10-21T08:47:50.399562Z","iopub.execute_input":"2024-10-21T08:47:50.399985Z","iopub.status.idle":"2024-10-21T08:47:50.40728Z","shell.execute_reply.started":"2024-10-21T08:47:50.399947Z","shell.execute_reply":"2024-10-21T08:47:50.406171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"dfRe = pd.read_csv(\"/kaggle/input/jane-street-real-time-market-data-forecasting/responders.csv\")\ndfRe.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-21T08:50:57.67487Z","iopub.execute_input":"2024-10-21T08:50:57.675325Z","iopub.status.idle":"2024-10-21T08:50:57.699538Z","shell.execute_reply.started":"2024-10-21T08:50:57.675281Z","shell.execute_reply":"2024-10-21T08:50:57.698208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count = 0\nfor i in range(len(dfRe[\"tag_0\"])):\n    if dfRe[\"tag_0\"].iloc[i] == True and dfRe[\"tag_1\"].iloc[i] == True and dfRe[\"tag_2\"].iloc[i] == True and dfRe[\"tag_3\"].iloc[i] == True and dfRe[\"tag_4\"].iloc[i] == True:\n        count += 1\nprint(count)","metadata":{"execution":{"iopub.status.busy":"2024-10-21T08:52:21.373895Z","iopub.execute_input":"2024-10-21T08:52:21.374332Z","iopub.status.idle":"2024-10-21T08:52:21.383242Z","shell.execute_reply.started":"2024-10-21T08:52:21.374291Z","shell.execute_reply":"2024-10-21T08:52:21.381962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vRETO = dfRe[\"tag_0\"].value_counts()\nprint(vRETO)","metadata":{"execution":{"iopub.status.busy":"2024-10-21T08:52:57.741025Z","iopub.execute_input":"2024-10-21T08:52:57.741433Z","iopub.status.idle":"2024-10-21T08:52:57.748851Z","shell.execute_reply.started":"2024-10-21T08:52:57.741394Z","shell.execute_reply":"2024-10-21T08:52:57.747682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path = \"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=4/part-0.parquet\"\ndf = pd.read_parquet(file_path)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-30T09:43:37.025803Z","iopub.execute_input":"2024-10-30T09:43:37.026327Z","iopub.status.idle":"2024-10-30T09:43:45.751459Z","shell.execute_reply.started":"2024-10-30T09:43:37.026273Z","shell.execute_reply":"2024-10-30T09:43:45.750212Z"},"trusted":true},"execution_count":2,"outputs":[{"execution_count":2,"output_type":"execute_result","data":{"text/plain":"   date_id  time_id  symbol_id    weight  feature_00  feature_01  feature_02  \\\n0      680        0          0  2.298160    0.851814    1.197591    0.219422   \n1      680        0          1  3.928745    0.534441    1.079740    0.038748   \n2      680        0          2  1.340433   -0.227643    0.764146   -0.243349   \n3      680        0          3  1.695526    0.267686    1.193612   -0.388798   \n4      680        0          5  2.700766    0.952372    0.861269   -0.375405   \n\n   feature_03  feature_04  feature_05  ...  feature_78  responder_0  \\\n0    0.411698    2.057359   -0.542597  ...   -0.195100    -0.304665   \n1    0.275343    2.135057   -0.541966  ...   -0.470271     0.089769   \n2    0.247027    2.347248   -0.478477  ...    0.152837     0.218281   \n3    0.030673    2.175273   -0.408371  ...   11.179185    -0.012298   \n4    0.259099    2.497325   -0.618828  ...   -0.770234    -0.229585   \n\n   responder_1  responder_2  responder_3  responder_4  responder_5  \\\n0     0.164485    -0.205231     0.191064    -1.413209     0.375675   \n1     0.011395     0.092348     0.473781     0.397024     0.777026   \n2     0.060373    -0.164715    -0.132612     0.543831    -0.123519   \n3     1.047678    -0.696032     0.960062     2.328890     0.718955   \n4    -0.240741    -0.887929    -0.061485     0.691569     1.016049   \n\n   responder_6  responder_7  responder_8  \n0     0.929775    -1.574939     1.101371  \n1     0.826995     0.569681     1.986971  \n2    -0.296969     0.547286    -0.049303  \n3     2.047506     3.691308     3.031337  \n4     0.103898     0.814866     2.073280  \n\n[5 rows x 92 columns]","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>date_id</th>\n      <th>time_id</th>\n      <th>symbol_id</th>\n      <th>weight</th>\n      <th>feature_00</th>\n      <th>feature_01</th>\n      <th>feature_02</th>\n      <th>feature_03</th>\n      <th>feature_04</th>\n      <th>feature_05</th>\n      <th>...</th>\n      <th>feature_78</th>\n      <th>responder_0</th>\n      <th>responder_1</th>\n      <th>responder_2</th>\n      <th>responder_3</th>\n      <th>responder_4</th>\n      <th>responder_5</th>\n      <th>responder_6</th>\n      <th>responder_7</th>\n      <th>responder_8</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>680</td>\n      <td>0</td>\n      <td>0</td>\n      <td>2.298160</td>\n      <td>0.851814</td>\n      <td>1.197591</td>\n      <td>0.219422</td>\n      <td>0.411698</td>\n      <td>2.057359</td>\n      <td>-0.542597</td>\n      <td>...</td>\n      <td>-0.195100</td>\n      <td>-0.304665</td>\n      <td>0.164485</td>\n      <td>-0.205231</td>\n      <td>0.191064</td>\n      <td>-1.413209</td>\n      <td>0.375675</td>\n      <td>0.929775</td>\n      <td>-1.574939</td>\n      <td>1.101371</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>680</td>\n      <td>0</td>\n      <td>1</td>\n      <td>3.928745</td>\n      <td>0.534441</td>\n      <td>1.079740</td>\n      <td>0.038748</td>\n      <td>0.275343</td>\n      <td>2.135057</td>\n      <td>-0.541966</td>\n      <td>...</td>\n      <td>-0.470271</td>\n      <td>0.089769</td>\n      <td>0.011395</td>\n      <td>0.092348</td>\n      <td>0.473781</td>\n      <td>0.397024</td>\n      <td>0.777026</td>\n      <td>0.826995</td>\n      <td>0.569681</td>\n      <td>1.986971</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>680</td>\n      <td>0</td>\n      <td>2</td>\n      <td>1.340433</td>\n      <td>-0.227643</td>\n      <td>0.764146</td>\n      <td>-0.243349</td>\n      <td>0.247027</td>\n      <td>2.347248</td>\n      <td>-0.478477</td>\n      <td>...</td>\n      <td>0.152837</td>\n      <td>0.218281</td>\n      <td>0.060373</td>\n      <td>-0.164715</td>\n      <td>-0.132612</td>\n      <td>0.543831</td>\n      <td>-0.123519</td>\n      <td>-0.296969</td>\n      <td>0.547286</td>\n      <td>-0.049303</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>680</td>\n      <td>0</td>\n      <td>3</td>\n      <td>1.695526</td>\n      <td>0.267686</td>\n      <td>1.193612</td>\n      <td>-0.388798</td>\n      <td>0.030673</td>\n      <td>2.175273</td>\n      <td>-0.408371</td>\n      <td>...</td>\n      <td>11.179185</td>\n      <td>-0.012298</td>\n      <td>1.047678</td>\n      <td>-0.696032</td>\n      <td>0.960062</td>\n      <td>2.328890</td>\n      <td>0.718955</td>\n      <td>2.047506</td>\n      <td>3.691308</td>\n      <td>3.031337</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>680</td>\n      <td>0</td>\n      <td>5</td>\n      <td>2.700766</td>\n      <td>0.952372</td>\n      <td>0.861269</td>\n      <td>-0.375405</td>\n      <td>0.259099</td>\n      <td>2.497325</td>\n      <td>-0.618828</td>\n      <td>...</td>\n      <td>-0.770234</td>\n      <td>-0.229585</td>\n      <td>-0.240741</td>\n      <td>-0.887929</td>\n      <td>-0.061485</td>\n      <td>0.691569</td>\n      <td>1.016049</td>\n      <td>0.103898</td>\n      <td>0.814866</td>\n      <td>2.073280</td>\n    </tr>\n  </tbody>\n</table>\n<p>5 rows × 92 columns</p>\n</div>"},"metadata":{}}]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-30T09:44:08.406848Z","iopub.execute_input":"2024-10-30T09:44:08.407337Z","iopub.status.idle":"2024-10-30T09:44:08.415218Z","shell.execute_reply.started":"2024-10-30T09:44:08.407292Z","shell.execute_reply":"2024-10-30T09:44:08.413897Z"},"trusted":true},"execution_count":3,"outputs":[{"execution_count":3,"output_type":"execute_result","data":{"text/plain":"(5022952, 92)"},"metadata":{}}]},{"cell_type":"markdown","source":"# Value count of date id","metadata":{}},{"cell_type":"code","source":"vCODI = df[\"date_id\"].value_counts()\nprint(vCODI)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T09:46:57.735234Z","iopub.execute_input":"2024-10-30T09:46:57.735691Z","iopub.status.idle":"2024-10-30T09:46:57.79638Z","shell.execute_reply.started":"2024-10-30T09:46:57.735625Z","shell.execute_reply":"2024-10-30T09:46:57.795264Z"},"trusted":true},"execution_count":4,"outputs":[{"name":"stdout","text":"date_id\n765    30976\n778    30976\n772    30976\n773    30976\n774    30976\n       ...  \n730    27104\n795    25168\n790    22264\n828    20328\n789    18392\nName: count, Length: 170, dtype: int64\n","output_type":"stream"}]},{"cell_type":"markdown","source":"Most occuring dateId is 765","metadata":{}},{"cell_type":"code","source":"correlations = df.corr()\nresponder_correlations = correlations.loc['responder_0':'responder_1', 'feature_00':'feature_01']\nprint(responder_correlations)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T09:56:30.689065Z","iopub.execute_input":"2024-10-30T09:56:30.690308Z","iopub.status.idle":"2024-10-30T09:58:27.384414Z","shell.execute_reply.started":"2024-10-30T09:56:30.69026Z","shell.execute_reply":"2024-10-30T09:58:27.38315Z"},"trusted":true},"execution_count":8,"outputs":[{"name":"stdout","text":"             feature_00  feature_01\nresponder_0   -0.004318   -0.009329\nresponder_1   -0.016807   -0.027706\n","output_type":"stream"}]},{"cell_type":"markdown","source":"# Weight id, symbol id, time id, feautre00 and repondent 1 of first most occuring date id","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.impute import SimpleImputer\n\n# Assuming df is your loaded DataFrame with features and responders\n# Example: df = pd.read_parquet('path_to_file')\n\n# Define feature and responder columns\nfeature_columns = [f'feature_{i:02}' for i in range(79)]\nresponder_columns = [f'responder_{i}' for i in range(9)]  # Adjust this if there are more or fewer responders\n\n# Prepare the features (X) and targets (y)\nX = df[feature_columns]\ny = df[responder_columns]\n\n# Split data into training and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Handle missing values by imputing the mean for each column\nimputer = SimpleImputer(strategy='mean')\nX_train_imputed = imputer.fit_transform(X_train)\nX_test_imputed = imputer.transform(X_test)\n\n# Initialize and train the Random Forest Regressor\nmodel = RandomForestRegressor(n_estimators=100, random_state=42)\nmodel.fit(X_train_imputed, y_train)\n\n# Make predictions\ny_pred = model.predict(X_test_imputed)\n\n# Evaluate the model on each responder\nrmse_scores = {}\nfor i, responder in enumerate(responder_columns):\n    rmse = np.sqrt(mean_squared_error(y_test[responder], y_pred[:, i]))\n    rmse_scores[responder] = rmse\n    print(f\"RMSE for {responder}: {rmse}\")\n\n# Output RMSE scores for each responder\nprint(\"RMSE Scores:\", rmse_scores)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-30T10:02:22.653835Z","iopub.execute_input":"2024-10-30T10:02:22.654507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}