{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":11305158,"sourceType":"competition"}],"dockerImageVersionId":31234,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:33.995124Z","iopub.execute_input":"2026-01-03T13:07:33.995442Z","iopub.status.idle":"2026-01-03T13:07:34.035529Z","shell.execute_reply.started":"2026-01-03T13:07:33.995413Z","shell.execute_reply":"2026-01-03T13:07:34.034869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"initial_train = pd.read_parquet('/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=0/part-0.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:34.036858Z","iopub.execute_input":"2026-01-03T13:07:34.037164Z","iopub.status.idle":"2026-01-03T13:07:36.489354Z","shell.execute_reply.started":"2026-01-03T13:07:34.037140Z","shell.execute_reply":"2026-01-03T13:07:36.488542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"initial_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:36.490667Z","iopub.execute_input":"2026-01-03T13:07:36.490956Z","iopub.status.idle":"2026-01-03T13:07:36.727294Z","shell.execute_reply.started":"2026-01-03T13:07:36.490931Z","shell.execute_reply":"2026-01-03T13:07:36.726512Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"initial_train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:36.728379Z","iopub.execute_input":"2026-01-03T13:07:36.728824Z","iopub.status.idle":"2026-01-03T13:07:41.651965Z","shell.execute_reply.started":"2026-01-03T13:07:36.728768Z","shell.execute_reply":"2026-01-03T13:07:41.651235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"initial_train_0 = initial_train[initial_train['date_id'] == 0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:41.653866Z","iopub.execute_input":"2026-01-03T13:07:41.654184Z","iopub.status.idle":"2026-01-03T13:07:41.660503Z","shell.execute_reply.started":"2026-01-03T13:07:41.654160Z","shell.execute_reply":"2026-01-03T13:07:41.659846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"initial_train_0.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:41.661332Z","iopub.execute_input":"2026-01-03T13:07:41.661583Z","iopub.status.idle":"2026-01-03T13:07:41.818773Z","shell.execute_reply.started":"2026-01-03T13:07:41.661560Z","shell.execute_reply":"2026-01-03T13:07:41.817968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Find and rank columns/features with missing values\nfeature_cols = []\nfor col in initial_train_0.columns:\n    if col.startswith('fea'):\n        feature_cols.append(col)\ndf = initial_train_0[feature_cols].isna().sum()\ndf.sort_values(ascending = False, inplace = True)\ndf = df[df>0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:41.819852Z","iopub.execute_input":"2026-01-03T13:07:41.820561Z","iopub.status.idle":"2026-01-03T13:07:41.828602Z","shell.execute_reply.started":"2026-01-03T13:07:41.820529Z","shell.execute_reply":"2026-01-03T13:07:41.827974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (15,10))\nplt.barh(data = df, y = df.index, width = df.values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:41.829563Z","iopub.execute_input":"2026-01-03T13:07:41.829885Z","iopub.status.idle":"2026-01-03T13:07:42.184383Z","shell.execute_reply.started":"2026-01-03T13:07:41.829858Z","shell.execute_reply":"2026-01-03T13:07:42.183690Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_train = initial_train_0.drop(columns = df[df == df.max()].index.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:42.185352Z","iopub.execute_input":"2026-01-03T13:07:42.185615Z","iopub.status.idle":"2026-01-03T13:07:42.191705Z","shell.execute_reply.started":"2026-01-03T13:07:42.185591Z","shell.execute_reply":"2026-01-03T13:07:42.190856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_to_plot = ['feature_45' , 'feature_62', 'feature_39', 'feature_33']\nfig, ax = plt.subplots(1,4, figsize = (16,5))\n\nfor i, f in enumerate(features_to_plot):\n    sns.histplot(initial_train_0[f], ax = ax[i], kde = True)\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:42.192663Z","iopub.execute_input":"2026-01-03T13:07:42.192966Z","iopub.status.idle":"2026-01-03T13:07:44.719560Z","shell.execute_reply.started":"2026-01-03T13:07:42.192941Z","shell.execute_reply":"2026-01-03T13:07:44.718718Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Use ffill to fill in missing values to respect time signature","metadata":{}},{"cell_type":"code","source":"new_train.ffill()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:44.720579Z","iopub.execute_input":"2026-01-03T13:07:44.720951Z","iopub.status.idle":"2026-01-03T13:07:44.750738Z","shell.execute_reply.started":"2026-01-03T13:07:44.720924Z","shell.execute_reply":"2026-01-03T13:07:44.749810Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_to_plot = ['feature_45' , 'feature_62', 'feature_39', 'feature_33']\nfig, ax = plt.subplots(1,4, figsize = (16,5))\n\nfor i, f in enumerate(features_to_plot):\n    sns.histplot(new_train[f], ax = ax[i], kde = True)\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:44.751944Z","iopub.execute_input":"2026-01-03T13:07:44.752291Z","iopub.status.idle":"2026-01-03T13:07:45.623100Z","shell.execute_reply.started":"2026-01-03T13:07:44.752245Z","shell.execute_reply":"2026-01-03T13:07:45.622326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:45.624268Z","iopub.execute_input":"2026-01-03T13:07:45.624931Z","iopub.status.idle":"2026-01-03T13:07:45.631172Z","shell.execute_reply.started":"2026-01-03T13:07:45.624904Z","shell.execute_reply":"2026-01-03T13:07:45.630519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:45.634644Z","iopub.execute_input":"2026-01-03T13:07:45.634918Z","iopub.status.idle":"2026-01-03T13:07:45.657525Z","shell.execute_reply.started":"2026-01-03T13:07:45.634894Z","shell.execute_reply":"2026-01-03T13:07:45.656763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Analyse tag_0 features\ntag_0_features = meta[meta['tag_0'] == True]['feature'].tolist()\ntag_0_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:45.658540Z","iopub.execute_input":"2026-01-03T13:07:45.658897Z","iopub.status.idle":"2026-01-03T13:07:45.672195Z","shell.execute_reply.started":"2026-01-03T13:07:45.658860Z","shell.execute_reply":"2026-01-03T13:07:45.671470Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.heatmap(initial_train_0[tag_0_features].corr(), annot = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:45.673232Z","iopub.execute_input":"2026-01-03T13:07:45.673566Z","iopub.status.idle":"2026-01-03T13:07:46.067269Z","shell.execute_reply.started":"2026-01-03T13:07:45.673540Z","shell.execute_reply":"2026-01-03T13:07:46.066570Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target = new_train[['time_id', 'responder_6']]\ntarget.set_index('time_id', inplace = True)\nplt.figure(figsize = (10,5))\nplt.plot( target, label = 'Resp 6 over Day 0')\nplt.xlabel('time_id')\nplt.ylabel('responder_6')\nplt.grid()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:46.068232Z","iopub.execute_input":"2026-01-03T13:07:46.068527Z","iopub.status.idle":"2026-01-03T13:07:46.255291Z","shell.execute_reply.started":"2026-01-03T13:07:46.068497Z","shell.execute_reply":"2026-01-03T13:07:46.254482Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from statsmodels.tsa.stattools import adfuller\nresults = adfuller(target)\nresults","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:46.256357Z","iopub.execute_input":"2026-01-03T13:07:46.256686Z","iopub.status.idle":"2026-01-03T13:07:46.557391Z","shell.execute_reply.started":"2026-01-03T13:07:46.256651Z","shell.execute_reply":"2026-01-03T13:07:46.556757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target.cumsum().plot()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:46.558154Z","iopub.execute_input":"2026-01-03T13:07:46.558409Z","iopub.status.idle":"2026-01-03T13:07:46.741616Z","shell.execute_reply.started":"2026-01-03T13:07:46.558383Z","shell.execute_reply":"2026-01-03T13:07:46.740954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"resp = []\nfor col in new_train.columns:\n    if col.startswith('resp'):\n        resp.append(col)\nresp_df = new_train[resp]\ncorr_mat = resp_df.corr()\nsns.heatmap(corr_mat, annot = True)\n        ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:46.742650Z","iopub.execute_input":"2026-01-03T13:07:46.743375Z","iopub.status.idle":"2026-01-03T13:07:47.121552Z","shell.execute_reply.started":"2026-01-03T13:07:46.743344Z","shell.execute_reply":"2026-01-03T13:07:47.120852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compare feature correlations with resp_6, cum sum of resp_6\ndf = initial_train_0[feature_cols + ['responder_6']]\ncorr_mat = df.corr()['responder_6'].drop(index = 'responder_6', axis = 0).sort_values(ascending = False)\ncorr_mat","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:47.122543Z","iopub.execute_input":"2026-01-03T13:07:47.122869Z","iopub.status.idle":"2026-01-03T13:07:47.221322Z","shell.execute_reply.started":"2026-01-03T13:07:47.122840Z","shell.execute_reply":"2026-01-03T13:07:47.220518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_cols = []\nfor col in initial_train.columns:\n    if col.startswith('fea'):\n        feature_cols.append(col)\ndf = initial_train[feature_cols].isna().sum()\ndf.sort_values(ascending = False, inplace = True)\ndf = df[df>0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:47.222329Z","iopub.execute_input":"2026-01-03T13:07:47.223107Z","iopub.status.idle":"2026-01-03T13:07:48.573660Z","shell.execute_reply.started":"2026-01-03T13:07:47.223079Z","shell.execute_reply":"2026-01-03T13:07:48.572849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (15,10))\nplt.barh(data = df, y = df.index, width = df.values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:07:48.574722Z","iopub.execute_input":"2026-01-03T13:07:48.575172Z","iopub.status.idle":"2026-01-03T13:07:48.987693Z","shell.execute_reply.started":"2026-01-03T13:07:48.575137Z","shell.execute_reply":"2026-01-03T13:07:48.986862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_final = initial_train.drop(columns = df[df == df.max()].index.tolist())\ny = train_final['responder_6'].values\nX = train_final[[col for col in train_final.columns if col.startswith('fea') or col.startswith('sym')]]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:55:43.585212Z","iopub.execute_input":"2026-01-03T13:55:43.586074Z","iopub.status.idle":"2026-01-03T13:55:44.019694Z","shell.execute_reply.started":"2026-01-03T13:55:43.586033Z","shell.execute_reply":"2026-01-03T13:55:44.018984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_final","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:55:44.945192Z","iopub.execute_input":"2026-01-03T13:55:44.945514Z","iopub.status.idle":"2026-01-03T13:55:45.093373Z","shell.execute_reply.started":"2026-01-03T13:55:44.945483Z","shell.execute_reply":"2026-01-03T13:55:45.092579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = X.fillna(X.median())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:55:49.051602Z","iopub.execute_input":"2026-01-03T13:55:49.051973Z","iopub.status.idle":"2026-01-03T13:55:51.356605Z","shell.execute_reply.started":"2026-01-03T13:55:49.051942Z","shell.execute_reply":"2026-01-03T13:55:51.355196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"weights = train_final['weight'].values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:55:52.491047Z","iopub.execute_input":"2026-01-03T13:55:52.491382Z","iopub.status.idle":"2026-01-03T13:55:52.495546Z","shell.execute_reply.started":"2026-01-03T13:55:52.491343Z","shell.execute_reply":"2026-01-03T13:55:52.494778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def custom_score(y_pred, y_true, weights):\n    r2_score = 1 - (np.sum(weights * (y_true-y_pred)**2))/(np.sum(weights*(y_true**2)))\n    return r2_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:55:54.328481Z","iopub.execute_input":"2026-01-03T13:55:54.329269Z","iopub.status.idle":"2026-01-03T13:55:54.333566Z","shell.execute_reply.started":"2026-01-03T13:55:54.329235Z","shell.execute_reply":"2026-01-03T13:55:54.332814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train test split\nX = X.values\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test, weights_tr, weights_test = train_test_split(X,y,weights, test_size = 0.02, shuffle = False)\nfrom sklearn.linear_model import Ridge\nmodel = Ridge(alpha = 1)\nmodel.fit(X_train, y_train, sample_weight = weights_tr)\ny_pred = model.predict(X_test)\nprint(f'Training score:{custom_score(model.predict(X_train),y_train, weights_tr)}')\nprint(f'Validation/Test score: {custom_score(y_pred,y_test,weights_test)}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:55:55.874155Z","iopub.execute_input":"2026-01-03T13:55:55.874496Z","iopub.status.idle":"2026-01-03T13:55:57.998058Z","shell.execute_reply.started":"2026-01-03T13:55:55.874465Z","shell.execute_reply":"2026-01-03T13:55:57.996740Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Zero-Model Score: {custom_score(y_test, np.zeros_like(y_test), weights_test)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:19:05.264290Z","iopub.execute_input":"2026-01-03T13:19:05.264981Z","iopub.status.idle":"2026-01-03T13:19:05.269777Z","shell.execute_reply.started":"2026-01-03T13:19:05.264948Z","shell.execute_reply":"2026-01-03T13:19:05.269047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.plot(y_pred)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:36:58.890474Z","iopub.execute_input":"2026-01-03T13:36:58.890858Z","iopub.status.idle":"2026-01-03T13:36:59.104935Z","shell.execute_reply.started":"2026-01-03T13:36:58.890824Z","shell.execute_reply":"2026-01-03T13:36:59.104122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.plot(y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:35:53.140702Z","iopub.execute_input":"2026-01-03T13:35:53.141441Z","iopub.status.idle":"2026-01-03T13:35:53.710725Z","shell.execute_reply.started":"2026-01-03T13:35:53.141407Z","shell.execute_reply":"2026-01-03T13:35:53.709916Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"It seems like the model is failing to capture the scale of the target, maybe due to a lot of noise present and low correlations?","metadata":{}},{"cell_type":"code","source":"# Distribution of weights_tr vs weights_test\nax = sns.histplot(np.log(weights_tr), kde = True)\nax.set_xlabel('Log Weights')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:56:12.862001Z","iopub.execute_input":"2026-01-03T13:56:12.862735Z","iopub.status.idle":"2026-01-03T13:56:21.684175Z","shell.execute_reply.started":"2026-01-03T13:56:12.862690Z","shell.execute_reply":"2026-01-03T13:56:21.683357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_final['symbol_id'].value_counts()\nsymbols = train_final['symbol_id'].value_counts().index.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T14:04:03.318878Z","iopub.execute_input":"2026-01-03T14:04:03.319677Z","iopub.status.idle":"2026-01-03T14:04:03.345598Z","shell.execute_reply.started":"2026-01-03T14:04:03.319637Z","shell.execute_reply":"2026-01-03T14:04:03.344694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ax = sns.histplot(np.log(weights_test), kde = True)\nax.set_xlabel('Log Weights')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T13:53:00.921658Z","iopub.execute_input":"2026-01-03T13:53:00.922367Z","iopub.status.idle":"2026-01-03T13:53:01.409006Z","shell.execute_reply.started":"2026-01-03T13:53:00.922333Z","shell.execute_reply":"2026-01-03T13:53:01.408258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Does symbol_id affect weight distribution?\nfig, ax = plt.subplots(4, 5,figsize  = (20,25))\nax = ax.flatten()\nfor i, sym in enumerate(symbols):\n    df = train_final[train_final['symbol_id'] == sym]\n    weights = np.log(df['weight'])\n    sns.histplot(weights, kde = True, ax = ax[i])\n    ax[i].set_xlabel(f'Log Weights for Symbol = {sym}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T14:10:22.048187Z","iopub.execute_input":"2026-01-03T14:10:22.048948Z","iopub.status.idle":"2026-01-03T14:10:35.088203Z","shell.execute_reply.started":"2026-01-03T14:10:22.048908Z","shell.execute_reply":"2026-01-03T14:10:35.087278Z"}},"outputs":[],"execution_count":null}]}