{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-24T08:52:45.538225Z","iopub.execute_input":"2024-10-24T08:52:45.538807Z","iopub.status.idle":"2024-10-24T08:52:46.126263Z","shell.execute_reply.started":"2024-10-24T08:52:45.538748Z","shell.execute_reply":"2024-10-24T08:52:46.12495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import polars as pl\nimport pandas as pd\nimport numpy as np\nfrom sklearn.linear_model import Ridge\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings('ignore')\nimport matplotlib.pyplot as plt\nfrom statsmodels.tsa.seasonal import seasonal_decompose\n\nimport kaggle_evaluation.jane_street_inference_server\n\nimport random\n\ndef seed_everything(seed):\n    np.random.seed(seed)\n    random.seed(seed)\n\nseed_everything(seed=2024)\n\ntarget = \"responder_6\"\nop_path = f\"/kaggle/working\"\nip_path = f\"/kaggle/input/janestreet2024-dataload-v1\"\nstate = 42\nmethod = \"LGBM1R\"","metadata":{"execution":{"iopub.status.busy":"2024-10-24T08:52:46.135938Z","iopub.execute_input":"2024-10-24T08:52:46.137108Z","iopub.status.idle":"2024-10-24T08:52:48.330386Z","shell.execute_reply.started":"2024-10-24T08:52:46.137053Z","shell.execute_reply":"2024-10-24T08:52:48.329117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=9/part-0.parquet\")\ntrain=train.to_pandas()\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T08:52:48.332723Z","iopub.execute_input":"2024-10-24T08:52:48.333374Z","iopub.status.idle":"2024-10-24T08:53:00.286973Z","shell.execute_reply.started":"2024-10-24T08:52:48.333326Z","shell.execute_reply":"2024-10-24T08:53:00.285555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"symbol_counts = train['symbol_id'].value_counts()\nplt.bar(symbol_counts.index, symbol_counts.values)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T08:53:00.289537Z","iopub.execute_input":"2024-10-24T08:53:00.289985Z","iopub.status.idle":"2024-10-24T08:53:00.783482Z","shell.execute_reply.started":"2024-10-24T08:53:00.289943Z","shell.execute_reply":"2024-10-24T08:53:00.782114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Weight analysis","metadata":{}},{"cell_type":"markdown","source":"\nPortfolio weight represents the proportion of an individual asset's value relative to the total value of the portfolio. It indicates the percentage of the total investment allocated to a specific asset, helping to determine the exposure to that asset.\n\nFormula\nThe portfolio weight of an asset is calculated using the formula:\n\nPortfolio Weight of Asset = (Value of the Asset) / (Total Value of the Portfolio)\nValue of the Asset: The current market value of the individual asset.\nTotal Value of the Portfolio: The sum of the market values of all assets within the portfolio.\nExample\nIf the total portfolio value is $100,000 and one of the assets is worth $20,000, the portfolio weight of that asset would be:\n\n\nPortfolio Weight = 20,000 / 100,000 = 0.2 or 20%\n","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nplt.plot(train['weight'][:500])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T09:00:05.866526Z","iopub.execute_input":"2024-10-24T09:00:05.867094Z","iopub.status.idle":"2024-10-24T09:00:06.159466Z","shell.execute_reply.started":"2024-10-24T09:00:05.867045Z","shell.execute_reply":"2024-10-24T09:00:06.158104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train['weight'] > 5][:2]","metadata":{"execution":{"iopub.status.busy":"2024-10-24T09:07:54.320187Z","iopub.execute_input":"2024-10-24T09:07:54.321191Z","iopub.status.idle":"2024-10-24T09:07:54.512356Z","shell.execute_reply.started":"2024-10-24T09:07:54.321136Z","shell.execute_reply":"2024-10-24T09:07:54.51097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Weight Decomposition\n\nTime series decomposition is a technique used to break down a time series into several components that represent different aspects of the data. This method helps in understanding the underlying patterns within the time series and is commonly used in forecasting and anomaly detection.","metadata":{}},{"cell_type":"code","source":"decomposition = seasonal_decompose(train['weight'][:1000], model='additive', period=19)\n\n# Create a figure with a specific size\nfig, (ax1, ax2, ax3, ax4) = plt.subplots(4, 1, figsize=(18, 12))  # Change figsize as needed\n\n# Plot the decomposition components\ndecomposition.observed.plot(ax=ax1, legend=False, title='Observed')\ndecomposition.trend.plot(ax=ax2, legend=False, title='Trend')\ndecomposition.seasonal.plot(ax=ax3, legend=False, title='Seasonal')\ndecomposition.resid.plot(ax=ax4, legend=False, title='Residual')\n\n# Show the plot\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T09:11:26.783837Z","iopub.execute_input":"2024-10-24T09:11:26.784412Z","iopub.status.idle":"2024-10-24T09:11:27.991077Z","shell.execute_reply.started":"2024-10-24T09:11:26.784364Z","shell.execute_reply":"2024-10-24T09:11:27.989686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(14, 7))\n\n# Plot the first 1000 records only\nplt.scatter(train['time_id'][:1000], train['weight'][:1000], label='weight')\nplt.scatter(train['time_id'][:1000], train['responder_6'][:1000], label='Feature 2')\n\nplt.legend()\nplt.title('Feature 1 vs Feature 2 (First 1000 Records)')\nplt.xlabel('Date')\nplt.ylabel('Values')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T09:15:22.195231Z","iopub.execute_input":"2024-10-24T09:15:22.195774Z","iopub.status.idle":"2024-10-24T09:15:22.645217Z","shell.execute_reply.started":"2024-10-24T09:15:22.195728Z","shell.execute_reply":"2024-10-24T09:15:22.643702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\nsns.histplot(train['weight'], bins=50)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T08:53:01.315728Z","iopub.execute_input":"2024-10-24T08:53:01.316224Z","iopub.status.idle":"2024-10-24T08:53:10.86167Z","shell.execute_reply.started":"2024-10-24T08:53:01.316178Z","shell.execute_reply":"2024-10-24T08:53:10.859802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pairs with highest wieght overtime","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\naggregated = train.groupby('symbol_id')[['weight']].agg(['sum'])\nplt.bar(aggregated.index, aggregated['weight']['sum'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T08:53:10.864111Z","iopub.execute_input":"2024-10-24T08:53:10.864577Z","iopub.status.idle":"2024-10-24T08:53:11.773163Z","shell.execute_reply.started":"2024-10-24T08:53:10.86453Z","shell.execute_reply":"2024-10-24T08:53:11.771463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Distribution of weights over time","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\naggregated = train.groupby('date_id')[['weight']].agg(['sum'])\nplt.bar(aggregated.index, aggregated['weight']['sum'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T08:53:15.579413Z","iopub.execute_input":"2024-10-24T08:53:15.580183Z","iopub.status.idle":"2024-10-24T08:53:16.400193Z","shell.execute_reply.started":"2024-10-24T08:53:15.580123Z","shell.execute_reply":"2024-10-24T08:53:16.39825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"STD of weight is\", train['weight'].std())","metadata":{"execution":{"iopub.status.busy":"2024-10-24T08:53:16.404464Z","iopub.execute_input":"2024-10-24T08:53:16.405151Z","iopub.status.idle":"2024-10-24T08:53:16.485162Z","shell.execute_reply.started":"2024-10-24T08:53:16.405096Z","shell.execute_reply":"2024-10-24T08:53:16.482905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nplt.scatter(train['weight'][:15000], train['responder_6'][:15000])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T08:53:16.487323Z","iopub.execute_input":"2024-10-24T08:53:16.487798Z","iopub.status.idle":"2024-10-24T08:53:16.849741Z","shell.execute_reply.started":"2024-10-24T08:53:16.487752Z","shell.execute_reply":"2024-10-24T08:53:16.847553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Responder analysis","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nplt.plot(train['responder_6'][:5000])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T08:53:16.85319Z","iopub.execute_input":"2024-10-24T08:53:16.854088Z","iopub.status.idle":"2024-10-24T08:53:17.286292Z","shell.execute_reply.started":"2024-10-24T08:53:16.854014Z","shell.execute_reply":"2024-10-24T08:53:17.284456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['responder_6'].describe()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T08:53:17.288779Z","iopub.execute_input":"2024-10-24T08:53:17.289497Z","iopub.status.idle":"2024-10-24T08:53:17.62619Z","shell.execute_reply.started":"2024-10-24T08:53:17.289432Z","shell.execute_reply":"2024-10-24T08:53:17.624419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Responder Decompostion","metadata":{}},{"cell_type":"code","source":"train[train['responder_6'] > 3][:3]","metadata":{"execution":{"iopub.status.busy":"2024-10-24T09:17:35.968505Z","iopub.execute_input":"2024-10-24T09:17:35.969035Z","iopub.status.idle":"2024-10-24T09:17:36.050985Z","shell.execute_reply.started":"2024-10-24T09:17:35.968965Z","shell.execute_reply":"2024-10-24T09:17:36.04968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"decomposition = seasonal_decompose(train['responder_6'][:1000], model='additive', period=9)\n\n# Create a figure with a specific size\nfig, (ax1, ax2, ax3, ax4) = plt.subplots(4, 1, figsize=(18, 12))  # Change figsize as needed\n\n# Plot the decomposition components\ndecomposition.observed.plot(ax=ax1, legend=False, title='Observed')\ndecomposition.trend.plot(ax=ax2, legend=False, title='Trend')\ndecomposition.seasonal.plot(ax=ax3, legend=False, title='Seasonal')\ndecomposition.resid.plot(ax=ax4, legend=False, title='Residual')\n\n# Show the plot\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T09:17:42.165982Z","iopub.execute_input":"2024-10-24T09:17:42.166484Z","iopub.status.idle":"2024-10-24T09:17:43.80385Z","shell.execute_reply.started":"2024-10-24T09:17:42.166436Z","shell.execute_reply":"2024-10-24T09:17:43.80239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Responder has scaled between [-5, 5] with a mean score of -3.7","metadata":{}},{"cell_type":"code","source":"sns.histplot(train['responder_6'], bins=50)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T08:53:17.629034Z","iopub.execute_input":"2024-10-24T08:53:17.629613Z","iopub.status.idle":"2024-10-24T08:53:27.661298Z","shell.execute_reply.started":"2024-10-24T08:53:17.629554Z","shell.execute_reply":"2024-10-24T08:53:27.659477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the cumulative sum\nplt.figure(figsize=(16, 6))\ntrain['cumulative_sum'] = train['responder_6'].cumsum()\nplt.plot(train['cumulative_sum'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T08:53:27.663969Z","iopub.execute_input":"2024-10-24T08:53:27.664463Z","iopub.status.idle":"2024-10-24T08:53:28.99568Z","shell.execute_reply.started":"2024-10-24T08:53:27.66441Z","shell.execute_reply":"2024-10-24T08:53:28.994098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the cumulative sum\ntrain['cumulative_sum'] = train['weight'].cumsum()\nplt.plot(train['cumulative_sum'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T08:53:28.998895Z","iopub.execute_input":"2024-10-24T08:53:28.999812Z","iopub.status.idle":"2024-10-24T08:53:30.19788Z","shell.execute_reply.started":"2024-10-24T08:53:28.999747Z","shell.execute_reply":"2024-10-24T08:53:30.196575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Features analysis","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nplt.plot(train['feature_00'][:500])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T08:53:30.199438Z","iopub.execute_input":"2024-10-24T08:53:30.199951Z","iopub.status.idle":"2024-10-24T08:53:30.521782Z","shell.execute_reply.started":"2024-10-24T08:53:30.199885Z","shell.execute_reply":"2024-10-24T08:53:30.520541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(train['feature_00'], bins=50)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T18:40:18.434651Z","iopub.execute_input":"2024-10-19T18:40:18.43513Z","iopub.status.idle":"2024-10-19T18:40:27.017734Z","shell.execute_reply.started":"2024-10-19T18:40:18.435078Z","shell.execute_reply":"2024-10-19T18:40:27.015988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nsns.histplot(train['date_id'], bins=50)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T18:40:27.019946Z","iopub.execute_input":"2024-10-19T18:40:27.020411Z","iopub.status.idle":"2024-10-19T18:40:36.015352Z","shell.execute_reply.started":"2024-10-19T18:40:27.020362Z","shell.execute_reply":"2024-10-19T18:40:36.013561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['feature_00_cumulative_sum'] = train['feature_00'].cumsum()\nplt.plot(train['feature_00_cumulative_sum'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T12:43:23.9535Z","iopub.execute_input":"2024-10-19T12:43:23.954017Z","iopub.status.idle":"2024-10-19T12:43:24.898856Z","shell.execute_reply.started":"2024-10-19T12:43:23.953964Z","shell.execute_reply":"2024-10-19T12:43:24.897698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(train['feature_01'], bins=50)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T12:43:24.899978Z","iopub.execute_input":"2024-10-19T12:43:24.900349Z","iopub.status.idle":"2024-10-19T12:43:32.587593Z","shell.execute_reply.started":"2024-10-19T12:43:24.900294Z","shell.execute_reply":"2024-10-19T12:43:32.586434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\ntrain['feature_01_cumulative_sum'] = train['feature_01'].cumsum()\nplt.plot(train['feature_01_cumulative_sum'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:11:14.666636Z","iopub.execute_input":"2024-10-19T13:11:14.667113Z","iopub.status.idle":"2024-10-19T13:11:15.64809Z","shell.execute_reply.started":"2024-10-19T13:11:14.667067Z","shell.execute_reply":"2024-10-19T13:11:15.646704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explanetory Data Analysis","metadata":{}},{"cell_type":"code","source":"features_agg_sum = pd.DataFrame(train.iloc[:, 4:70].sum(), columns=['Value'])\nfeatures_agg_sum.plot(figsize=(24, 6))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:11:54.741933Z","iopub.execute_input":"2024-10-19T13:11:54.742502Z","iopub.status.idle":"2024-10-19T13:11:56.325309Z","shell.execute_reply.started":"2024-10-19T13:11:54.742451Z","shell.execute_reply":"2024-10-19T13:11:56.323791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['time_id'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T12:55:56.840248Z","iopub.execute_input":"2024-10-19T12:55:56.841532Z","iopub.status.idle":"2024-10-19T12:55:56.906089Z","shell.execute_reply.started":"2024-10-19T12:55:56.841477Z","shell.execute_reply":"2024-10-19T12:55:56.904618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=train[train['time_id']==0]\ndisplay(df)","metadata":{"execution":{"iopub.status.busy":"2024-10-20T08:26:40.646901Z","iopub.execute_input":"2024-10-20T08:26:40.647402Z","iopub.status.idle":"2024-10-20T08:26:40.669283Z","shell.execute_reply.started":"2024-10-20T08:26:40.647346Z","shell.execute_reply":"2024-10-20T08:26:40.66808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Corrolation Analysis","metadata":{}},{"cell_type":"code","source":"null_cols = []\n\n\n# remove all null columns\nnull_cols = train.columns[pd.isnull(train).all()].tolist()\ntrain = train.drop(null_cols, axis=1)\ntrain.shape\n\nnanValues = train.isnull().sum()\ntrain.dropna(inplace=True)\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-20T08:23:15.586543Z","iopub.execute_input":"2024-10-20T08:23:15.586989Z","iopub.status.idle":"2024-10-20T08:23:19.62229Z","shell.execute_reply.started":"2024-10-20T08:23:15.586938Z","shell.execute_reply":"2024-10-20T08:23:19.621051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# check the corrolation of features with target feature","metadata":{}},{"cell_type":"code","source":"# Drop all responder columns except responder_6\nfeature_columns = train.columns.difference(['responder_0', 'responder_1', 'responder_2', 'responder_3', \n                                         'responder_4', 'responder_5', 'responder_7', 'responder_8', \n                                         'responder_6'])\n\n# Compute the correlation of features with responder_6\ncorr_with_responder6 = train[feature_columns].corrwith(train['responder_6'])\n\n# Visualize the correlation with a bar plot\nplt.figure(figsize=(16, 12))\nsns.barplot(y=corr_with_responder6.index, x=corr_with_responder6.values, palette='coolwarm')\nplt.title('Correlation of Features with responder_6')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-20T08:27:16.640556Z","iopub.execute_input":"2024-10-20T08:27:16.640999Z","iopub.status.idle":"2024-10-20T08:27:30.007122Z","shell.execute_reply.started":"2024-10-20T08:27:16.640955Z","shell.execute_reply":"2024-10-20T08:27:30.005887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop all responder columns except responder_6\ndata = train.drop(columns=['responder_0', 'responder_1', 'responder_2', 'responder_3', \n                        'responder_4', 'responder_5', 'responder_7', 'responder_8'])\n\n# Set up the figure size for a large heatmap\nplt.figure(figsize=(20, 16))\n\n# Generate the heatmap for the dataset correlations\nsns.heatmap(data.corr(), annot=False, cmap='coolwarm', linewidths=0.5)\n\n# Add title\nplt.title('Heatmap of Correlations for All Columns (Including responder_6)')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-20T08:24:45.820703Z","iopub.execute_input":"2024-10-20T08:24:45.821666Z","iopub.status.idle":"2024-10-20T08:26:40.64455Z","shell.execute_reply.started":"2024-10-20T08:24:45.821613Z","shell.execute_reply":"2024-10-20T08:26:40.643299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols=train.columns.tolist()\n\nfor col in cols[3:]:\n    plt.figure(figsize=(12, 6)) \n    for symbol in df['symbol_id'].unique():\n        subset = df[df['symbol_id'] == symbol]\n        plt.plot(subset['date_id'], subset[col], label=symbol)\n\n    plt.xlabel('Date')\n    plt.ylabel(f'{col} Value')\n    plt.title(f'Symbol-wise {col}')\n    plt.legend(title='Symbol ID', bbox_to_anchor=(1.05, 1), loc='upper left')\n\n    plt.xticks(rotation=45)\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T12:57:40.300401Z","iopub.execute_input":"2024-10-19T12:57:40.30088Z","iopub.status.idle":"2024-10-19T12:59:02.070688Z","shell.execute_reply.started":"2024-10-19T12:57:40.300838Z","shell.execute_reply":"2024-10-19T12:59:02.069067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Runing the model","metadata":{}},{"cell_type":"code","source":"def custom_metric(y_true,y_pred,weight):\n    weighted_r2=1-(np.sum(weight*(y_true-y_pred)**2)/np.sum(weight*y_true**2))\n    return weighted_r2\nprint(\"read data\")\ntrain=pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=9/part-0.parquet\")\ntrain=train.to_pandas()\nprint(\"get X,y\")\n#相关性绝对值大于0.01的feature列\ncols=['feature_04', 'feature_06', 'feature_07', 'feature_08', 'feature_15', 'feature_16', 'feature_17', 'feature_19', 'feature_25', 'feature_36', 'feature_45', 'feature_56', 'feature_60', 'feature_66']\nX=train[cols].fillna(3).values\ny=train['responder_6'].values\nprint(\"train test split\")\nsplit=1300000#大约是8:2\nweights=train['weight'].values\ntrain_X,train_y,test_X,test_y,train_weight,test_weight=X[:-split],y[:-split],X[-split:],y[-split:],weights[:-split],weights[-split:]\nprint(f\"train_X.shape:{train_X.shape},test_X.shape:{test_X.shape}\")\nprint(\"fit and predict\")\nmodel=Ridge()\nmodel.fit(train_X,train_y)\ntrain_pred=model.predict(train_X)\ntest_pred=model.predict(test_X)\nprint(f\"train weighted_r2:{custom_metric(train_y,train_pred,weight=train_weight)}\")\nprint(f\"test weighted_r2:{custom_metric(test_y,test_pred,weight=test_weight)}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-16T12:29:31.839539Z","iopub.execute_input":"2024-10-16T12:29:31.840042Z","iopub.status.idle":"2024-10-16T12:29:45.826789Z","shell.execute_reply.started":"2024-10-16T12:29:31.840003Z","shell.execute_reply":"2024-10-16T12:29:45.825012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(test,lags):\n    cols=['feature_04', 'feature_06', 'feature_07', 'feature_08', 'feature_15', 'feature_16', 'feature_17', 'feature_19', 'feature_25', 'feature_36', 'feature_45', 'feature_56', 'feature_60', 'feature_66']\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    test_preds=model.predict(test[cols].to_pandas().fillna(3).values)\n    predictions = predictions.with_columns(pl.Series('responder_6', test_preds.ravel()))\n    return predictions\n\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"execution":{"iopub.status.busy":"2024-10-16T09:28:51.560008Z","iopub.execute_input":"2024-10-16T09:28:51.560499Z","iopub.status.idle":"2024-10-16T09:28:51.817574Z","shell.execute_reply.started":"2024-10-16T09:28:51.560456Z","shell.execute_reply":"2024-10-16T09:28:51.816302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}