{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport polars as pl\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom statsmodels.tsa.seasonal import seasonal_decompose\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.metrics import mean_squared_error\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-04T20:56:25.926591Z","iopub.execute_input":"2025-01-04T20:56:25.926988Z","iopub.status.idle":"2025-01-04T20:56:27.905722Z","shell.execute_reply.started":"2025-01-04T20:56:25.926948Z","shell.execute_reply":"2025-01-04T20:56:27.904839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# credit - https://www.kaggle.com/code/sahanabasapathi/jane-street-linearregression/edit\ndef import_unordered_data(num_samples=int(5e5), verbose=False):\n    \n    base_path = \"/kaggle/input/jane-street-real-time-market-data-forecasting/\"\n    random_state = 42\n    samples = []\n    total_entries_count = 0\n    \n    for i in range(4):\n\n        file_path = f\"train.parquet/partition_id={i}/part-0.parquet\"\n        \n        if verbose:\n            print(f\"importing file: {file_path}\")\n\n        try:\n            sample = pd.read_parquet(os.path.join(base_path, file_path))\n            total_entries_count += len(sample)\n            \n            if verbose:\n                print(f\"number of entries in '{file_path}': {len(sample):,}\")\n    \n            if num_samples < len(sample):\n                sample = sample.sample(n=num_samples, random_state=random_state)\n            samples.append(sample)\n    \n            if verbose:\n                print(' ')\n                \n        except Exception as e:\n            print(f\"error: {e}\")\n\n    sample_df = pd.concat(samples, ignore_index=True)\n    \n    if verbose:\n        print(f\"importing files complete\")\n        print(f'number of entries in full dataset: {total_entries_count:,}')\n        print(f'number of entries in dataframe: {len(sample_df):,}')\n\n    return sample_df\ntrain_df = import_unordered_data(verbose=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T20:56:29.437091Z","iopub.execute_input":"2025-01-04T20:56:29.437609Z","iopub.status.idle":"2025-01-04T20:56:54.209088Z","shell.execute_reply.started":"2025-01-04T20:56:29.437576Z","shell.execute_reply":"2025-01-04T20:56:54.208047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns_with_missing_values = train_df.columns[train_df.isnull().any()]\nprint(columns_with_missing_values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T23:40:22.283225Z","iopub.execute_input":"2025-01-02T23:40:22.283641Z","iopub.status.idle":"2025-01-02T23:40:22.665823Z","shell.execute_reply.started":"2025-01-02T23:40:22.283603Z","shell.execute_reply":"2025-01-02T23:40:22.664680Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"non_numeric_columns = train_df.select_dtypes(exclude='number').columns\nprint(non_numeric_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T02:34:20.043167Z","iopub.execute_input":"2025-01-03T02:34:20.043582Z","iopub.status.idle":"2025-01-03T02:34:20.056254Z","shell.execute_reply.started":"2025-01-03T02:34:20.043545Z","shell.execute_reply":"2025-01-03T02:34:20.054812Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Auto-Arima \n\nhttps://alkaline-ml.com/pmdarima/modules/generated/pmdarima.arima.auto_arima.html","metadata":{}},{"cell_type":"code","source":"'''\nSome viz before that \nOkay, I don't think we have enough memory to perform this \n\nresults - don't see anything obvious\n'''\n\nviz_df = train_df.drop(['date_id','time_id','symbol_id','responder_0', 'responder_1', 'responder_2','responder_3', 'responder_4', 'responder_5', 'responder_6',\n       'responder_7', 'responder_8', 'weight'], axis=1)\ncorr_matrix = viz_df.corr()\nplt.figure(figsize=(12, 10))  \nsns.heatmap(corr_matrix, cmap='coolwarm', fmt=\".2f\")\nplt.title(\"Correlation Map of Features\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T00:49:45.034847Z","iopub.execute_input":"2024-12-20T00:49:45.036265Z","iopub.status.idle":"2024-12-20T00:50:57.222503Z","shell.execute_reply.started":"2024-12-20T00:49:45.036204Z","shell.execute_reply":"2024-12-20T00:50:57.221276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nLooks like responder 6 has some corelation with responser 0, 1 and 2\n0,1,2 and 6,7,8 are also somewhat co-related\n\n\n'''\nsubset = train_df[['responder_0', 'responder_1', 'responder_2','responder_3', 'responder_4', 'responder_5', 'responder_6',\n       'responder_7', 'responder_8']]\ncorr_matrix = subset.corr()\nplt.figure(figsize=(6, 5))  \nsns.heatmap(corr_matrix, cmap='coolwarm', fmt=\".2f\")\nplt.title(\"Correlation Map of Features\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T02:22:40.690151Z","iopub.execute_input":"2024-12-22T02:22:40.690554Z","iopub.status.idle":"2024-12-22T02:22:42.377416Z","shell.execute_reply.started":"2024-12-22T02:22:40.690506Z","shell.execute_reply":"2024-12-22T02:22:42.376414Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"code","source":"'''\n#https://otexts.com/fpp2/stationarity.html\n\nObservations\nWe have 20 symbol ids\nLot's of missing values\nAll features are numeric \nI think it's worth checking how the responder's values change over time - if it's stationary, then AR, ARIMA models might not cut it\n\n\nConclusions\nRemove feature 21,26,27,31 since all or most of it is null\n\n\n'''","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#credit - https://www.kaggle.com/code/allegich/jane-street-time-series-analysis-eda-ensemble\n\n#df_train = sample_df\nplt.figure(figsize=(20, 3))    # Plot missing values\nplt.bar(x=train_df.isna().sum().index, height=(train_df.isna().sum().values/len(train_df))*100, color=\"red\", label='missing')   # analog: using missingno\nplt.xticks(rotation=90)\nplt.title(f'Percentage of missing values over the {len(train_df)} samples which have a target')\nplt.grid()\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T03:10:37.982954Z","iopub.execute_input":"2024-12-22T03:10:37.983361Z","iopub.status.idle":"2024-12-22T03:10:39.924276Z","shell.execute_reply.started":"2024-12-22T03:10:37.983306Z","shell.execute_reply":"2024-12-22T03:10:39.923210Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train_df['symbol_id'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:59:40.698981Z","iopub.execute_input":"2024-12-22T23:59:40.699464Z","iopub.status.idle":"2024-12-22T23:59:40.735189Z","shell.execute_reply.started":"2024-12-22T23:59:40.699413Z","shell.execute_reply":"2024-12-22T23:59:40.734050Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def clean(df):\n    '''\n    let's not remove any cols since that's causing problems\n    '''\n    # to_remove = ['feature_21', 'feature_26', 'feature_27', 'feature_31', 'feature_00', 'feature_01',\n    #    'feature_02', 'feature_03', 'feature_04']\n    # for col in to_remove:\n    #     df.drop([col], axis=1, inplace = True)\n\n    # df.drop(['feature_21', 'feature_26', 'feature_27', 'feature_31', 'feature_00', 'feature_01',\n    #    'feature_02', 'feature_03', 'feature_04'], axis=1, inplace=True)\n    \n    #remove those cols that start with responder since it's not in test\n    res_features = [col for col in df.columns if col.startswith(\"responder_\")]\n    df.drop(res_features, axis=1, inplace=True)\n    df.replace([np.inf, -np.inf], 0, inplace=True)\n    df_imputed = df.apply(lambda col: col.fillna(-1, axis=0))\n    return df_imputed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T20:56:59.913836Z","iopub.execute_input":"2025-01-04T20:56:59.914225Z","iopub.status.idle":"2025-01-04T20:56:59.920808Z","shell.execute_reply.started":"2025-01-04T20:56:59.914190Z","shell.execute_reply":"2025-01-04T20:56:59.919639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"w = train_df['weight']\ny = train_df['responder_6']\ntrain_df.drop(['responder_6','weight'], axis = 1, inplace=True)\nX = train_df\ndf_ = clean(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T20:57:02.236940Z","iopub.execute_input":"2025-01-04T20:57:02.237832Z","iopub.status.idle":"2025-01-04T20:57:05.022181Z","shell.execute_reply.started":"2025-01-04T20:57:02.237792Z","shell.execute_reply":"2025-01-04T20:57:05.021213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T20:57:05.023590Z","iopub.execute_input":"2025-01-04T20:57:05.023901Z","iopub.status.idle":"2025-01-04T20:57:05.031096Z","shell.execute_reply.started":"2025-01-04T20:57:05.023870Z","shell.execute_reply":"2025-01-04T20:57:05.030146Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Ridge Regression ","metadata":{}},{"cell_type":"code","source":"def ridge_reg(X,y,w):\n    \n    X_train, X_test, y_train, y_test, w_train, w_test = train_test_split(X, y, w, test_size=0.2, random_state=42)\n    scaler = StandardScaler()\n    X_train_scaled = scaler.fit_transform(X_train)\n    X_test_scaled = scaler.transform(X_test)\n    \n    ridge_model = RidgeCV(store_cv_values=True).fit(X_train_scaled, y_train, w_train)  # ridgeCV tunes the alpha parameter\n    \n    #R2 is used while calling the score\n    r2_score = ridge_model.score(X_test_scaled, y_test, sample_weight=w_test)\n\n    print(\"R2 score:\",r2_score)\n    \n    # print(\"Ridge coefficients:\", ridge_model.coef_)\n    coef_df = pd.DataFrame({'Feature': X.columns, 'Coefficient': ridge_model.coef_})\n    sorted_coef_df = coef_df.reindex(coef_df['Coefficient'].abs().sort_values(ascending=False).index)\n    print(sorted_coef_df.head())\n    return ridge_model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T21:18:46.492335Z","iopub.execute_input":"2025-01-04T21:18:46.493348Z","iopub.status.idle":"2025-01-04T21:18:46.500559Z","shell.execute_reply.started":"2025-01-04T21:18:46.493305Z","shell.execute_reply":"2025-01-04T21:18:46.499375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rm = ridge_reg(df_,y,w)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T21:18:47.366616Z","iopub.execute_input":"2025-01-04T21:18:47.367305Z","iopub.status.idle":"2025-01-04T21:19:09.976378Z","shell.execute_reply.started":"2025-01-04T21:18:47.367267Z","shell.execute_reply":"2025-01-04T21:19:09.974645Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"#credit - https://www.kaggle.com/code/gkitchen/jane-street-first-steps\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None):\n    \n    dts = ['date_id', 'time_id', 'symbol_id']\n    global lags_\n    if lags is not None:\n        lags_ = lags\n    \n    # Convert Polars DataFrame to Pandas DataFrame for prediction\n    test_pandas = clean(test.to_pandas())\n    test_pandas.shape\n    # Extract the features for the model input\n    features = [col for col in test_pandas.columns if col.startswith(\"feature_\")]\n    responder_6_predictions = rm.predict(test_pandas[features + dts])\n    \n    # Convert predictions to a Polars Series\n    predictions_series = pl.Series(\n        name='responder_6',\n        values=np.clip(responder_6_predictions, a_min=-5, a_max=5)\n    )\n    \n    # Add the predictions to the original Polars DataFrame\n    predictions = test.select('row_id').with_columns([predictions_series])\n\n    print(predictions)\n    return predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T21:20:24.317592Z","iopub.execute_input":"2025-01-04T21:20:24.317975Z","iopub.status.idle":"2025-01-04T21:20:24.325449Z","shell.execute_reply.started":"2025-01-04T21:20:24.317941Z","shell.execute_reply":"2025-01-04T21:20:24.324214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T21:20:26.141584Z","iopub.execute_input":"2025-01-04T21:20:26.141948Z","iopub.status.idle":"2025-01-04T21:20:26.199359Z","shell.execute_reply.started":"2025-01-04T21:20:26.141916Z","shell.execute_reply":"2025-01-04T21:20:26.198389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}