{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":178547,"sourceType":"modelInstanceVersion","modelInstanceId":152096,"modelId":174550}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error\nfrom keras.models import Sequential\nfrom keras.layers import Dense, LSTM\nimport keras\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-29T07:55:17.321105Z","iopub.execute_input":"2024-12-29T07:55:17.321405Z","iopub.status.idle":"2024-12-29T07:55:30.226392Z","shell.execute_reply.started":"2024-12-29T07:55:17.321374Z","shell.execute_reply":"2024-12-29T07:55:30.225125Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Prophet forecasting model**","metadata":{}},{"cell_type":"code","source":"conda install -c anaconda ephem","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"conda install -c conda-forge pystan","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"conda install -c conda-forge fbprophet","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"conda install -c conda-forge libarchive","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T18:51:17.881181Z","iopub.execute_input":"2024-12-01T18:51:17.881601Z","iopub.status.idle":"2024-12-01T18:53:04.423365Z","shell.execute_reply.started":"2024-12-01T18:51:17.881563Z","shell.execute_reply":"2024-12-01T18:53:04.421922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!conda install -c conda-forge fbprophet -y","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T18:59:48.297058Z","iopub.execute_input":"2024-12-01T18:59:48.297512Z","iopub.status.idle":"2024-12-01T19:03:44.962728Z","shell.execute_reply.started":"2024-12-01T18:59:48.297476Z","shell.execute_reply":"2024-12-01T19:03:44.960046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install --upgrade plotly","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T19:03:44.968025Z","iopub.execute_input":"2024-12-01T19:03:44.969326Z","iopub.status.idle":"2024-12-01T19:04:32.759305Z","shell.execute_reply.started":"2024-12-01T19:03:44.969181Z","shell.execute_reply":"2024-12-01T19:04:32.757723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport pyarrow.parquet as pa\nimport matplotlib.pyplot as plt\nimport pyarrow.parquet as pa\nimport os\nimport polars as pl\n\nimport kaggle_evaluation.jane_street_inference_server\n\nfrom prophet import Prophet\n\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nplt.style.use('ggplot')\nplt.style.use('fivethirtyeight')\n\ndef mean_absolute_percentage_error(y_true, y_pred): \n    \"\"\"Calculates MAPE given y_true and y_pred\"\"\"\n    y_true, y_pred = np.array(y_true), np.array(y_pred)\n    return np.mean(np.abs((y_true - y_pred) / y_true)) * 100","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T21:12:40.981518Z","iopub.execute_input":"2024-12-29T21:12:40.981936Z","iopub.status.idle":"2024-12-29T21:12:42.992088Z","shell.execute_reply.started":"2024-12-29T21:12:40.981899Z","shell.execute_reply":"2024-12-29T21:12:42.991024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef import_training_data(input_path:str):\n    table = pa.read_table(input_path) \n    # tabl\n    df = table.to_pandas() \n    return df\n    # Taking tanspose so the printing dataset will easy. \n# df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T21:12:42.993867Z","iopub.execute_input":"2024-12-29T21:12:42.994308Z","iopub.status.idle":"2024-12-29T21:12:42.999720Z","shell.execute_reply.started":"2024-12-29T21:12:42.994276Z","shell.execute_reply":"2024-12-29T21:12:42.998364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prep_training_data(df:pd.DataFrame) -> pd.DataFrame:\n    df['date_time'] = df['date_id'] + df['time_id']\n    # df['date_id'].astype(int)\n    df_short = df[['date_time', 'responder_6', 'symbol_id']]\n    df_short\n    temp_company_1 = df_short[df_short['symbol_id'] == 1][['date_time', 'responder_6']]\n    # temp_company_1 = temp_company_1.reset_index(drop=True)['responder_6']\n    temp_company_1\n    \n    temp_company_1 = temp_company_1.set_index('date_time')\n    return temp_company_1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T21:12:43.001246Z","iopub.execute_input":"2024-12-29T21:12:43.001889Z","iopub.status.idle":"2024-12-29T21:12:43.013304Z","shell.execute_reply.started":"2024-12-29T21:12:43.001839Z","shell.execute_reply":"2024-12-29T21:12:43.012198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# temp_company_1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T21:12:43.116192Z","iopub.execute_input":"2024-12-29T21:12:43.116536Z","iopub.status.idle":"2024-12-29T21:12:43.121219Z","shell.execute_reply.started":"2024-12-29T21:12:43.116507Z","shell.execute_reply":"2024-12-29T21:12:43.120147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef split_data(df: pd.DataFrame) -> pd.DataFrame:\n    split_date = 800\n    pjme_train = df.loc[df.index <= split_date].copy()\n    pjme_test = df.loc[df.index > split_date].copy()\n    return pjme_train, pjme_test\n# pjme_train = df\n# pjme_test = df\n\n# Plot train and test so you can see where we have split\n# pjme_test \\\n#     .rename(columns={'responder_6': 'TEST SET'}) \\\n#     .join(pjme_train.rename(columns={'responder_6': 'TRAINING SET'}),\n#           how='outer') \\\n#     .plot(figsize=(10, 5), title='PJM East', style='.', ms=1)\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T21:12:43.451166Z","iopub.execute_input":"2024-12-29T21:12:43.451606Z","iopub.status.idle":"2024-12-29T21:12:43.457368Z","shell.execute_reply.started":"2024-12-29T21:12:43.451520Z","shell.execute_reply":"2024-12-29T21:12:43.456238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_phrophet(df_train:pd.DataFrame):\n    pjme_train_prophet = df_train.reset_index() \\\n        .rename(columns={'date_time':'ds',\n                         'responder_6':'y'})\n\n    return pjme_train_prophet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T21:12:43.794380Z","iopub.execute_input":"2024-12-29T21:12:43.795568Z","iopub.status.idle":"2024-12-29T21:12:43.800947Z","shell.execute_reply.started":"2024-12-29T21:12:43.795514Z","shell.execute_reply":"2024-12-29T21:12:43.799527Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def fit_phrophet(train_output):\n    # %%time\n    model = Prophet()\n    return model.fit(train_output)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T21:12:44.133464Z","iopub.execute_input":"2024-12-29T21:12:44.133977Z","iopub.status.idle":"2024-12-29T21:12:44.139821Z","shell.execute_reply.started":"2024-12-29T21:12:44.133929Z","shell.execute_reply":"2024-12-29T21:12:44.138710Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def make_prophet_ready():\n    input_path = '/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=0/part-0.parquet'\n    df = import_training_data(input_path)\n    df_clean = prep_training_data(df)\n    df_train, df_test = split_data(df_clean)\n    output_train = train_phrophet(df_train)\n    return fit_phrophet(output_train)\n\nmodel = make_prophet_ready()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T21:12:44.488827Z","iopub.execute_input":"2024-12-29T21:12:44.489720Z","iopub.status.idle":"2024-12-29T21:13:07.968301Z","shell.execute_reply.started":"2024-12-29T21:12:44.489653Z","shell.execute_reply":"2024-12-29T21:13:07.966808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef make_prediction(df_test:pd.DataFrame): #model comes from the fit_phrophet\n    pjme_test_prophet = df_test.reset_index() \\\n        .rename(columns={'date_time':'ds',\n                         'responder_6':'y'})\n    \n    return model.predict(pjme_test_prophet)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T21:13:07.970275Z","iopub.execute_input":"2024-12-29T21:13:07.970611Z","iopub.status.idle":"2024-12-29T21:13:07.976036Z","shell.execute_reply.started":"2024-12-29T21:13:07.970579Z","shell.execute_reply":"2024-12-29T21:13:07.974970Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def make_predictions_ready(predictions_df: pd.DataFrame) -> pd.DataFrame:\n    predictions = predictions_df.reset_index().rename(columns={'index':'row_id', 'yhat':'responder_6'})\n    return predictions[['row_id', 'responder_6']]\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T21:13:07.977395Z","iopub.execute_input":"2024-12-29T21:13:07.977763Z","iopub.status.idle":"2024-12-29T21:13:07.992094Z","shell.execute_reply.started":"2024-12-29T21:13:07.977731Z","shell.execute_reply":"2024-12-29T21:13:07.991058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mean_absolute_percentage_error(y_true=pjme_test['responder_6'],\n                   y_pred=pjme_test_fcst['yhat'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T21:13:07.994040Z","iopub.execute_input":"2024-12-29T21:13:07.994425Z","iopub.status.idle":"2024-12-29T21:13:08.365326Z","shell.execute_reply.started":"2024-12-29T21:13:07.994393Z","shell.execute_reply":"2024-12-29T21:13:08.363735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"future = model.make_future_dataframe(periods=365*24, freq='h', include_history=False)\nforecast = model.predict(future)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T21:13:08.366322Z","iopub.status.idle":"2024-12-29T21:13:08.366729Z","shell.execute_reply.started":"2024-12-29T21:13:08.366531Z","shell.execute_reply":"2024-12-29T21:13:08.366549Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"forecast[['ds','yhat']].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T21:13:08.368313Z","iopub.status.idle":"2024-12-29T21:13:08.368826Z","shell.execute_reply.started":"2024-12-29T21:13:08.368581Z","shell.execute_reply":"2024-12-29T21:13:08.368610Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pyarrow.parquet as pa\n# test = pd.read_parquet('/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet') \ntest = pd.read_parquet('/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet/date_id=0/')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T21:14:31.478216Z","iopub.execute_input":"2024-12-29T21:14:31.478884Z","iopub.status.idle":"2024-12-29T21:14:31.503776Z","shell.execute_reply.started":"2024-12-29T21:14:31.478846Z","shell.execute_reply":"2024-12-29T21:14:31.502868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lags_ : pl.DataFrame | None = None\n\n\n# Replace this function with your inference code.\n# You can return either a Pandas or Polars dataframe, though Polars is recommended.\n# Each batch of predictions (except the very first) must be returned within 1 minute of the batch features being provided.\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    # All the responders from the previous day are passed in at time_id == 0. We save them in a global variable for access at every time_id.\n    # Use them as extra features, if you like.\n    global lags_\n    if lags is not None:\n        lags_ = lags\n\n    # Replace this section with your own predictions\n    # predictions = test.select(\n    #     'row_id',\n    #     pl.lit(0.0).alias('responder_6'),\n    # )\n    # print(test.columns)\n    # test = prep_training_data(test)\n    test['date_time'] = test['date_id'] + test['time_id']\n    predictions_raw = make_prediction(test)\n    # print(predictions_raw)\n\n    predictions_test = make_predictions_ready(predictions_raw)\n    predictions = pl.DataFrame(\n        {\n            \"row_id\": predictions_test[\"row_id\"],\n            \"responder_6\": predictions_test['responder_6'],\n        }\n    )\n    \n    if isinstance(predictions, pl.DataFrame):\n        # print('polar')\n        # print (predictions.columns == ['row_id', 'responder_6'])\n        assert predictions.columns == ['row_id', 'responder_6']\n    elif isinstance(predictions, pd.DataFrame):\n        # print(pandas)\n        assert (predictions.columns == ['row_id', 'responder_6']).all()\n    else:\n        raise TypeError('The predict function must return a DataFrame')\n    # Confirm has as many rows as the test data.\n    assert len(predictions) == len(test)\n    print(type(predictions))\n    print(predictions)\n    # predictions = predictions.to_frame()\n    return predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T21:52:46.919632Z","iopub.execute_input":"2024-12-29T21:52:46.920098Z","iopub.status.idle":"2024-12-29T21:52:46.929516Z","shell.execute_reply.started":"2024-12-29T21:52:46.920047Z","shell.execute_reply":"2024-12-29T21:52:46.928241Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T21:52:49.028672Z","iopub.execute_input":"2024-12-29T21:52:49.029173Z","iopub.status.idle":"2024-12-29T21:52:49.146483Z","shell.execute_reply.started":"2024-12-29T21:52:49.029124Z","shell.execute_reply":"2024-12-29T21:52:49.144971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"temp = predict(test, None)\ntemp","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T21:52:49.228608Z","iopub.execute_input":"2024-12-29T21:52:49.229791Z","iopub.status.idle":"2024-12-29T21:52:49.288396Z","shell.execute_reply.started":"2024-12-29T21:52:49.229731Z","shell.execute_reply":"2024-12-29T21:52:49.287177Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}