{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:36:04.984106Z","iopub.execute_input":"2024-12-21T09:36:04.984465Z","iopub.status.idle":"2024-12-21T09:36:05.415525Z","shell.execute_reply.started":"2024-12-21T09:36:04.984434Z","shell.execute_reply":"2024-12-21T09:36:05.414475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"###Import Libraries and Initialize Variables\n\nimport pandas as pd\n\n# Initialize a list to hold samples from each file\nsamples = []\n\n###Step 2: Load Data from Multiple Files\n\nfor i in range(10):\n    # Define the path for each partition file\n    file_path = f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}/part-0.parquet\"\n    \n    # Read the data from the Parquet file into a DataFrame\n    chunk = pd.read_parquet(file_path)\n    \n    # Take a sample of 100,000 rows from the current DataFrame\n    sample_chunk = chunk.sample(n=100000, random_state=42)\n    \n    # Append the sampled data to the list\n    samples.append(sample_chunk)\n\n###Step 3: Combine the Samples into a Single DataFrame\n\n# Combine all sampled data into one DataFrame\nsample_df = pd.concat(samples, ignore_index=True)\n\n# Display the first rows of the combined DataFrame\nsample_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:36:05.417056Z","iopub.execute_input":"2024-12-21T09:36:05.417601Z","iopub.status.idle":"2024-12-21T09:37:28.804127Z","shell.execute_reply.started":"2024-12-21T09:36:05.417565Z","shell.execute_reply":"2024-12-21T09:37:28.802918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install prophet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:37:28.805929Z","iopub.execute_input":"2024-12-21T09:37:28.806253Z","iopub.status.idle":"2024-12-21T09:37:34.593562Z","shell.execute_reply.started":"2024-12-21T09:37:28.806224Z","shell.execute_reply":"2024-12-21T09:37:34.592242Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from datetime import timedelta\nfrom prophet import Prophet\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error\n\n# Step 1: Prepare the data\n# Assuming date_id is an integer sequence starting from 0 or 1\nstart_date = pd.Timestamp(\"2020-01-01\")  # Replace with an appropriate start date\nsample_df['ds'] = sample_df['date_id'].apply(lambda x: start_date + timedelta(days=int(x)))\n\n# Aggregate responder_6 by date\naggregated_df = sample_df.groupby('ds').agg({'responder_6': 'mean'}).reset_index()\n\n# Rename columns for Prophet\naggregated_df.rename(columns={'responder_6': 'y'}, inplace=True)\n\n# Step 2: Initialize and train the Prophet model\nmodel = Prophet(\n    daily_seasonality=False,  # Disable if not applicable\n    weekly_seasonality=True,  # Enable weekly seasonality\n    yearly_seasonality=True   # Enable yearly seasonality\n)\n\n# Fit the model to the data\nmodel.fit(aggregated_df)\n\n# Step 3: Make future predictions\nfuture = model.make_future_dataframe(periods=30)  # Predict for the next 30 days\nforecast = model.predict(future)\n\n# Step 4: Plot actual data and forecast\nplt.figure(figsize=(12, 6))\nplt.plot(aggregated_df['ds'], aggregated_df['y'], label='Actual', color='blue')\nplt.plot(forecast['ds'], forecast['yhat'], label='Forecast', color='orange')\nplt.fill_between(\n    forecast['ds'],\n    forecast['yhat_lower'],\n    forecast['yhat_upper'],\n    color='orange',\n    alpha=0.3,\n    label='Uncertainty Interval'\n)\nplt.legend()\nplt.title(\"Actual vs Forecast - Responder_6\")\nplt.xlabel(\"Date\")\nplt.ylabel(\"Responder_6\")\nplt.grid()\nplt.show()\n\n# Step 5: Analyze model components\nmodel.plot_components(forecast)\nplt.show()\n\n# Step 6: Evaluate model performance\n# Filter for predictions on existing data\neval_forecast = forecast[forecast['ds'].isin(aggregated_df['ds'])]\nactuals = aggregated_df.set_index('ds').loc[eval_forecast['ds']]['y']\npredicted = eval_forecast['yhat']\n\nmae = mean_absolute_error(actuals, predicted)\nrmse = np.sqrt(mean_squared_error(actuals, predicted))\n\nprint(f\"Model Performance Metrics:\")\nprint(f\"Mean Absolute Error (MAE): {mae:.4f}\")\nprint(f\"Root Mean Squared Error (RMSE): {rmse:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T09:37:34.595147Z","iopub.execute_input":"2024-12-21T09:37:34.595503Z","iopub.status.idle":"2024-12-21T09:37:47.767355Z","shell.execute_reply.started":"2024-12-21T09:37:34.595460Z","shell.execute_reply":"2024-12-21T09:37:47.766357Z"}},"outputs":[],"execution_count":null}]}