{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom statsmodels.tsa.stattools import adfuller\nfrom statsmodels.tsa.arima.model import ARIMA\nfrom sklearn.metrics import mean_squared_error","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-09T07:40:09.654365Z","iopub.execute_input":"2024-09-09T07:40:09.654750Z","iopub.status.idle":"2024-09-09T07:40:11.038765Z","shell.execute_reply.started":"2024-09-09T07:40:09.654714Z","shell.execute_reply":"2024-09-09T07:40:11.037828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_df = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-09-09T07:40:13.337230Z","iopub.execute_input":"2024-09-09T07:40:13.338435Z","iopub.status.idle":"2024-09-09T07:41:20.683295Z","shell.execute_reply.started":"2024-09-09T07:40:13.338361Z","shell.execute_reply":"2024-09-09T07:41:20.682281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_df.t_dat = pd.to_datetime(transactions_df.t_dat)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T07:41:20.685007Z","iopub.execute_input":"2024-09-09T07:41:20.685348Z","iopub.status.idle":"2024-09-09T07:41:26.279690Z","shell.execute_reply.started":"2024-09-09T07:41:20.685306Z","shell.execute_reply":"2024-09-09T07:41:26.278678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dailySales = transactions_df.groupby('t_dat')['price'].sum().reset_index()","metadata":{"execution":{"iopub.status.busy":"2024-09-09T07:41:26.280920Z","iopub.execute_input":"2024-09-09T07:41:26.281310Z","iopub.status.idle":"2024-09-09T07:41:26.960108Z","shell.execute_reply.started":"2024-09-09T07:41:26.281269Z","shell.execute_reply":"2024-09-09T07:41:26.959167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set `t_dat` as the index\ndailySales.set_index('t_dat', inplace=True)\n\n# Ensure data is sorted by the index\ndailySales.sort_index(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T07:41:26.962153Z","iopub.execute_input":"2024-09-09T07:41:26.962467Z","iopub.status.idle":"2024-09-09T07:41:26.968907Z","shell.execute_reply.started":"2024-09-09T07:41:26.962436Z","shell.execute_reply":"2024-09-09T07:41:26.967916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result = adfuller(dailySales.index)  # Ensure dailySales is a pandas Series or a 1D array-like object\n\n# Output the results\nprint('ADF Statistic:', result[0])\nprint('p-value:', result[1])\nprint('Critical Values:', result[4])\n\n# Interpretation\nif result[1] > 0.05:\n    print(\"The data is likely non-stationary (Fail to reject H0)\")\nelse:\n    print(\"The data is likely stationary (Reject H0)\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-09T07:41:26.970130Z","iopub.execute_input":"2024-09-09T07:41:26.970485Z","iopub.status.idle":"2024-09-09T07:41:27.024896Z","shell.execute_reply.started":"2024-09-09T07:41:26.970452Z","shell.execute_reply":"2024-09-09T07:41:27.022259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ensure 'date' is a datetime type\ndailySales.index = pd.to_datetime(dailySales.index)\n\n# Set the frequency to daily ('D')\ndailySales = dailySales.asfreq('D')\n","metadata":{"execution":{"iopub.status.busy":"2024-09-09T07:41:27.026187Z","iopub.execute_input":"2024-09-09T07:41:27.029729Z","iopub.status.idle":"2024-09-09T07:41:27.041702Z","shell.execute_reply.started":"2024-09-09T07:41:27.029686Z","shell.execute_reply":"2024-09-09T07:41:27.040416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = dailySales.iloc[:-365]\ntest = dailySales.iloc[-365:]","metadata":{"execution":{"iopub.status.busy":"2024-09-09T07:41:27.043120Z","iopub.execute_input":"2024-09-09T07:41:27.044365Z","iopub.status.idle":"2024-09-09T07:41:27.050711Z","shell.execute_reply.started":"2024-09-09T07:41:27.044304Z","shell.execute_reply":"2024-09-09T07:41:27.049317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example with ARIMA(1,1,1)\nmodel = ARIMA(train['price'], order=(2, 1, 2))\nmodel_fit = model.fit()\n\n# Summary of the model\nprint(model_fit.summary())","metadata":{"execution":{"iopub.status.busy":"2024-09-08T15:20:37.782110Z","iopub.execute_input":"2024-09-08T15:20:37.783116Z","iopub.status.idle":"2024-09-08T15:20:38.016556Z","shell.execute_reply.started":"2024-09-08T15:20:37.783062Z","shell.execute_reply":"2024-09-08T15:20:38.015470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot residuals\nresiduals = model_fit.resid\nplt.figure(figsize=(20, 4))\n\nplt.subplot(121)\nplt.plot(residuals)\nplt.title('Residuals')\n\nplt.subplot(122)\nplt.hist(residuals, bins=30)\nplt.title('Histogram of Residuals')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-09-08T15:21:09.290175Z","iopub.execute_input":"2024-09-08T15:21:09.291068Z","iopub.status.idle":"2024-09-08T15:21:09.863517Z","shell.execute_reply.started":"2024-09-08T15:21:09.291026Z","shell.execute_reply":"2024-09-08T15:21:09.862562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from statsmodels.tsa.arima.model import ARIMA\nimport pandas as pd\n\n# Example: tuning ARIMA parameters\np = [0, 1, 2]  # Autoregressive terms\nd = [0, 1, 2]  # Differencing terms\nq = [0, 1, 2]  # Moving average terms\n\nbest_aic = float('inf')\nbest_order = None\nbest_model = None\n\nfor param in [(x, y, z) for x in p for y in d for z in q]:\n    try:\n        model = ARIMA(dailySales['price'], order=param)\n        model_fit = model.fit()\n        aic = model_fit.aic\n        if aic < best_aic:\n            best_aic = aic\n            best_order = param\n            best_model = model_fit\n    except:\n        continue\n\nprint('Best ARIMA Order:', best_order)\nprint(best_model.summary())\n","metadata":{"execution":{"iopub.status.busy":"2024-09-09T08:48:22.853576Z","iopub.execute_input":"2024-09-09T08:48:22.854307Z","iopub.status.idle":"2024-09-09T08:48:27.660749Z","shell.execute_reply.started":"2024-09-09T08:48:22.854269Z","shell.execute_reply":"2024-09-09T08:48:27.659779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from statsmodels.tsa.statespace.sarimax import SARIMAX\nseasonal_order = (2, 1, 2, 180)  # Example seasonal order; adjust p, d, q as needed\n\n# Fit the SARIMA model\nmodel_sarima = SARIMAX(train['price'], order=(2, 1, 2), seasonal_order=seasonal_order)\nmodel_fit_sarima = model_sarima.fit()\n\n# Print the summary of the model\nprint(model_fit_sarima.summary())","metadata":{"execution":{"iopub.status.busy":"2024-09-09T08:01:45.003872Z","iopub.execute_input":"2024-09-09T08:01:45.004631Z","iopub.status.idle":"2024-09-09T08:39:04.185485Z","shell.execute_reply.started":"2024-09-09T08:01:45.004586Z","shell.execute_reply":"2024-09-09T08:39:04.184510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Forecast the validation period\nforecast_steps = len(test)\nforecast_sarima = model_fit_sarima.get_forecast(steps=forecast_steps)\nforecast_sarima_mean = forecast_sarima.predicted_mean\n\n# Create a DataFrame for the forecast\nforecast_index = test.index\nforecast_df_sarima = pd.DataFrame({'forecast': forecast_sarima_mean}, index=forecast_index)\n\n# Plot the forecast against the actual validation data\nimport matplotlib.pyplot as plt\nplt.figure(figsize=(12, 6))\nplt.plot(train.index, train['price'], label='Training Data')\nplt.plot(test.index, test['price'], label='Validation Data')\nplt.plot(forecast_df_sarima.index, forecast_df_sarima['forecast'], label='Forecast (SARIMA)', color='green')\nplt.legend()\nplt.show()\n\n# Calculate Mean Squared Error on the validation set\nfrom sklearn.metrics import mean_squared_error\nmse_sarima = mean_squared_error(test['price'], forecast_sarima_mean)\nprint(f'Mean Squared Error (SARIMA): {mse_sarima}')\n","metadata":{"execution":{"iopub.status.busy":"2024-09-09T08:46:55.115890Z","iopub.execute_input":"2024-09-09T08:46:55.116332Z","iopub.status.idle":"2024-09-09T08:47:02.196174Z","shell.execute_reply.started":"2024-09-09T08:46:55.116289Z","shell.execute_reply":"2024-09-09T08:47:02.195324Z"},"trusted":true},"execution_count":null,"outputs":[]}]}