{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-24T10:19:35.876646Z","iopub.execute_input":"2022-07-24T10:19:35.877023Z","iopub.status.idle":"2022-07-24T10:19:35.887463Z","shell.execute_reply.started":"2022-07-24T10:19:35.876992Z","shell.execute_reply":"2022-07-24T10:19:35.886250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ts_train=pd.read_csv('../input/store-sales-time-series-forecasting/train.csv')\nts_train.tail()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:35.919561Z","iopub.execute_input":"2022-07-24T10:19:35.920560Z","iopub.status.idle":"2022-07-24T10:19:37.904120Z","shell.execute_reply.started":"2022-07-24T10:19:35.920518Z","shell.execute_reply":"2022-07-24T10:19:37.903185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setup feedback system\nfrom learntools.core import binder\nbinder.bind(globals())\nfrom learntools.time_series.ex3 import *\n\n# Setup notebook\nfrom pathlib import Path\nfrom learntools.time_series.style import *  # plot style settings\nfrom learntools.time_series.utils import plot_periodogram, seasonal_plot\n\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport seaborn as sns\nfrom sklearn.linear_model import LinearRegression\nfrom statsmodels.tsa.deterministic import CalendarFourier, DeterministicProcess\n\n\ncomp_dir = Path('../input/store-sales-time-series-forecasting')\n\nholidays_events = pd.read_csv(\n    comp_dir / \"holidays_events.csv\",\n    dtype={\n        'type': 'category',\n        'locale': 'category',\n        'locale_name': 'category',\n        'description': 'category',\n        'transferred': 'bool',\n    },\n    parse_dates=['date'],\n    infer_datetime_format=True,\n)\nholidays_events = holidays_events.set_index('date').to_period('D')\n\nstore_sales = pd.read_csv(\n    comp_dir / 'train.csv',\n    usecols=['store_nbr', 'family', 'date', 'sales'],\n    dtype={\n        'store_nbr': 'category',\n        'family': 'category',\n        'sales': 'float32',\n    },\n    parse_dates=['date'],\n    infer_datetime_format=True,\n)\nstore_sales['date'] = store_sales.date.dt.to_period('D')\nstore_sales = store_sales.set_index(['store_nbr', 'family', 'date']).sort_index()\naverage_sales = (\n    store_sales\n    .groupby('date').mean() # group sales by date and average\n    .squeeze() \n    .loc['2017'] # only take the 2017 data.\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:37.906365Z","iopub.execute_input":"2022-07-24T10:19:37.907131Z","iopub.status.idle":"2022-07-24T10:19:41.888528Z","shell.execute_reply.started":"2022-07-24T10:19:37.907063Z","shell.execute_reply":"2022-07-24T10:19:41.887422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"store_sales['sales']","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:41.889764Z","iopub.execute_input":"2022-07-24T10:19:41.890234Z","iopub.status.idle":"2022-07-24T10:19:41.904776Z","shell.execute_reply.started":"2022-07-24T10:19:41.890190Z","shell.execute_reply":"2022-07-24T10:19:41.903812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"avg_sales=store_sales.groupby('family').mean()\nplt.figure(figsize=(20,10))\nsns.barplot(x='sales',y=avg_sales.index,data=avg_sales)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:41.907042Z","iopub.execute_input":"2022-07-24T10:19:41.907408Z","iopub.status.idle":"2022-07-24T10:19:42.620302Z","shell.execute_reply.started":"2022-07-24T10:19:41.907377Z","shell.execute_reply":"2022-07-24T10:19:42.618788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"average_sales.plot()\nplt.title('Average sales in 2017')\nplt.ylabel('Sales')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:42.622212Z","iopub.execute_input":"2022-07-24T10:19:42.623000Z","iopub.status.idle":"2022-07-24T10:19:42.980526Z","shell.execute_reply.started":"2022-07-24T10:19:42.622949Z","shell.execute_reply":"2022-07-24T10:19:42.979333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from statsmodels.tsa.stattools import adfuller\nfrom numpy import log\nresult = adfuller(average_sales)\nprint('ADF Statistic: %f' % result[0])\nprint('p-value: %f' % result[1])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:42.982426Z","iopub.execute_input":"2022-07-24T10:19:42.982785Z","iopub.status.idle":"2022-07-24T10:19:43.000478Z","shell.execute_reply.started":"2022-07-24T10:19:42.982753Z","shell.execute_reply":"2022-07-24T10:19:42.999677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\nfrom warnings import simplefilter\n\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport seaborn as sns\nfrom sklearn.linear_model import LinearRegression\nfrom statsmodels.tsa.deterministic import CalendarFourier, DeterministicProcess\n\nsimplefilter(\"ignore\")\n\n# Set Matplotlib defaults\nplt.style.use(\"seaborn-whitegrid\")\nplt.rc(\"figure\", autolayout=True, figsize=(11, 5))\nplt.rc(\n    \"axes\",\n    labelweight=\"bold\",\n    labelsize=\"large\",\n    titleweight=\"bold\",\n    titlesize=16,\n    titlepad=10,\n)\nplot_params = dict(\n    color=\"0.75\",\n    style=\".-\",\n    markeredgecolor=\"0.25\",\n    markerfacecolor=\"0.25\",\n    legend=False,\n)\n%config InlineBackend.figure_format = 'retina'\n\n\n# annotations: https://stackoverflow.com/a/49238256/5769929\ndef seasonal_plot(X, y, period, freq, ax=None):\n    if ax is None:\n        _, ax = plt.subplots()\n    palette = sns.color_palette(\"husl\", n_colors=X[period].nunique(),)\n    ax = sns.lineplot(\n        x=freq,\n        y=y,\n        hue=period,\n        data=X,\n        ci=False,\n        ax=ax,\n        palette=palette,\n        legend=False,\n    )\n    ax.set_title(f\"Seasonal Plot ({period}/{freq})\")\n    for line, name in zip(ax.lines, X[period].unique()):\n        y_ = line.get_ydata()[-1]\n        ax.annotate(\n            name,\n            xy=(1, y_),\n            xytext=(6, 0),\n            color=line.get_color(),\n            xycoords=ax.get_yaxis_transform(),\n            textcoords=\"offset points\",\n            size=14,\n            va=\"center\",\n        )\n    return ax\n\n\ndef plot_periodogram(ts, detrend='linear', ax=None):\n    from scipy.signal import periodogram\n    fs = pd.Timedelta(\"1Y\") / pd.Timedelta(\"1D\")\n    freqencies, spectrum = periodogram(\n        ts,\n        fs=fs,\n        detrend=detrend,\n        window=\"boxcar\",\n        scaling='spectrum',\n    )\n    if ax is None:\n        _, ax = plt.subplots()\n    ax.step(freqencies, spectrum, color=\"purple\")\n    ax.set_xscale(\"log\")\n    ax.set_xticks([1, 2, 4, 6, 12, 26, 52, 104])\n    ax.set_xticklabels(\n        [\n            \"Annual (1)\",\n            \"Semiannual (2)\",\n            \"Quarterly (4)\",\n            \"Bimonthly (6)\",\n            \"Monthly (12)\",\n            \"Biweekly (26)\",\n            \"Weekly (52)\",\n            \"Semiweekly (104)\",\n        ],\n        rotation=30,\n    )\n    ax.ticklabel_format(axis=\"y\", style=\"sci\", scilimits=(0, 0))\n    ax.set_ylabel(\"Variance\")\n    ax.set_title(\"Periodogram\")\n    return ax\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:43.002542Z","iopub.execute_input":"2022-07-24T10:19:43.003007Z","iopub.status.idle":"2022-07-24T10:19:43.044112Z","shell.execute_reply.started":"2022-07-24T10:19:43.002964Z","shell.execute_reply":"2022-07-24T10:19:43.043067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = average_sales.to_frame()\nX[\"week\"] = X.index.week\nX[\"day\"] = X.index.dayofweek\nseasonal_plot(X, y='sales', period='week', freq='day');","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:43.045761Z","iopub.execute_input":"2022-07-24T10:19:43.046574Z","iopub.status.idle":"2022-07-24T10:19:43.850645Z","shell.execute_reply.started":"2022-07-24T10:19:43.046539Z","shell.execute_reply":"2022-07-24T10:19:43.849521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_periodogram(average_sales);","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:43.852437Z","iopub.execute_input":"2022-07-24T10:19:43.853160Z","iopub.status.idle":"2022-07-24T10:19:44.321661Z","shell.execute_reply.started":"2022-07-24T10:19:43.853114Z","shell.execute_reply":"2022-07-24T10:19:44.320477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = average_sales.copy()\n\n# YOUR CODE HERE\nfourier = CalendarFourier(freq='M',order=4)\ndp = DeterministicProcess(\n    index=y.index,\n    constant=True,\n    order=1,\n    seasonal = True,\n    additional_terms=[fourier],\n    drop=True,\n)\nX = dp.in_sample() ","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:44.327514Z","iopub.execute_input":"2022-07-24T10:19:44.328293Z","iopub.status.idle":"2022-07-24T10:19:44.345959Z","shell.execute_reply.started":"2022-07-24T10:19:44.328242Z","shell.execute_reply":"2022-07-24T10:19:44.344947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = LinearRegression().fit(X, y)\ny_pred = pd.Series(\n    model.predict(X),\n    index=X.index,\n    name='Fitted',\n)\n\ny_pred = pd.Series(model.predict(X), index=X.index)\nax = y.plot(**plot_params, alpha=0.5, title=\"Average Sales\", ylabel=\"items sold\")\nax = y_pred.plot(ax=ax, label=\"Seasonal\")\nax.legend();","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:44.347430Z","iopub.execute_input":"2022-07-24T10:19:44.348216Z","iopub.status.idle":"2022-07-24T10:19:44.828757Z","shell.execute_reply.started":"2022-07-24T10:19:44.348178Z","shell.execute_reply":"2022-07-24T10:19:44.827190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_deseason = y - y_pred\n\nfig, (ax1, ax2) = plt.subplots(2, 1, sharex=True, sharey=True, figsize=(10, 7))\nax1 = plot_periodogram(y, ax=ax1)\nax1.set_title(\"Product Sales Frequency Components\")\nax2 = plot_periodogram(y_deseason, ax=ax2);\nax2.set_title(\"Deseasonalized\");","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:44.830806Z","iopub.execute_input":"2022-07-24T10:19:44.831354Z","iopub.status.idle":"2022-07-24T10:19:45.598987Z","shell.execute_reply.started":"2022-07-24T10:19:44.831304Z","shell.execute_reply":"2022-07-24T10:19:45.597902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"holidays = (\n    holidays_events\n    .query(\"locale in ['National', 'Regional']\")\n    .loc['2017':'2017-08-15', ['description']]\n    .assign(description=lambda x: x.description.cat.remove_unused_categories())\n)\ndisplay(holidays)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:45.600611Z","iopub.execute_input":"2022-07-24T10:19:45.600931Z","iopub.status.idle":"2022-07-24T10:19:45.620126Z","shell.execute_reply.started":"2022-07-24T10:19:45.600902Z","shell.execute_reply":"2022-07-24T10:19:45.618713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.subplot(2,1,1)\naverage_sales.plot(**plot_params)\nplt.plot_date(holidays.index, average_sales[holidays.index], color='C3')\nplt.title('National and Regional Holidays');\n\nplt.subplot(2,1,2)\ny_deseason.plot(**plot_params)\nplt.plot_date(holidays.index, y_deseason[holidays.index], color='C3')\nplt.title('National and Regional Holidays(deseasoned)')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:45.621936Z","iopub.execute_input":"2022-07-24T10:19:45.622930Z","iopub.status.idle":"2022-07-24T10:19:46.285078Z","shell.execute_reply.started":"2022-07-24T10:19:45.622884Z","shell.execute_reply":"2022-07-24T10:19:46.281451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\nohe = OneHotEncoder(sparse=False)\n\nX_holidays = pd.DataFrame(\n    ohe.fit_transform(holidays),\n    index=holidays.index,\n    columns=holidays.description.unique(),\n)\n# Pandas solution\nX_holidays = pd.get_dummies(holidays)\n\nX2 = X.join(X_holidays, on='date').fillna(0.0)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:46.286424Z","iopub.execute_input":"2022-07-24T10:19:46.286750Z","iopub.status.idle":"2022-07-24T10:19:46.303948Z","shell.execute_reply.started":"2022-07-24T10:19:46.286721Z","shell.execute_reply":"2022-07-24T10:19:46.302319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(X2)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:46.305286Z","iopub.execute_input":"2022-07-24T10:19:46.305582Z","iopub.status.idle":"2022-07-24T10:19:46.355286Z","shell.execute_reply.started":"2022-07-24T10:19:46.305554Z","shell.execute_reply":"2022-07-24T10:19:46.354482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = LinearRegression().fit(X2, y)\ny_pred2 = pd.Series(\n    model.predict(X2),\n    index=X2.index,\n    name='Fitted',\n)\n\ny_pred2 = pd.Series(model.predict(X2), index=X2.index)\nax = y.plot(**plot_params, alpha=0.5, title=\"Average Sales\", ylabel=\"items sold\")\nax = y_pred2.plot(ax=ax, label=\"Seasonal\")\nax.legend();","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:46.356438Z","iopub.execute_input":"2022-07-24T10:19:46.357500Z","iopub.status.idle":"2022-07-24T10:19:46.838029Z","shell.execute_reply.started":"2022-07-24T10:19:46.357467Z","shell.execute_reply":"2022-07-24T10:19:46.836859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_total=(y_pred + y_pred2)/2\nplt.figure(figsize=(20,10))\nax = y.plot(**plot_params, alpha=0.5, title=\"Holiday Average Sales Prediction(Monthly)\", ylabel=\"items sold\")\nax = y_pred_total.plot(ax=ax, label=\"Seasonal\",color='C0')\nplt.plot_date(holidays.index, average_sales[holidays.index], color='C3',label='Holiday Real')\nplt.plot_date(holidays.index, y_pred_total[holidays.index], color='C1',label='Holiday Predicted')\nax.legend();","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:46.839594Z","iopub.execute_input":"2022-07-24T10:19:46.840583Z","iopub.status.idle":"2022-07-24T10:19:47.567848Z","shell.execute_reply.started":"2022-07-24T10:19:46.840544Z","shell.execute_reply":"2022-07-24T10:19:47.566837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = store_sales.unstack(['store_nbr', 'family']).loc[\"2017\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:47.569136Z","iopub.execute_input":"2022-07-24T10:19:47.569454Z","iopub.status.idle":"2022-07-24T10:19:48.771745Z","shell.execute_reply.started":"2022-07-24T10:19:47.569425Z","shell.execute_reply":"2022-07-24T10:19:48.770714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create training data\nfourier = CalendarFourier(freq='M', order=4)\ndp = DeterministicProcess(\n    index=y.index,\n    constant=True,\n    order=1,\n    seasonal=True,\n    additional_terms=[fourier],\n    drop=True,\n)\nX = dp.in_sample()\nX['NewYear'] = (X.index.dayofyear == 1)\n# X['Holiday'] = hol\nmodel = LinearRegression(fit_intercept=False)\nmodel.fit(X, y)\ny_pred = pd.DataFrame(model.predict(X), index=X.index, columns=y.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:48.773232Z","iopub.execute_input":"2022-07-24T10:19:48.773771Z","iopub.status.idle":"2022-07-24T10:19:48.836645Z","shell.execute_reply.started":"2022-07-24T10:19:48.773739Z","shell.execute_reply":"2022-07-24T10:19:48.835368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STORE_NBR = '1'  # 1 - 54\nFAMILY = 'PRODUCE'\n# Uncomment to see a list of product families\ndisplay(store_sales.index.get_level_values('family').unique())\n\nax = y.loc(axis=1)['sales', STORE_NBR, FAMILY].plot(**plot_params)\nax = y_pred.loc(axis=1)['sales', STORE_NBR, FAMILY].plot(ax=ax)\nax.set_title(f'{FAMILY} Sales at Store {STORE_NBR}');","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:48.838212Z","iopub.execute_input":"2022-07-24T10:19:48.839374Z","iopub.status.idle":"2022-07-24T10:19:49.419335Z","shell.execute_reply.started":"2022-07-24T10:19:48.839326Z","shell.execute_reply":"2022-07-24T10:19:49.418240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_test = pd.read_csv(\n#     comp_dir / 'test.csv',\n#     dtype={\n#         'store_nbr': 'category',\n#         'family': 'category',\n#         'onpromotion': 'uint32',\n#     },\n#     parse_dates=['date'],\n#     infer_datetime_format=True,\n# )\n# df_test['date'] = df_test.date.dt.to_period('D')\n# df_test = df_test.set_index(['store_nbr', 'family', 'date']).sort_index()\n\n# # Create features for test set\n# X_test = dp.out_of_sample(steps=16)\n# X_test.index.name = 'date'\n# test_hol=[]\n# for i in X_test.index.dayofyear:\n#         if any(i==holidays.index.dayofyear):\n#             test_hol.append(True)\n#         else:\n#             test_hol.append(False)   \n# X_test['NewYear'] = (X_test.index.dayofyear == 1)\n# # X_test['Holiday'] = test_hol\n\n\n# y_submit = pd.DataFrame(model.predict(X_test), index=X_test.index, columns=y.columns)\n# y_submit = y_submit.stack(['store_nbr', 'family'])\n# y_submit = y_submit.join(df_test.id).reindex(columns=['id', 'sales'])\n# y_submit.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:49.420850Z","iopub.execute_input":"2022-07-24T10:19:49.421660Z","iopub.status.idle":"2022-07-24T10:19:49.915675Z","shell.execute_reply.started":"2022-07-24T10:19:49.421608Z","shell.execute_reply":"2022-07-24T10:19:49.914720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_test","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:49.917184Z","iopub.execute_input":"2022-07-24T10:19:49.918363Z","iopub.status.idle":"2022-07-24T10:19:49.937957Z","shell.execute_reply.started":"2022-07-24T10:19:49.918316Z","shell.execute_reply":"2022-07-24T10:19:49.936854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_test","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:49.939202Z","iopub.execute_input":"2022-07-24T10:19:49.939486Z","iopub.status.idle":"2022-07-24T10:19:49.981914Z","shell.execute_reply.started":"2022-07-24T10:19:49.939460Z","shell.execute_reply":"2022-07-24T10:19:49.980720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_submit","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:19:49.985517Z","iopub.execute_input":"2022-07-24T10:19:49.985889Z","iopub.status.idle":"2022-07-24T10:19:50.007061Z","shell.execute_reply.started":"2022-07-24T10:19:49.985857Z","shell.execute_reply":"2022-07-24T10:19:50.005849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Next We check for cyclicity\n# Create Hybrid model","metadata":{}},{"cell_type":"code","source":"#deseasoned \ny_ds=y_deseason.copy()\ny_ds=y_ds.to_frame(name='sales')\ny_ds","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:20:22.003567Z","iopub.execute_input":"2022-07-24T10:20:22.003941Z","iopub.status.idle":"2022-07-24T10:20:22.016381Z","shell.execute_reply.started":"2022-07-24T10:20:22.003911Z","shell.execute_reply":"2022-07-24T10:20:22.015623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame({\n    'y': y_ds.sales,\n    'y_lag_1': y_ds.sales.shift(1),\n    'y_lag_2': y_ds.sales.shift(2),    \n})\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:21:00.037149Z","iopub.execute_input":"2022-07-24T10:21:00.037694Z","iopub.status.idle":"2022-07-24T10:21:00.058565Z","shell.execute_reply.started":"2022-07-24T10:21:00.037647Z","shell.execute_reply":"2022-07-24T10:21:00.057001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from warnings import simplefilter\n\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom scipy.signal import periodogram\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.model_selection import train_test_split\nfrom statsmodels.graphics.tsaplots import plot_pacf\n\nsimplefilter(\"ignore\")\n\n# Set Matplotlib defaults\nplt.style.use(\"seaborn-whitegrid\")\nplt.rc(\"figure\", autolayout=True, figsize=(11, 4))\nplt.rc(\n    \"axes\",\n    labelweight=\"bold\",\n    labelsize=\"large\",\n    titleweight=\"bold\",\n    titlesize=16,\n    titlepad=10,\n)\nplot_params = dict(\n    color=\"0.75\",\n    style=\".-\",\n    markeredgecolor=\"0.25\",\n    markerfacecolor=\"0.25\",\n)\n%config InlineBackend.figure_format = 'retina'\n\n\ndef lagplot(x, y=None, lag=1, standardize=False, ax=None, **kwargs):\n    from matplotlib.offsetbox import AnchoredText\n    x_ = x.shift(lag)\n    if standardize:\n        x_ = (x_ - x_.mean()) / x_.std()\n    if y is not None:\n        y_ = (y - y.mean()) / y.std() if standardize else y\n    else:\n        y_ = x\n    corr = y_.corr(x_)\n    if ax is None:\n        fig, ax = plt.subplots()\n    scatter_kws = dict(\n        alpha=0.75,\n        s=3,\n    )\n    line_kws = dict(color='C3', )\n    ax = sns.regplot(x=x_,\n                     y=y_,\n                     scatter_kws=scatter_kws,\n                     line_kws=line_kws,\n                     lowess=True,\n                     ax=ax,\n                     **kwargs)\n    at = AnchoredText(\n        f\"{corr:.2f}\",\n        prop=dict(size=\"large\"),\n        frameon=True,\n        loc=\"upper left\",\n    )\n    at.patch.set_boxstyle(\"square, pad=0.0\")\n    ax.add_artist(at)\n    ax.set(title=f\"Lag {lag}\", xlabel=x_.name, ylabel=y_.name)\n    return ax\n\n\ndef plot_lags(x, y=None, lags=6, nrows=1, lagplot_kwargs={}, **kwargs):\n    import math\n    kwargs.setdefault('nrows', nrows)\n    kwargs.setdefault('ncols', math.ceil(lags / nrows))\n    kwargs.setdefault('figsize', (kwargs['ncols'] * 2, nrows * 2 + 0.5))\n    fig, axs = plt.subplots(sharex=True, sharey=True, squeeze=False, **kwargs)\n    for ax, k in zip(fig.get_axes(), range(kwargs['nrows'] * kwargs['ncols'])):\n        if k + 1 <= lags:\n            ax = lagplot(x, y, lag=k + 1, ax=ax, **lagplot_kwargs)\n            ax.set_title(f\"Lag {k + 1}\", fontdict=dict(fontsize=14))\n            ax.set(xlabel=\"\", ylabel=\"\")\n        else:\n            ax.axis('off')\n    plt.setp(axs[-1, :], xlabel=x.name)\n    plt.setp(axs[:, 0], ylabel=y.name if y is not None else x.name)\n    fig.tight_layout(w_pad=0.1, h_pad=0.1)\n    return fig\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:21:05.224305Z","iopub.execute_input":"2022-07-24T10:21:05.224726Z","iopub.status.idle":"2022-07-24T10:21:05.272747Z","shell.execute_reply.started":"2022-07-24T10:21:05.224680Z","shell.execute_reply":"2022-07-24T10:21:05.271840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = plot_lags(y_ds.sales, lags=12, nrows=2)\n_ = plot_pacf(y_ds.sales, lags=12)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:36:23.097014Z","iopub.execute_input":"2022-07-24T10:36:23.097565Z","iopub.status.idle":"2022-07-24T10:36:25.471918Z","shell.execute_reply.started":"2022-07-24T10:36:23.097516Z","shell.execute_reply":"2022-07-24T10:36:25.470542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_lags(ts, lags):\n    return pd.concat(\n        {\n            f'y_lag_{i}': ts.shift(i)\n            for i in range(1, lags + 1)\n        },\n        axis=1)\n\n\nX = make_lags(y_ds.sales, lags=2)\nX = X.fillna(0.0)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:21:16.727598Z","iopub.execute_input":"2022-07-24T10:21:16.727999Z","iopub.status.idle":"2022-07-24T10:21:16.737009Z","shell.execute_reply.started":"2022-07-24T10:21:16.727965Z","shell.execute_reply":"2022-07-24T10:21:16.736027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create target series and data splits\nfrom xgboost import XGBRegressor as xgbr\ny = y_ds.sales.copy()\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=60, shuffle=False)\n\n# Fit and predict\nmodel = xgbr()  # `fit_intercept=True` since we didn't use DeterministicProcess\nmodel.fit(X_train, y_train)\ny_pred = pd.Series(model.predict(X_train), index=y_train.index)\ny_fore = pd.Series(model.predict(X_test), index=y_test.index)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:21:20.885362Z","iopub.execute_input":"2022-07-24T10:21:20.885802Z","iopub.status.idle":"2022-07-24T10:21:21.167686Z","shell.execute_reply.started":"2022-07-24T10:21:20.885765Z","shell.execute_reply":"2022-07-24T10:21:21.166775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax = y_train.plot(**plot_params)\nax = y_test.plot(**plot_params)\nax = y_pred.plot(ax=ax)\n_ = y_fore.plot(ax=ax, color='C3')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:21:24.079643Z","iopub.execute_input":"2022-07-24T10:21:24.080493Z","iopub.status.idle":"2022-07-24T10:21:24.494823Z","shell.execute_reply.started":"2022-07-24T10:21:24.080451Z","shell.execute_reply":"2022-07-24T10:21:24.493448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax = y_test.plot(**plot_params)\n_ = y_fore.plot(ax=ax, color='C3')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:21:36.245066Z","iopub.execute_input":"2022-07-24T10:21:36.245560Z","iopub.status.idle":"2022-07-24T10:21:36.717046Z","shell.execute_reply.started":"2022-07-24T10:21:36.245509Z","shell.execute_reply":"2022-07-24T10:21:36.715910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_ma = y_test.rolling(\n    window=7,    \n    center=True\n).mean()            \n\n\n# Plot\nax = y_ma.plot(style='.-',label='deseasoned')\nax = y_test.plot()\nax.set_title(\"Seven-Day Moving Average\");\nax.legend()\n# Check your answer\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:35:02.729758Z","iopub.execute_input":"2022-07-24T10:35:02.730180Z","iopub.status.idle":"2022-07-24T10:35:03.254680Z","shell.execute_reply.started":"2022-07-24T10:35:02.730145Z","shell.execute_reply":"2022-07-24T10:35:03.253167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setup feedback system\nfrom learntools.core import binder\nbinder.bind(globals())\nfrom learntools.time_series.ex5 import *\n\n# Setup notebook\nfrom pathlib import Path\nfrom learntools.time_series.style import *  # plot style settings\n\nimport matplotlib.pyplot as plt\nimport pandas as pd\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.preprocessing import LabelEncoder\nfrom statsmodels.tsa.deterministic import DeterministicProcess\nfrom xgboost import XGBRegressor\n\n\ncomp_dir = Path('../input/store-sales-time-series-forecasting')\ndata_dir = Path(\"../input/ts-course-data\")\n\nstore_sales = pd.read_csv(\n    comp_dir / 'train.csv',\n    usecols=['store_nbr', 'family', 'date', 'sales', 'onpromotion'],\n    dtype={\n        'store_nbr': 'category',\n        'family': 'category',\n        'sales': 'float32',\n    },\n    parse_dates=['date'],\n    infer_datetime_format=True,\n)\nstore_sales['date'] = store_sales.date.dt.to_period('D')\nstore_sales = store_sales.set_index(['store_nbr', 'family', 'date']).sort_index()\n\nfamily_sales = (\n    store_sales\n    .groupby(['family', 'date'])\n    .mean()\n    .unstack('family')\n    .loc['2017']\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:42:01.709621Z","iopub.execute_input":"2022-07-24T10:42:01.710043Z","iopub.status.idle":"2022-07-24T10:42:05.972674Z","shell.execute_reply.started":"2022-07-24T10:42:01.710008Z","shell.execute_reply":"2022-07-24T10:42:05.971409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"family_sales.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:42:11.855326Z","iopub.execute_input":"2022-07-24T10:42:11.855864Z","iopub.status.idle":"2022-07-24T10:42:11.899372Z","shell.execute_reply.started":"2022-07-24T10:42:11.855816Z","shell.execute_reply":"2022-07-24T10:42:11.898145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# You'll add fit and predict methods to this minimal class\nclass BoostedHybrid:\n    def __init__(self, model_1, model_2):\n        self.model_1 = model_1\n        self.model_2 = model_2\n        self.y_columns = None  # store column names from fit method\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:42:28.270959Z","iopub.execute_input":"2022-07-24T10:42:28.271392Z","iopub.status.idle":"2022-07-24T10:42:28.276792Z","shell.execute_reply.started":"2022-07-24T10:42:28.271355Z","shell.execute_reply":"2022-07-24T10:42:28.275954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fit(self, X_1, X_2, y):\n    # YOUR CODE HERE: fit self.model_1\n    self.model_1.fit(X_1,y)\n\n    y_fit = pd.DataFrame(\n        # YOUR CODE HERE: make predictions with self.model_1\n        self.model_1.predict(X_1),\n        index=X_1.index, columns=y.columns,\n    )\n\n    # YOUR CODE HERE: compute residuals\n    y_resid = y-y_fit\n    y_resid = y_resid.stack().squeeze() # wide to long\n\n    # YOUR CODE HERE: fit self.model_2 on residuals\n    self.model_2.fit(X_2, y_resid)\n\n    # Save column names for predict method\n    self.y_columns = y.columns\n    # Save data for question checking\n    self.y_fit = y_fit\n    self.y_resid = y_resid\n\n\n# Add method to class\nBoostedHybrid.fit = fit","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:42:47.574547Z","iopub.execute_input":"2022-07-24T10:42:47.574972Z","iopub.status.idle":"2022-07-24T10:42:47.583059Z","shell.execute_reply.started":"2022-07-24T10:42:47.574938Z","shell.execute_reply":"2022-07-24T10:42:47.581874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(self, X_1, X_2):\n    y_pred = pd.DataFrame(\n        # YOUR CODE HERE: predict with self.model_1\n        self.model_1.predict(X_1),\n        index=X_1.index, columns=self.y_columns,\n    )\n    y_pred = y_pred.stack().squeeze()  # wide to long\n\n    # YOUR CODE HERE: add self.model_2 predictions to y_pred\n    y_pred += self.model_2.predict(X_2)\n    \n    return y_pred.unstack()  # long to wide\n\n\n# Add method to class\nBoostedHybrid.predict = predict\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:43:03.122874Z","iopub.execute_input":"2022-07-24T10:43:03.123265Z","iopub.status.idle":"2022-07-24T10:43:03.130373Z","shell.execute_reply.started":"2022-07-24T10:43:03.123232Z","shell.execute_reply":"2022-07-24T10:43:03.129347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Target series\ny = family_sales.loc[:, 'sales']\n\n\n# X_1: Features for Linear Regression\ndp = DeterministicProcess(index=y.index, order=1)\nX_1 = dp.in_sample()\n\n\n# X_2: Features for XGBoost\nX_2 = family_sales.drop('sales', axis=1).stack()  # onpromotion feature\n\n# Label encoding for 'family'\nle = LabelEncoder()  # from sklearn.preprocessing\nX_2 = X_2.reset_index('family')\nX_2['family'] = le.fit_transform(X_2['family'])\n\n# Label encoding for seasonality\nX_2[\"day\"] = X_2.index.day  # values are day of the month","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:43:14.488306Z","iopub.execute_input":"2022-07-24T10:43:14.488749Z","iopub.status.idle":"2022-07-24T10:43:14.515262Z","shell.execute_reply.started":"2022-07-24T10:43:14.488713Z","shell.execute_reply":"2022-07-24T10:43:14.514159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # YOUR CODE HERE: Create LinearRegression + XGBRegressor hybrid with BoostedHybrid\n# model = BoostedHybrid(LinearRegression(),XGBRegressor())\n\n# # YOUR CODE HERE: Fit and predict\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:44:22.867500Z","iopub.execute_input":"2022-07-24T10:44:22.867926Z","iopub.status.idle":"2022-07-24T10:44:24.643601Z","shell.execute_reply.started":"2022-07-24T10:44:22.867892Z","shell.execute_reply":"2022-07-24T10:44:24.642662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyearth import Earth\nfrom sklearn.linear_model import ElasticNet, Lasso, Ridge\n\n# Model 2\nfrom sklearn.ensemble import ExtraTreesRegressor, RandomForestRegressor\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.neural_network import MLPRegressor\n\n# Boosted Hybrid\n\n# YOUR CODE HERE: Try different combinations of the algorithms above\nmodel = BoostedHybrid(\n    model_1=Earth(),\n    model_2=xgbr(),\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:56:23.156451Z","iopub.execute_input":"2022-07-24T10:56:23.156873Z","iopub.status.idle":"2022-07-24T10:56:23.164493Z","shell.execute_reply.started":"2022-07-24T10:56:23.156839Z","shell.execute_reply":"2022-07-24T10:56:23.163023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train, y_valid = y[:\"2017-07-01\"], y[\"2017-07-02\":]\nX1_train, X1_valid = X_1[: \"2017-07-01\"], X_1[\"2017-07-02\" :]\nX2_train, X2_valid = X_2.loc[:\"2017-07-01\"], X_2.loc[\"2017-07-02\":]\n\n# Some of the algorithms above do best with certain kinds of\n# preprocessing on the features (like standardization), but this is\n# just a demo.\nmodel.fit(X1_train, X2_train, y_train)\ny_fit = model.predict(X1_train, X2_train).clip(0.0)\ny_pred = model.predict(X1_valid, X2_valid).clip(0.0)\n\nfamilies = y.columns[0:6]\naxs = y.loc(axis=1)[families].plot(\n    subplots=True, sharex=True, figsize=(11, 9), **plot_params, alpha=0.5,\n)\n_ = y_fit.loc(axis=1)[families].plot(subplots=True, sharex=True, color='C0', ax=axs)\n_ = y_pred.loc(axis=1)[families].plot(subplots=True, sharex=True, color='C3', ax=axs)\nfor ax, family in zip(axs, families):\n    ax.legend([])\n    ax.set_ylabel(family)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T10:56:23.242482Z","iopub.execute_input":"2022-07-24T10:56:23.242858Z","iopub.status.idle":"2022-07-24T10:56:29.663692Z","shell.execute_reply.started":"2022-07-24T10:56:23.242826Z","shell.execute_reply":"2022-07-24T10:56:29.662340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(\n    comp_dir / 'test.csv',\n    dtype={\n        'store_nbr': 'category',\n        'family': 'category',\n        'onpromotion': 'uint32',\n    },\n    parse_dates=['date'],\n    infer_datetime_format=True,\n)\ndf_test['date'] = df_test.date.dt.to_period('D')\ndf_test = df_test.set_index(['store_nbr', 'family', 'date']).sort_index()\n\nfamily_sales_test = (\n    df_test\n    .groupby(['family', 'date'])\n    .mean()\n    .unstack('family')\n)\n# # Create features for test set\n# X_test = dp.out_of_sample(steps=16)\n# X_test.index.name = 'date'\n# test_hol=[]\n# for i in X_test.index.dayofyear:\n#         if any(i==holidays.index.dayofyear):\n#             test_hol.append(True)\n#         else:\n#             test_hol.append(False)   \n# X_test['NewYear'] = (X_test.index.dayofyear == 1)\n# # X_test['Holiday'] = test_hol\n\n\n# y_submit = pd.DataFrame(model.predict(X_test), index=X_test.index, columns=y.columns)\n# y_submit = y_submit.stack(['store_nbr', 'family'])\n# y_submit = y_submit.join(df_test.id).reindex(columns=['id', 'sales'])\n# y_submit.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:10:20.734720Z","iopub.execute_input":"2022-07-24T11:10:20.735286Z","iopub.status.idle":"2022-07-24T11:10:20.812349Z","shell.execute_reply.started":"2022-07-24T11:10:20.735243Z","shell.execute_reply":"2022-07-24T11:10:20.811172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X2_train","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:14:27.771464Z","iopub.execute_input":"2022-07-24T11:14:27.772383Z","iopub.status.idle":"2022-07-24T11:14:27.786951Z","shell.execute_reply.started":"2022-07-24T11:14:27.772341Z","shell.execute_reply":"2022-07-24T11:14:27.785662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X2_test = family_sales_test.drop('id', axis=1).stack()  # onpromotion feature\n\n# Label encoding for 'family'\nle = LabelEncoder()  # from sklearn.preprocessing\nX2_test = X2_test.reset_index('family')\nX2_test['family'] = le.fit_transform(X2_test['family'])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:17:14.701915Z","iopub.execute_input":"2022-07-24T11:17:14.702425Z","iopub.status.idle":"2022-07-24T11:17:14.728030Z","shell.execute_reply.started":"2022-07-24T11:17:14.702387Z","shell.execute_reply":"2022-07-24T11:17:14.726809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X2_test[\"day\"] = X2_test.index.day \nX2_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:18:43.430820Z","iopub.execute_input":"2022-07-24T11:18:43.431736Z","iopub.status.idle":"2022-07-24T11:18:43.445503Z","shell.execute_reply.started":"2022-07-24T11:18:43.431689Z","shell.execute_reply":"2022-07-24T11:18:43.444579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dp = DeterministicProcess(index=family_sales_test.index, order=1)\nX1_test = dp.in_sample()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:19:53.975177Z","iopub.execute_input":"2022-07-24T11:19:53.976270Z","iopub.status.idle":"2022-07-24T11:19:53.985220Z","shell.execute_reply.started":"2022-07-24T11:19:53.976235Z","shell.execute_reply":"2022-07-24T11:19:53.984084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X2_test","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:20:14.724830Z","iopub.execute_input":"2022-07-24T11:20:14.727345Z","iopub.status.idle":"2022-07-24T11:20:14.741062Z","shell.execute_reply.started":"2022-07-24T11:20:14.727301Z","shell.execute_reply":"2022-07-24T11:20:14.739947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X1_test, X2_test).clip(0.0)\ny_pred.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:22:16.834581Z","iopub.execute_input":"2022-07-24T11:22:16.835234Z","iopub.status.idle":"2022-07-24T11:22:16.882086Z","shell.execute_reply.started":"2022-07-24T11:22:16.835183Z","shell.execute_reply":"2022-07-24T11:22:16.881133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_submit = pd.DataFrame(y_pred, index=family_sales_test.index, columns=y_pred.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:41:54.337431Z","iopub.execute_input":"2022-07-24T11:41:54.338114Z","iopub.status.idle":"2022-07-24T11:41:54.343798Z","shell.execute_reply.started":"2022-07-24T11:41:54.338064Z","shell.execute_reply":"2022-07-24T11:41:54.342990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_submit = y_submit.stack(['family'])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:41:55.628237Z","iopub.execute_input":"2022-07-24T11:41:55.628857Z","iopub.status.idle":"2022-07-24T11:41:55.634625Z","shell.execute_reply.started":"2022-07-24T11:41:55.628823Z","shell.execute_reply":"2022-07-24T11:41:55.633514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_submit=y_submit.to_frame(name='sales')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:41:56.706335Z","iopub.execute_input":"2022-07-24T11:41:56.707254Z","iopub.status.idle":"2022-07-24T11:41:56.712489Z","shell.execute_reply.started":"2022-07-24T11:41:56.707204Z","shell.execute_reply":"2022-07-24T11:41:56.711431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_submit.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:41:57.822356Z","iopub.execute_input":"2022-07-24T11:41:57.823016Z","iopub.status.idle":"2022-07-24T11:41:57.835618Z","shell.execute_reply.started":"2022-07-24T11:41:57.822975Z","shell.execute_reply":"2022-07-24T11:41:57.834207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_submit = y_submit.join(df_test.id).reindex(columns=['id', 'sales'])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:42:02.697965Z","iopub.execute_input":"2022-07-24T11:42:02.698872Z","iopub.status.idle":"2022-07-24T11:42:03.615354Z","shell.execute_reply.started":"2022-07-24T11:42:02.698829Z","shell.execute_reply":"2022-07-24T11:42:03.614201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_submit.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:43:46.884765Z","iopub.execute_input":"2022-07-24T11:43:46.885941Z","iopub.status.idle":"2022-07-24T11:43:47.731633Z","shell.execute_reply.started":"2022-07-24T11:43:46.885898Z","shell.execute_reply":"2022-07-24T11:43:47.730803Z"},"trusted":true},"execution_count":null,"outputs":[]}]}