{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-06T22:57:08.208660Z","iopub.execute_input":"2022-08-06T22:57:08.210052Z","iopub.status.idle":"2022-08-06T22:57:08.245323Z","shell.execute_reply.started":"2022-08-06T22:57:08.209889Z","shell.execute_reply":"2022-08-06T22:57:08.244182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nimport seaborn as sns\nsns.set()\nfrom itertools import cycle\n\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler\n\n# from fbprophet import Prophet\n# from fbprophet.plot import plot_plotly\nimport plotly.offline as py\npy.init_notebook_mode()\n\n\nimport time\nfrom tqdm import tqdm_notebook as tqdm\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n        \nplt.style.use('bmh')\ncolor_pal = plt.rcParams['axes.prop_cycle'].by_key()['color']\ncolor_cycle = cycle(plt.rcParams['axes.prop_cycle'].by_key()['color'])","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:27:23.271181Z","iopub.execute_input":"2022-08-09T07:27:23.271676Z","iopub.status.idle":"2022-08-09T07:27:24.148342Z","shell.execute_reply.started":"2022-08-09T07:27:23.271570Z","shell.execute_reply":"2022-08-09T07:27:24.146685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load data\ntrain = pd.read_csv(\"/kaggle/input/m5-forecasting-accuracy/sales_train_validation.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:27:24.150185Z","iopub.execute_input":"2022-08-09T07:27:24.150512Z","iopub.status.idle":"2022-08-09T07:27:30.800283Z","shell.execute_reply.started":"2022-08-09T07:27:24.150481Z","shell.execute_reply":"2022-08-09T07:27:30.799199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:27:30.801657Z","iopub.execute_input":"2022-08-09T07:27:30.802007Z","iopub.status.idle":"2022-08-09T07:27:30.809865Z","shell.execute_reply.started":"2022-08-09T07:27:30.801975Z","shell.execute_reply":"2022-08-09T07:27:30.808712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load Calendar information\ncalendar = pd.read_csv(\"/kaggle/input/m5-forecasting-accuracy/calendar.csv\")\ncalendar.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:27:30.812574Z","iopub.execute_input":"2022-08-09T07:27:30.813668Z","iopub.status.idle":"2022-08-09T07:27:30.846025Z","shell.execute_reply.started":"2022-08-09T07:27:30.813604Z","shell.execute_reply":"2022-08-09T07:27:30.844925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sell prices information\nsell_prices = pd.read_csv(\"/kaggle/input/m5-forecasting-accuracy/sell_prices.csv\")\nsell_prices.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:27:30.849366Z","iopub.execute_input":"2022-08-09T07:27:30.850176Z","iopub.status.idle":"2022-08-09T07:27:35.501317Z","shell.execute_reply.started":"2022-08-09T07:27:30.850140Z","shell.execute_reply":"2022-08-09T07:27:35.500033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Explore how it's sales look across the training data.\n\n* We have historic sales data in the `sales_train_validation dataset` where rows exist in this dataset for days d_1, d_2, …, d_i, … d_1941: The number of units sold at day i, starting from 2011-01-29. We are given the department, category, state, and store id of the item.","metadata":{}},{"cell_type":"code","source":"d_cols = [c for c in train.columns if 'd_' in c] # sales data columns\n#train[d_cols]\n# Below we are chaining the following steps in pandas:\n# 1. Select the item.\n# 2. Set the id as the index, Keep only sales data columns\n# 3. Transform so it's a column\n# 4. Plot the data\ntrain.loc[train['id'] == 'FOODS_3_090_CA_3_validation'] \\\n    .set_index('id')[d_cols] \\\n    .T \\\n    .plot(figsize=(15, 5),\n          title='FOODS_3_090_CA_3 sales by \"d\" number',\n          color=next(color_cycle))\nplt.legend('')\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:27:35.502968Z","iopub.execute_input":"2022-08-09T07:27:35.503438Z","iopub.status.idle":"2022-08-09T07:27:35.852549Z","shell.execute_reply.started":"2022-08-09T07:27:35.503389Z","shell.execute_reply":"2022-08-09T07:27:35.851232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Note there are days where it appears the item is unavailable and sales flatline","metadata":{}},{"cell_type":"code","source":"# we can use calendar data to look at dates of days\n# Merge calendar on our items' data\nexample = train.loc[train['id'] == 'FOODS_3_090_CA_3_validation'][d_cols].T\nexample = example.rename(columns={8412:'FOODS_3_090_CA_3'}) # Name it correctly\nexample = example.reset_index().rename(columns={'index': 'd'}) # make the index \"d\"\nexample = example.merge(calendar, how='left', validate='1:1')\nexample.set_index('date')['FOODS_3_090_CA_3'] \\\n    .plot(figsize=(15, 5),\n          color=next(color_cycle),\n          title='FOODS_3_090_CA_3 sales by actual sale dates')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:27:35.854493Z","iopub.execute_input":"2022-08-09T07:27:35.855021Z","iopub.status.idle":"2022-08-09T07:27:36.156115Z","shell.execute_reply.started":"2022-08-09T07:27:35.854973Z","shell.execute_reply":"2022-08-09T07:27:36.155035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# From example we can find weekly and annual trends\nfig, (ax1, ax2, ax3) = plt.subplots(1, 3, figsize=(15, 3))\n#example.groupby('wday').mean()['FOODS_3_090_CA_3']\nexample.groupby('wday').mean()['FOODS_3_090_CA_3'] \\\n        .plot(kind='line',\n              title='average sale: day of week',\n              lw=5,\n              color=color_pal[0],\n              ax=ax1)\nexample.groupby('month').mean()['FOODS_3_090_CA_3'] \\\n        .plot(kind='line',\n              title='average sale: month',\n              lw=5,\n              color=color_pal[4],\n\n              ax=ax2)\nexample.groupby('year').mean()['FOODS_3_090_CA_3'] \\\n        .plot(kind='line',\n              lw=5,\n              title='average sale: year',\n              color=color_pal[2],\n\n              ax=ax3)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:27:36.157744Z","iopub.execute_input":"2022-08-09T07:27:36.158109Z","iopub.status.idle":"2022-08-09T07:27:36.711172Z","shell.execute_reply.started":"2022-08-09T07:27:36.158078Z","shell.execute_reply":"2022-08-09T07:27:36.710024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(15, 5))\nstores = []\nfor store, d in sell_prices.query('item_id == \"FOODS_3_090\"').groupby('store_id'):\n    d.plot(x='wm_yr_wk',\n          y='sell_price',\n          style='.',\n          color=next(color_cycle),\n          figsize=(15, 5),\n          title='FOODS_3_090 sale price over time',\n         ax=ax,\n          legend=store)\n    stores.append(store)\n    plt.legend()\nplt.legend(stores)\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:27:36.714307Z","iopub.execute_input":"2022-08-09T07:27:36.714685Z","iopub.status.idle":"2022-08-09T07:27:37.397033Z","shell.execute_reply.started":"2022-08-09T07:27:36.714623Z","shell.execute_reply":"2022-08-09T07:27:37.395723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Note:\n\n* It looks to me like the price of this item is growing.\n* Different stores have different selling prices.","metadata":{}},{"cell_type":"markdown","source":"### Lets put it all together to plot 20 different items and their sales","metadata":{}},{"cell_type":"code","source":"twenty_examples = train.sample(20) \\\n        .set_index('id')[d_cols] \\\n    .T \\\n    .merge(calendar.set_index('d')['date'],\n           left_index=True,\n           right_index=True,\n            validate='1:1') \\\n    .set_index('date')\nfig, axs = plt.subplots(10, 2, figsize=(15, 20))\naxs = axs.flatten()\nax_idx = 0\nfor item in twenty_examples.columns:\n    twenty_examples[item].plot(title=item,\n                              color=next(color_cycle),\n                              ax=axs[ax_idx])\n    ax_idx += 1\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:27:37.398847Z","iopub.execute_input":"2022-08-09T07:27:37.399301Z","iopub.status.idle":"2022-08-09T07:27:40.774482Z","shell.execute_reply.started":"2022-08-09T07:27:37.399255Z","shell.execute_reply":"2022-08-09T07:27:40.773184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Note that :\n\n* It is common to see an item unavailable for a period of time.\n* Some items only sell 1 or less in a day, making it very hard to predict.\n* Other items show spikes in their demand ","metadata":{}},{"cell_type":"code","source":"train.groupby('cat_id').count()['id'] \\\n    .sort_values() \\\n    .plot(kind='barh', figsize=(15, 5), title='Count of Items by Category')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:27:40.776255Z","iopub.execute_input":"2022-08-09T07:27:40.776770Z","iopub.status.idle":"2022-08-09T07:27:41.529870Z","shell.execute_reply.started":"2022-08-09T07:27:40.776722Z","shell.execute_reply":"2022-08-09T07:27:41.528799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"past_sales = train.set_index('id')[d_cols] \\\n    .T \\\n    .merge(calendar.set_index('d')['date'],\n           left_index=True,\n           right_index=True,\n            validate='1:1')\\\n    .set_index('date')","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:27:41.531711Z","iopub.execute_input":"2022-08-09T07:27:41.532298Z","iopub.status.idle":"2022-08-09T07:27:42.489408Z","shell.execute_reply.started":"2022-08-09T07:27:41.532266Z","shell.execute_reply":"2022-08-09T07:27:42.488001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  Plot the total demand over time for each categorical","metadata":{}},{"cell_type":"code","source":"# for i in train['cat_id'].unique():\n#     items_col = [c for c in past_sales.columns if i in c]\n#     print(past_sales[items_col].columns)\n    \nfor i in train['cat_id'].unique():\n    items_col = [c for c in past_sales.columns if i in c]\n    past_sales[items_col] \\\n        .sum(axis=1) \\\n        .plot(figsize=(15, 5),\n              alpha=0.8,\n              title='Total Sales by Item Type')\nplt.legend(train['cat_id'].unique())\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:27:42.491240Z","iopub.execute_input":"2022-08-09T07:27:42.492164Z","iopub.status.idle":"2022-08-09T07:27:43.123560Z","shell.execute_reply.started":"2022-08-09T07:27:42.492126Z","shell.execute_reply":"2022-08-09T07:27:43.122376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We can see the some items come into supply that previously didn't exist. Similarly some items stop being sold completely.","metadata":{}},{"cell_type":"code","source":"past_sales_clipped = past_sales.clip(0, 1) # 0 -> not selling, 1 -> selling\nfor i in train['cat_id'].unique():\n    items_col = [c for c in past_sales.columns if i in c]\n    (past_sales_clipped[items_col] \\\n        .mean(axis=1) * 100) \\\n        .plot(figsize=(15, 5),\n              alpha=0.8,\n              title='Inventory Sale Percentage by Date',\n              style='.')\nplt.ylabel('% of Inventory with at least 1 sale')\nplt.legend(train['cat_id'].unique())\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:27:43.124928Z","iopub.execute_input":"2022-08-09T07:27:43.125240Z","iopub.status.idle":"2022-08-09T07:27:46.614410Z","shell.execute_reply.started":"2022-08-09T07:27:43.125211Z","shell.execute_reply":"2022-08-09T07:27:46.611657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sell_prices['Category'] = sell_prices['item_id'].str.split('_', expand=True)[0]\nfig, axs = plt.subplots(1, 3, figsize=(15, 4))\ni = 0\nfor cat, d in sell_prices.groupby('Category'):\n    ax = d['sell_price'].apply(np.log1p) \\\n        .plot(kind='hist',\n                         bins=20,\n                         title=f'Distribution of {cat} prices',\n                         ax=axs[i],\n                                         color=next(color_cycle))\n    ax.set_xlabel('Log(price)')\n    i += 1\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:27:46.618343Z","iopub.execute_input":"2022-08-09T07:27:46.618748Z","iopub.status.idle":"2022-08-09T07:28:07.181018Z","shell.execute_reply.started":"2022-08-09T07:27:46.618713Z","shell.execute_reply":"2022-08-09T07:28:07.179703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#10 unique stores.\n# train.groupby('store_id').count()['id'] \\\n#     .sort_values() \\\n#     .plot(kind='barh', figsize=(15, 5), title='Count of Items by Stores')\n# plt.show()\nstore_list = sell_prices['store_id'].unique()\n# loop on each store and compute sumtion of sales all products and compute average for each 90 days\nfor s in store_list:\n    store_items = [c for c in past_sales.columns if s in c]\n    past_sales[store_items] \\\n        .sum(axis=1) \\\n        .rolling(90).mean() \\\n        .plot(figsize=(15, 5),\n              alpha=0.8,\n              title='Rolling 90 Day Average Total Sales (10 stores)')\nplt.legend(store_list)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:28:07.182584Z","iopub.execute_input":"2022-08-09T07:28:07.183047Z","iopub.status.idle":"2022-08-09T07:28:08.484399Z","shell.execute_reply.started":"2022-08-09T07:28:07.183005Z","shell.execute_reply":"2022-08-09T07:28:08.483109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Note that:\n\n* `CA_3` have highest total salse.\n* some stores are more steady than others.\n* `CA_2` seems to have a big change occur in 2015.\n* `WI_2` seems to have a big change occur in 2012.\n* some stores have abrupt changes in their demand\n\n\n","metadata":{}},{"cell_type":"code","source":"# plot a rolling 7 day(weekly) total demand count by store. \nfig, axes = plt.subplots(5, 2, figsize=(15, 10), sharex=True)\naxes = axes.flatten()\nax_idx = 0\nfor s in store_list:\n    store_items = [c for c in past_sales.columns if s in c]\n    past_sales[store_items] \\\n        .sum(axis=1) \\\n        .rolling(7).mean() \\\n        .plot(alpha=1,\n              ax=axes[ax_idx],\n              title=s,\n              lw=3,\n              color=next(color_cycle))\n    ax_idx += 1\n# plt.legend(store_list)\nplt.suptitle('Weekly Sale Trends by Store ID')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:28:08.488160Z","iopub.execute_input":"2022-08-09T07:28:08.488508Z","iopub.status.idle":"2022-08-09T07:28:10.335046Z","shell.execute_reply.started":"2022-08-09T07:28:08.488477Z","shell.execute_reply":"2022-08-09T07:28:10.333743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('The lowest sale date was:', past_sales.sum(axis=1).sort_values().index[0],\n     'with', past_sales.sum(axis=1).sort_values().values[0], 'sales')\nprint('The highest sale date was:', past_sales.sum(axis=1).sort_values(ascending=False).index[0],\n     'with', past_sales.sum(axis=1).sort_values(ascending=False).values[0], 'sales')","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:28:10.336557Z","iopub.execute_input":"2022-08-09T07:28:10.337013Z","iopub.status.idle":"2022-08-09T07:28:10.818120Z","shell.execute_reply.started":"2022-08-09T07:28:10.336971Z","shell.execute_reply":"2022-08-09T07:28:10.816681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Sales Heatmap Calendar\nfrom matplotlib.patches import Polygon\nfrom datetime import datetime\nfrom dateutil.relativedelta import relativedelta\n\ndef calmap(ax, year, data):\n    ax.tick_params('x', length=0, labelsize=\"medium\", which='major')\n    ax.tick_params('y', length=0, labelsize=\"x-small\", which='major')\n\n    # Month borders\n    xticks, labels = [], []\n    start = datetime(year,1,1).weekday()\n    for month in range(1,13):\n        first = datetime(year, month, 1)\n        last = first + relativedelta(months=1, days=-1)\n\n        y0 = first.weekday()\n        y1 = last.weekday()\n        x0 = (int(first.strftime(\"%j\"))+start-1)//7\n        x1 = (int(last.strftime(\"%j\"))+start-1)//7\n\n        P = [ (x0,   y0), (x0,    7),  (x1,   7),\n              (x1,   y1+1), (x1+1,  y1+1), (x1+1, 0),\n              (x0+1,  0), (x0+1,  y0) ]\n        xticks.append(x0 +(x1-x0+1)/2)\n        labels.append(first.strftime(\"%b\"))\n        poly = Polygon(P, edgecolor=\"black\", facecolor=\"None\",\n                       linewidth=1, zorder=20, clip_on=False)\n        ax.add_artist(poly)\n    \n    ax.set_xticks(xticks)\n    ax.set_xticklabels(labels)\n    ax.set_yticks(0.5 + np.arange(7))\n    ax.set_yticklabels([\"Mon\", \"Tue\", \"Wed\", \"Thu\", \"Fri\", \"Sat\", \"Sun\"])\n    ax.set_title(\"{}\".format(year), weight=\"semibold\")\n    \n    # Clearing first and last day from the data\n    valid = datetime(year, 1, 1).weekday()\n    data[:valid,0] = np.nan\n    valid = datetime(year, 12, 31).weekday()\n    # data[:,x1+1:] = np.nan\n    data[valid+1:,x1] = np.nan\n\n    # Showing data\n    ax.imshow(data, extent=[0,53,0,7], zorder=10, vmin=-1, vmax=1,\n              cmap=\"RdYlBu_r\", origin=\"lower\", alpha=.75)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:28:10.819582Z","iopub.execute_input":"2022-08-09T07:28:10.819990Z","iopub.status.idle":"2022-08-09T07:28:10.834130Z","shell.execute_reply.started":"2022-08-09T07:28:10.819955Z","shell.execute_reply":"2022-08-09T07:28:10.833284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nsscale = StandardScaler()\npast_sales.index = pd.to_datetime(past_sales.index)\nfor i in train['cat_id'].unique():\n    fig, axes = plt.subplots(3, 1, figsize=(20, 8))\n    items_col = [c for c in past_sales.columns if i in c]\n    \n    sales2013 = past_sales.loc[past_sales.index.isin(pd.date_range('31-Dec-2012',\n                                                                   periods=371))][items_col].mean(axis=1)\n    vals = np.hstack(sscale.fit_transform(sales2013.values.reshape(-1, 1)))\n    calmap(axes[0], 2013, vals.reshape(53,7).T)\n    sales2014 = past_sales.loc[past_sales.index.isin(pd.date_range('30-Dec-2013',\n                                                                   periods=371))][items_col].mean(axis=1)\n    vals = np.hstack(sscale.fit_transform(sales2014.values.reshape(-1, 1)))\n    calmap(axes[1], 2014, vals.reshape(53,7).T)\n    sales2015 = past_sales.loc[past_sales.index.isin(pd.date_range('29-Dec-2014',\n                                                                   periods=371))][items_col].mean(axis=1)\n    vals = np.hstack(sscale.fit_transform(sales2015.values.reshape(-1, 1)))\n    calmap(axes[2], 2015, vals.reshape(53,7).T)\n    \n    plt.suptitle(i, fontsize=30, x=0.4, y=1.01)\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:28:10.835436Z","iopub.execute_input":"2022-08-09T07:28:10.835967Z","iopub.status.idle":"2022-08-09T07:28:14.159511Z","shell.execute_reply.started":"2022-08-09T07:28:10.835933Z","shell.execute_reply":"2022-08-09T07:28:14.158693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#past_sales.index = pd.to_datetime(past_sales.index)\n#past_sales.index.year \n#past_sales.loc[past_sales.index.isin(pd.date_range('31-Dec-2011',periods=371))]\n#past_sales.loc[past_sales.index.year == 2013]\n#past_sales.loc[past_sales.index.isin(pd.date_range('31-Dec-2012',periods=371))]","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:28:14.160567Z","iopub.execute_input":"2022-08-09T07:28:14.161266Z","iopub.status.idle":"2022-08-09T07:28:14.166000Z","shell.execute_reply.started":"2022-08-09T07:28:14.161234Z","shell.execute_reply":"2022-08-09T07:28:14.165082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Note that: \n\n* Food tends to have lower number of purchases as the month goes on. \n* Household and Hobby items sell much less in January - after the Holiday season is over.\n* Cleary weekends are more popular shopping days regardless of the item category.\n* It appears that walmarts are closed on Chirstmas day. \n* The highest demand day of all the data was on Sunday March 6th, 2016.","metadata":{}},{"cell_type":"code","source":"thirty_day_avg_map =train.set_index('id')[d_cols[-30:]].mean(axis=1).to_dict()\nsubmission = pd.read_csv('/kaggle/input/m5-forecasting-accuracy/sample_submission.csv')\nsubmission.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:40:42.644628Z","iopub.execute_input":"2022-08-09T07:40:42.645031Z","iopub.status.idle":"2022-08-09T07:40:43.169101Z","shell.execute_reply.started":"2022-08-09T07:40:42.645000Z","shell.execute_reply":"2022-08-09T07:40:43.167732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submission['id'].map(thirty_day_avg_map)\nfcols = [f for f in submission.columns if 'F' in f]\nfor f in fcols:\n    submission[f] = submission['id'].map(thirty_day_avg_map).fillna(0)\n    \nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:44:00.361803Z","iopub.execute_input":"2022-08-09T07:44:00.362195Z","iopub.status.idle":"2022-08-09T07:44:02.359243Z","shell.execute_reply.started":"2022-08-09T07:44:00.362163Z","shell.execute_reply":"2022-08-09T07:44:02.357846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['store_id'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T22:33:53.351609Z","iopub.execute_input":"2022-08-08T22:33:53.351971Z","iopub.status.idle":"2022-08-08T22:33:53.364399Z","shell.execute_reply.started":"2022-08-08T22:33:53.351925Z","shell.execute_reply":"2022-08-08T22:33:53.363567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* ### Reading the competiton guideline we can find out that we have to deal with grouped time series of unit sales data. \n\nthis mean that A measure of the total amount of revenue a product generates divided by the total number of units of that product that were sold in a given time period in case Unit sales of all products. ","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}