{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%pip install lightgbm","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:13.512944Z","iopub.execute_input":"2022-07-19T03:24:13.513318Z","iopub.status.idle":"2022-07-19T03:24:22.777161Z","shell.execute_reply.started":"2022-07-19T03:24:13.513290Z","shell.execute_reply":"2022-07-19T03:24:22.775612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import imp\nimport pandas as pd\nfrom matplotlib import pyplot as plt\nimport numpy as np\nimport lightgbm as lgb\nfrom statsmodels.graphics.tsaplots import plot_acf, plot_pacf\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:22.779601Z","iopub.execute_input":"2022-07-19T03:24:22.780132Z","iopub.status.idle":"2022-07-19T03:24:22.786276Z","shell.execute_reply.started":"2022-07-19T03:24:22.780086Z","shell.execute_reply":"2022-07-19T03:24:22.784850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/demand-forecasting-kernels-only/train.csv', parse_dates=['date'])\ntest_df = pd.read_csv('../input/demand-forecasting-kernels-only/test.csv', parse_dates=['date'])\n# sample_sub = pd.read_csv('./input/demand-forecasting-kernels-only/sample_submission.csv')\ndf = pd.concat([train_df, test_df], sort=False)\n\nprint(train_df.shape, test_df.shape, df.shape, \"\\n\")\ntrain_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:22.787751Z","iopub.execute_input":"2022-07-19T03:24:22.788733Z","iopub.status.idle":"2022-07-19T03:24:23.244424Z","shell.execute_reply.started":"2022-07-19T03:24:22.788560Z","shell.execute_reply":"2022-07-19T03:24:23.243176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"def check_df(dataframe, head=5):\n    print(\"##################### Shape #####################\")\n    print(dataframe.shape)\n    print(\"##################### Types #####################\")\n    print(dataframe.dtypes)\n    print(\"##################### Head #####################\")\n    print(dataframe.head(head))\n    print(\"##################### Tail #####################\")\n    print(dataframe.tail(head))\n    print(\"##################### NA #####################\")\n    print(dataframe.isnull().sum())\n    print(\"##################### Quantiles #####################\")\n    print(dataframe.quantile([0, 0.05, 0.50, 0.95, 0.99, 1]).T)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:23.246675Z","iopub.execute_input":"2022-07-19T03:24:23.246985Z","iopub.status.idle":"2022-07-19T03:24:23.253172Z","shell.execute_reply.started":"2022-07-19T03:24:23.246959Z","shell.execute_reply":"2022-07-19T03:24:23.252255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_df(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:23.254309Z","iopub.execute_input":"2022-07-19T03:24:23.254849Z","iopub.status.idle":"2022-07-19T03:24:23.368856Z","shell.execute_reply.started":"2022-07-19T03:24:23.254817Z","shell.execute_reply":"2022-07-19T03:24:23.367404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['date'].min(), train_df['date'].max(),test_df['date'].min(), test_df['date'].max()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:23.370491Z","iopub.execute_input":"2022-07-19T03:24:23.371242Z","iopub.status.idle":"2022-07-19T03:24:23.389328Z","shell.execute_reply.started":"2022-07-19T03:24:23.371197Z","shell.execute_reply":"2022-07-19T03:24:23.388259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_df(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:23.391021Z","iopub.execute_input":"2022-07-19T03:24:23.391832Z","iopub.status.idle":"2022-07-19T03:24:23.487223Z","shell.execute_reply.started":"2022-07-19T03:24:23.391788Z","shell.execute_reply":"2022-07-19T03:24:23.485866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:23.488577Z","iopub.execute_input":"2022-07-19T03:24:23.488976Z","iopub.status.idle":"2022-07-19T03:24:23.515535Z","shell.execute_reply.started":"2022-07-19T03:24:23.488945Z","shell.execute_reply":"2022-07-19T03:24:23.514792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:23.517042Z","iopub.execute_input":"2022-07-19T03:24:23.517390Z","iopub.status.idle":"2022-07-19T03:24:23.688388Z","shell.execute_reply.started":"2022-07-19T03:24:23.517361Z","shell.execute_reply":"2022-07-19T03:24:23.687205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[['item']].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:23.691838Z","iopub.execute_input":"2022-07-19T03:24:23.692177Z","iopub.status.idle":"2022-07-19T03:24:23.711041Z","shell.execute_reply.started":"2022-07-19T03:24:23.692142Z","shell.execute_reply":"2022-07-19T03:24:23.709956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[['store']].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:23.712418Z","iopub.execute_input":"2022-07-19T03:24:23.713036Z","iopub.status.idle":"2022-07-19T03:24:23.732720Z","shell.execute_reply.started":"2022-07-19T03:24:23.713002Z","shell.execute_reply":"2022-07-19T03:24:23.731602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby('store').agg({'item': 'nunique'})","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:23.734479Z","iopub.execute_input":"2022-07-19T03:24:23.735274Z","iopub.status.idle":"2022-07-19T03:24:23.792439Z","shell.execute_reply.started":"2022-07-19T03:24:23.735226Z","shell.execute_reply":"2022-07-19T03:24:23.791289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby(['store', 'item']).agg({'sales': ['sum','mean', 'std', 'median'],})","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:23.794072Z","iopub.execute_input":"2022-07-19T03:24:23.794397Z","iopub.status.idle":"2022-07-19T03:24:23.916374Z","shell.execute_reply.started":"2022-07-19T03:24:23.794369Z","shell.execute_reply":"2022-07-19T03:24:23.915035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = df['sales'].dropna().values","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:23.917996Z","iopub.execute_input":"2022-07-19T03:24:23.918588Z","iopub.status.idle":"2022-07-19T03:24:23.931801Z","shell.execute_reply.started":"2022-07-19T03:24:23.918540Z","shell.execute_reply":"2022-07-19T03:24:23.930609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 5, figsize=(20, 10))\nfor i in range(1,11):\n    if i < 6:\n        train_df[train_df.store == i].sales.hist(ax=axes[0, i-1])\n        axes[0,i-1].set_title(\"Store \" + str(i), fontsize = 15)\n        \n    else:\n        train_df[train_df.store == i].sales.hist(ax=axes[1, i - 6])\n        axes[1,i-6].set_title(\"Store \" + str(i), fontsize = 15)\nplt.tight_layout(pad=4.5)\nplt.suptitle(\"Histogram: Sales\");","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:23.933490Z","iopub.execute_input":"2022-07-19T03:24:23.934058Z","iopub.status.idle":"2022-07-19T03:24:25.469305Z","shell.execute_reply.started":"2022-07-19T03:24:23.934025Z","shell.execute_reply":"2022-07-19T03:24:25.468029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"store = 1\nsub = train_df[train_df.store == store].set_index(\"date\")\n\nfig, axes = plt.subplots(10, 5, figsize=(20, 35))\nfor i in range(1,51):\n    if i < 6:\n        sub[sub.item == i].sales.plot(ax=axes[0, i-1], legend=True, label = \"Item \"+str(i)+\" Sales\")\n    if i >= 6 and i<11:\n        sub[sub.item == i].sales.plot(ax=axes[1, i - 6], legend=True, label = \"Item \"+str(i)+\" Sales\")\n    if i >= 11 and i<16:\n        sub[sub.item == i].sales.plot(ax=axes[2, i - 11], legend=True, label = \"Item \"+str(i)+\" Sales\")    \n    if i >= 16 and i<21:\n        sub[sub.item == i].sales.plot(ax=axes[3, i - 16], legend=True, label = \"Item \"+str(i)+\" Sales\")    \n    if i >= 21 and i<26:\n        sub[sub.item == i].sales.plot(ax=axes[4, i - 21], legend=True, label = \"Item \"+str(i)+\" Sales\")  \n    if i >= 26 and i<31:\n        sub[sub.item == i].sales.plot(ax=axes[5, i - 26], legend=True, label = \"Item \"+str(i)+\" Sales\")    \n    if i >= 31 and i<36:\n        sub[sub.item == i].sales.plot(ax=axes[6, i - 31], legend=True, label = \"Item \"+str(i)+\" Sales\")    \n    if i >= 36 and i<41:\n        sub[sub.item == i].sales.plot(ax=axes[7, i - 36], legend=True, label = \"Item \"+str(i)+\" Sales\")    \n    if i >= 41 and i<46:\n        sub[sub.item == i].sales.plot(ax=axes[8, i - 41], legend=True, label = \"Item \"+str(i)+\" Sales\") \n    if i >= 46 and i<51:\n        sub[sub.item == i].sales.plot(ax=axes[9, i - 46], legend=True, label = \"Item \"+str(i)+\" Sales\") \nplt.tight_layout(pad=4.5)\nplt.suptitle(\"Store 1 Item Status\");","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:25.470816Z","iopub.execute_input":"2022-07-19T03:24:25.471247Z","iopub.status.idle":"2022-07-19T03:24:35.317910Z","shell.execute_reply.started":"2022-07-19T03:24:25.471203Z","shell.execute_reply":"2022-07-19T03:24:35.316513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"store = 1\nsub = train_df[train_df.store == store].set_index(\"date\")\n\nfig, axes = plt.subplots(10, 5, figsize=(20, 35))\nfor i in range(1,51):\n    if i < 6:\n        plot_acf(sub[sub.item == i].sales, ax=axes[0, i-1])\n    #     sub[sub.item == i].sales.plot(ax=axes[0, i-1], legend=True, label = \"Item \"+str(i)+\" Sales\")\n    if i >= 6 and i<11:\n        plot_acf(sub[sub.item == i].sales, ax=axes[1, i-6])\n\n\n    #     sub[sub.item == i].sales.plot(ax=axes[1, i - 6], legend=True, label = \"Item \"+str(i)+\" Sales\")\n    if i >= 11 and i<16:\n        plot_acf(sub[sub.item == i].sales, ax=axes[2, i-11])\n\n\n        \n    #     sub[sub.item == i].sales.plot(ax=axes[2, i - 11], legend=True, label = \"Item \"+str(i)+\" Sales\")    \n    if i >= 16 and i<21:\n        plot_acf(sub[sub.item == i].sales, ax=axes[3, i-16])\n\n\n    #     sub[sub.item == i].sales.plot(ax=axes[3, i - 16], legend=True, label = \"Item \"+str(i)+\" Sales\")    \n    if i >= 21 and i<26:\n        plot_acf(sub[sub.item == i].sales, ax=axes[4, i-21])\n\n\n    #     sub[sub.item == i].sales.plot(ax=axes[4, i - 21], legend=True, label = \"Item \"+str(i)+\" Sales\")  \n    if i >= 26 and i<31:\n        plot_acf(sub[sub.item == i].sales, ax=axes[5, i-26])\n\n\n    #     sub[sub.item == i].sales.plot(ax=axes[5, i - 26], legend=True, label = \"Item \"+str(i)+\" Sales\")    \n    if i >= 31 and i<36:\n        plot_acf(sub[sub.item == i].sales, ax=axes[6, i-31])\n\n\n    #     sub[sub.item == i].sales.plot(ax=axes[6, i - 31], legend=True, label = \"Item \"+str(i)+\" Sales\")    \n    if i >= 36 and i<41:\n        plot_acf(sub[sub.item == i].sales, ax=axes[7, i-36])\n\n\n    #     sub[sub.item == i].sales.plot(ax=axes[7, i - 36], legend=True, label = \"Item \"+str(i)+\" Sales\")    \n    if i >= 41 and i<46:\n        plot_acf(sub[sub.item == i].sales, ax=axes[8, i-41])\n\n\n    #     sub[sub.item == i].sales.plot(ax=axes[8, i - 41], legend=True, label = \"Item \"+str(i)+\" Sales\") \n    if i >= 46 and i<51:\n        plot_acf(sub[sub.item == i].sales, ax=axes[9, i-46])\n\n\n    #     sub[sub.item == i].sales.plot(ax=axes[9, i - 46], legend=True, label = \"Item \"+str(i)+\" Sales\") \nplt.tight_layout(pad=4.5)\n\n# Presence of seasonlaity in data","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:35.319600Z","iopub.execute_input":"2022-07-19T03:24:35.320245Z","iopub.status.idle":"2022-07-19T03:24:41.669937Z","shell.execute_reply.started":"2022-07-19T03:24:35.320206Z","shell.execute_reply":"2022-07-19T03:24:41.668553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"store = 1\nsub = train_df[train_df.store == store].set_index(\"date\")\n\nfig, axes = plt.subplots(10, 5, figsize=(20, 35))\nfor i in range(1,51):\n    if i < 6:\n        plot_pacf(sub[sub.item == i].sales, ax=axes[0, i-1])\n    #     sub[sub.item == i].sales.plot(ax=axes[0, i-1], legend=True, label = \"Item \"+str(i)+\" Sales\")\n    if i >= 6 and i<11:\n        plot_pacf(sub[sub.item == i].sales, ax=axes[1, i-6])\n\n\n    #     sub[sub.item == i].sales.plot(ax=axes[1, i - 6], legend=True, label = \"Item \"+str(i)+\" Sales\")\n    if i >= 11 and i<16:\n        plot_pacf(sub[sub.item == i].sales, ax=axes[2, i-11])\n\n\n        \n    #     sub[sub.item == i].sales.plot(ax=axes[2, i - 11], legend=True, label = \"Item \"+str(i)+\" Sales\")    \n    if i >= 16 and i<21:\n        plot_pacf(sub[sub.item == i].sales, ax=axes[3, i-16])\n\n\n    #     sub[sub.item == i].sales.plot(ax=axes[3, i - 16], legend=True, label = \"Item \"+str(i)+\" Sales\")    \n    if i >= 21 and i<26:\n        plot_pacf(sub[sub.item == i].sales, ax=axes[4, i-21])\n\n\n    #     sub[sub.item == i].sales.plot(ax=axes[4, i - 21], legend=True, label = \"Item \"+str(i)+\" Sales\")  \n    if i >= 26 and i<31:\n        plot_pacf(sub[sub.item == i].sales, ax=axes[5, i-26])\n\n\n    #     sub[sub.item == i].sales.plot(ax=axes[5, i - 26], legend=True, label = \"Item \"+str(i)+\" Sales\")    \n    if i >= 31 and i<36:\n        plot_pacf(sub[sub.item == i].sales, ax=axes[6, i-31])\n\n\n    #     sub[sub.item == i].sales.plot(ax=axes[6, i - 31], legend=True, label = \"Item \"+str(i)+\" Sales\")    \n    if i >= 36 and i<41:\n        plot_pacf(sub[sub.item == i].sales, ax=axes[7, i-36])\n\n\n    #     sub[sub.item == i].sales.plot(ax=axes[7, i - 36], legend=True, label = \"Item \"+str(i)+\" Sales\")    \n    if i >= 41 and i<46:\n        plot_pacf(sub[sub.item == i].sales, ax=axes[8, i-41])\n\n\n    #     sub[sub.item == i].sales.plot(ax=axes[8, i - 41], legend=True, label = \"Item \"+str(i)+\" Sales\") \n    if i >= 46 and i<51:\n        plot_pacf(sub[sub.item == i].sales, ax=axes[9, i-46])\n\n\n    #     sub[sub.item == i].sales.plot(ax=axes[9, i - 46], legend=True, label = \"Item \"+str(i)+\" Sales\") \nplt.tight_layout(pad=4.5)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:41.671400Z","iopub.execute_input":"2022-07-19T03:24:41.671792Z","iopub.status.idle":"2022-07-19T03:24:49.443913Z","shell.execute_reply.started":"2022-07-19T03:24:41.671758Z","shell.execute_reply":"2022-07-19T03:24:49.442706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"# Date features\n\ndef create_date_features(df):\n    df['month'] = df.date.dt.month\n    df['day_of_month'] = df.date.dt.day\n    df['day_of_year'] = df.date.dt.dayofyear\n    df['week_of_year'] = df.date.dt.weekofyear\n    df['day_of_week'] = df.date.dt.dayofweek\n    df['year'] = df.date.dt.year\n    df['is_wknd'] = df.date.dt.weekday // 4  # Also included Friday as weekend\n    df['is_month_start'] = df.date.dt.is_month_start.astype(int)\n    df['is_month_end'] = df.date.dt.is_month_end.astype(int)\n    df['quarter'] = df.date.dt.quarter\n    df['is_quarter_start'] = df.date.dt.is_quarter_start.astype(int)\n    df['is_quarter_end'] = df.date.dt.is_quarter_end.astype(int)\n    df['is_christmas_week'] = (df.date.dt.weekofyear == 51).astype(int)\n    # 0: Winter - 1: Spring - 2: Summer - 3: Fall\n    df[\"season\"] = np.where(df.month.isin([12,1,2]), 0, 1)\n    df[\"season\"] = np.where(df.month.isin([6,7,8]), 2, df[\"season\"])\n    df[\"season\"] = np.where(df.month.isin([9, 10, 11]), 3, df[\"season\"])\n    return df\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:49.445508Z","iopub.execute_input":"2022-07-19T03:24:49.446230Z","iopub.status.idle":"2022-07-19T03:24:49.454701Z","shell.execute_reply.started":"2022-07-19T03:24:49.446196Z","shell.execute_reply":"2022-07-19T03:24:49.453700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def random_noise(dataframe):\n    return np.random.normal(scale=1.6, size=(len(dataframe),))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:49.455824Z","iopub.execute_input":"2022-07-19T03:24:49.456473Z","iopub.status.idle":"2022-07-19T03:24:49.472229Z","shell.execute_reply.started":"2022-07-19T03:24:49.456437Z","shell.execute_reply":"2022-07-19T03:24:49.471204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lag/Shift features\n# To capture seasonality\ndef lag_features(dataframe, lags):\n    for lag in lags:\n        dataframe['sales_lag_' + str(lag)] = dataframe.groupby(['store', 'item'])['sales'].transform(\n            lambda x: x.shift(lag)) +  random_noise(dataframe)\n    return dataframe","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:49.473668Z","iopub.execute_input":"2022-07-19T03:24:49.474527Z","iopub.status.idle":"2022-07-19T03:24:49.483865Z","shell.execute_reply.started":"2022-07-19T03:24:49.474494Z","shell.execute_reply":"2022-07-19T03:24:49.482818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Rolling Mean features\n# To capture trend\ndef roll_mean_features(dataframe, windows):\n    for window in windows:\n        dataframe['sales_roll_mean_' + str(window)] = dataframe.groupby(['store','item'])['sales'].\\\n                                                          transform(\n            lambda x : x.rolling(window=window, min_periods=10, win_type='triang').mean().shift(1))+random_noise(dataframe)\n    return dataframe","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:49.485261Z","iopub.execute_input":"2022-07-19T03:24:49.485942Z","iopub.status.idle":"2022-07-19T03:24:49.495319Z","shell.execute_reply.started":"2022-07-19T03:24:49.485899Z","shell.execute_reply":"2022-07-19T03:24:49.494199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Exponential Weighted Mean\n# Gives weight to recent sales\ndef ewm_features(dataframe, alphas, lags):\n    dataframe = dataframe.copy()\n    for alpha in alphas:\n        for lag in lags:\n            dataframe['sales_ewm_alpha' + str(alpha).replace('.','') + '_lag_' + str(lag)] = \\\n            dataframe.groupby(['store','item'])['sales'].transform(\n                lambda x: x.shift(lag).ewm(alpha=alpha).mean())\n    return dataframe\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:49.496361Z","iopub.execute_input":"2022-07-19T03:24:49.497078Z","iopub.status.idle":"2022-07-19T03:24:49.507347Z","shell.execute_reply.started":"2022-07-19T03:24:49.497047Z","shell.execute_reply":"2022-07-19T03:24:49.506205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:49.508501Z","iopub.execute_input":"2022-07-19T03:24:49.509138Z","iopub.status.idle":"2022-07-19T03:24:49.530416Z","shell.execute_reply.started":"2022-07-19T03:24:49.509101Z","shell.execute_reply":"2022-07-19T03:24:49.529571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = create_date_features(df)\nprint(df.shape)\ndf.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:49.531435Z","iopub.execute_input":"2022-07-19T03:24:49.532138Z","iopub.status.idle":"2022-07-19T03:24:51.150852Z","shell.execute_reply.started":"2022-07-19T03:24:49.532090Z","shell.execute_reply":"2022-07-19T03:24:51.149700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = roll_mean_features(df, [16, 23, 32, 62, 68, 98, 105, 126, 182, 211, 365, 546, 728])\nprint(df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:24:51.152168Z","iopub.execute_input":"2022-07-19T03:24:51.152483Z","iopub.status.idle":"2022-07-19T03:25:02.115271Z","shell.execute_reply.started":"2022-07-19T03:24:51.152457Z","shell.execute_reply":"2022-07-19T03:25:02.113771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = lag_features(df, [16, 23, 31, 38, 61, 68, 91, 98, 105, 112, 119, 126, 182, 211 ,364, 546, 728])\nprint(df.shape)\ndf.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:25:02.117746Z","iopub.execute_input":"2022-07-19T03:25:02.118673Z","iopub.status.idle":"2022-07-19T03:25:07.312869Z","shell.execute_reply.started":"2022-07-19T03:25:02.118604Z","shell.execute_reply":"2022-07-19T03:25:07.311657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nalphas = [0.95, 0.9, 0.8, 0.7, 0.5, 0.4, 0.3]\nlags = [23, 31, 62, 91, 98, 105, 112, 126, 180, 270, 365, 546, 728]\ndf = ewm_features(df, alphas, lags)\ndf.head()\nprint(df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:25:07.320120Z","iopub.execute_input":"2022-07-19T03:25:07.321068Z","iopub.status.idle":"2022-07-19T03:25:44.284503Z","shell.execute_reply.started":"2022-07-19T03:25:07.321026Z","shell.execute_reply":"2022-07-19T03:25:44.283274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # One Hot Encoding\n\n# df = pd.get_dummies(df, columns=['store','item','day_of_week','month'])\n# print(df.shape)\n# df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:25:44.285970Z","iopub.execute_input":"2022-07-19T03:25:44.286319Z","iopub.status.idle":"2022-07-19T03:25:44.291447Z","shell.execute_reply.started":"2022-07-19T03:25:44.286289Z","shell.execute_reply":"2022-07-19T03:25:44.289860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Cost Function\n\nhttps://en.wikipedia.org/wiki/Symmetric_mean_absolute_percentage_error","metadata":{}},{"cell_type":"code","source":"\ndef smape(preds, target):\n    n = len(preds)\n    masked_arr = ~((preds == 0) & (target == 0))\n    preds, target = preds[masked_arr], target[masked_arr]\n    num = np.abs(preds - target)\n    denom = np.abs(preds) + np.abs(target)\n    smape_val = (200 * np.sum(num / denom)) / n\n    return smape_val\n\n\ndef lgbm_smape(y_true, y_pred):\n    smape_val = smape(y_true, y_pred)\n    return 'SMAPE', smape_val, False","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:25:44.292989Z","iopub.execute_input":"2022-07-19T03:25:44.293751Z","iopub.status.idle":"2022-07-19T03:25:44.306132Z","shell.execute_reply.started":"2022-07-19T03:25:44.293711Z","shell.execute_reply":"2022-07-19T03:25:44.304899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train/Validation Split Dataset","metadata":{}},{"cell_type":"code","source":"train_df['date'].min(), train_df['date'].max(), test_df['date'].min(), test_df['date'].max()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:25:44.307982Z","iopub.execute_input":"2022-07-19T03:25:44.308515Z","iopub.status.idle":"2022-07-19T03:25:44.335749Z","shell.execute_reply.started":"2022-07-19T03:25:44.308483Z","shell.execute_reply":"2022-07-19T03:25:44.334730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Validation set includes 3 months (Oct. Nov. Dec. 2017)\ntrain = df.loc[(df['date']<'2017-10-01'), :]\nval = df.loc[(df['date']>='2017-10-01') & (df['date']<'2018-01-01'), :]\n\nprint(train.shape)\nprint(val.shape)\n# Remove cols as features are generated from them\ncols = [col for col in train.columns if col not in ['date','id','sales','year']]","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:25:44.337115Z","iopub.execute_input":"2022-07-19T03:25:44.337692Z","iopub.status.idle":"2022-07-19T03:25:46.424027Z","shell.execute_reply.started":"2022-07-19T03:25:44.337656Z","shell.execute_reply":"2022-07-19T03:25:46.422596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_train = train['sales']\nX_train = train[cols]\n\nY_val = val['sales']\nX_val = val[cols]\n\nY_train.shape, X_train.shape, Y_val.shape, X_val.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:25:46.425573Z","iopub.execute_input":"2022-07-19T03:25:46.426283Z","iopub.status.idle":"2022-07-19T03:25:46.955030Z","shell.execute_reply.started":"2022-07-19T03:25:46.426237Z","shell.execute_reply":"2022-07-19T03:25:46.953910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# First Model","metadata":{}},{"cell_type":"code","source":"first_model = lgb.LGBMRegressor(random_state=384).fit(X_train, Y_train, \n                                                      eval_metric= lambda y_true, y_pred: [lgbm_smape(y_true, y_pred)])\n\nprint(\"TRAIN SMAPE:\", smape(Y_train, first_model.predict(X_train)))\nprint(\"VALID SMAPE:\", smape(Y_val, first_model.predict(X_val)))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:25:46.956128Z","iopub.execute_input":"2022-07-19T03:25:46.956428Z","iopub.status.idle":"2022-07-19T03:26:31.021314Z","shell.execute_reply.started":"2022-07-19T03:25:46.956402Z","shell.execute_reply":"2022-07-19T03:26:31.020264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Importance","metadata":{}},{"cell_type":"code","source":"def plot_lgb_importances(model, plot=False, num=10):\n    \n    # LGBM API\n    #gain = model.feature_importance('gain')\n    #feat_imp = pd.DataFrame({'feature': model.feature_name(),\n    #                         'split': model.feature_importance('split'),\n    #                         'gain': 100 * gain / gain.sum()}).sort_values('gain', ascending=False)\n    \n    # SKLEARN API\n    gain = model.booster_.feature_importance(importance_type='gain')\n    feat_imp = pd.DataFrame({'feature': model.feature_name_,\n                             'split': model.booster_.feature_importance(importance_type='split'),\n                             'gain': 100 * gain / gain.sum()}).sort_values('gain', ascending=False)\n    if plot:\n        plt.figure(figsize=(10, 10))\n        sns.set(font_scale=1)\n        sns.barplot(x=\"gain\", y=\"feature\", data=feat_imp[0:25])\n        plt.title('feature')\n        plt.tight_layout()\n        plt.show()\n    else:\n        print(feat_imp.head(num))\n        return feat_imp\n\nfeature_imp_df = plot_lgb_importances(first_model, num=50)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:26:31.022590Z","iopub.execute_input":"2022-07-19T03:26:31.023506Z","iopub.status.idle":"2022-07-19T03:26:31.041254Z","shell.execute_reply.started":"2022-07-19T03:26:31.023465Z","shell.execute_reply":"2022-07-19T03:26:31.039894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_imp_df.shape, feature_imp_df[feature_imp_df.gain > 0].shape, feature_imp_df[feature_imp_df.gain > 0.57].shape","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:26:31.042728Z","iopub.execute_input":"2022-07-19T03:26:31.043748Z","iopub.status.idle":"2022-07-19T03:26:31.052546Z","shell.execute_reply.started":"2022-07-19T03:26:31.043710Z","shell.execute_reply":"2022-07-19T03:26:31.051656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_lgb_importances(first_model, plot=True, num=30)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:26:31.054285Z","iopub.execute_input":"2022-07-19T03:26:31.055056Z","iopub.status.idle":"2022-07-19T03:26:31.462896Z","shell.execute_reply.started":"2022-07-19T03:26:31.055019Z","shell.execute_reply":"2022-07-19T03:26:31.461531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Error Analysis","metadata":{}},{"cell_type":"code","source":"error = pd.DataFrame({\n    \"date\":val.date,\n    \"store\":X_val.store,\n    \"item\":X_val.item,\n    \"actual\":Y_val,\n    \"pred\":first_model.predict(X_val)\n}).reset_index(drop = True)\n\nerror[\"error\"] = np.abs(error.actual-error.pred)\n\nerror.sort_values(\"error\", ascending=False).head(20)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:26:31.464684Z","iopub.execute_input":"2022-07-19T03:26:31.465154Z","iopub.status.idle":"2022-07-19T03:26:31.667214Z","shell.execute_reply.started":"2022-07-19T03:26:31.465107Z","shell.execute_reply":"2022-07-19T03:26:31.665950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"error[[\"actual\", \"pred\", \"error\"]].describe([0.7, 0.8, 0.9, 0.95, 0.99]).T","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:26:31.668454Z","iopub.execute_input":"2022-07-19T03:26:31.668781Z","iopub.status.idle":"2022-07-19T03:26:31.702127Z","shell.execute_reply.started":"2022-07-19T03:26:31.668751Z","shell.execute_reply":"2022-07-19T03:26:31.701301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Mean Absolute Error\nerror.groupby([\"store\"]).error.mean().sort_values(ascending = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:26:31.703321Z","iopub.execute_input":"2022-07-19T03:26:31.703908Z","iopub.status.idle":"2022-07-19T03:26:31.714480Z","shell.execute_reply.started":"2022-07-19T03:26:31.703876Z","shell.execute_reply":"2022-07-19T03:26:31.713215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Mean Absolute Error\nerror.groupby([\"item\"]).error.mean().sort_values(ascending = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:26:31.715799Z","iopub.execute_input":"2022-07-19T03:26:31.716787Z","iopub.status.idle":"2022-07-19T03:26:31.727306Z","shell.execute_reply.started":"2022-07-19T03:26:31.716749Z","shell.execute_reply":"2022-07-19T03:26:31.726462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Store 1 Actual - Pred\nsub = error[error.store == 1].set_index(\"date\")\nfig, axes = plt.subplots(10, 5, figsize=(20, 35))\nfor i in range(1,51):\n    if i < 6:\n        sub[sub.item == i].actual.plot(ax=axes[0, i-1], legend=True, label = \"Item \"+str(i)+\" Sales\")\n        sub[sub.item == i].pred.plot(ax=axes[0, i - 1], legend=True, label=\"Item \" + str(i) + \" Pred\", linestyle = \"dashed\")\n    if i >= 6 and i<11:\n        sub[sub.item == i].actual.plot(ax=axes[1, i - 6], legend=True, label = \"Item \"+str(i)+\" Sales\")\n        sub[sub.item == i].pred.plot(ax=axes[1, i - 6], legend=True, label=\"Item \" + str(i) + \" Pred\",  linestyle=\"dashed\")\n    if i >= 11 and i<16:\n        sub[sub.item == i].actual.plot(ax=axes[2, i - 11], legend=True, label = \"Item \"+str(i)+\" Sales\")\n        sub[sub.item == i].pred.plot(ax=axes[2, i - 11], legend=True, label=\"Item \" + str(i) + \" Pred\", linestyle=\"dashed\")\n    if i >= 16 and i<21:\n        sub[sub.item == i].actual.plot(ax=axes[3, i - 16], legend=True, label = \"Item \"+str(i)+\" Sales\")\n        sub[sub.item == i].pred.plot(ax=axes[3, i - 16], legend=True, label=\"Item \" + str(i) + \" Pred\", linestyle=\"dashed\")\n    if i >= 21 and i<26:\n        sub[sub.item == i].actual.plot(ax=axes[4, i - 21], legend=True, label = \"Item \"+str(i)+\" Sales\")\n        sub[sub.item == i].pred.plot(ax=axes[4, i - 21], legend=True, label=\"Item \" + str(i) + \" Pred\", linestyle=\"dashed\")\n    if i >= 26 and i<31:\n        sub[sub.item == i].actual.plot(ax=axes[5, i - 26], legend=True, label = \"Item \"+str(i)+\" Sales\")\n        sub[sub.item == i].pred.plot(ax=axes[5, i - 26], legend=True, label=\"Item \" + str(i) + \" Pred\", linestyle=\"dashed\")\n    if i >= 31 and i<36:\n        sub[sub.item == i].actual.plot(ax=axes[6, i - 31], legend=True, label = \"Item \"+str(i)+\" Sales\")\n        sub[sub.item == i].pred.plot(ax=axes[6, i - 31], legend=True, label=\"Item \" + str(i) + \" Pred\", linestyle=\"dashed\")\n    if i >= 36 and i<41:\n        sub[sub.item == i].actual.plot(ax=axes[7, i - 36], legend=True, label = \"Item \"+str(i)+\" Sales\")\n        sub[sub.item == i].pred.plot(ax=axes[7, i - 36], legend=True, label=\"Item \" + str(i) + \" Pred\", linestyle=\"dashed\")\n    if i >= 41 and i<46:\n        sub[sub.item == i].actual.plot(ax=axes[8, i - 41], legend=True, label = \"Item \"+str(i)+\" Sales\")\n        sub[sub.item == i].pred.plot(ax=axes[8, i - 41], legend=True, label=\"Item \" + str(i) + \" Pred\",linestyle=\"dashed\")\n    if i >= 46 and i<51:\n        sub[sub.item == i].actual.plot(ax=axes[9, i - 46], legend=True, label = \"Item \"+str(i)+\" Sales\")\n        sub[sub.item == i].pred.plot(ax=axes[9, i - 46], legend=True, label=\"Item \" + str(i) + \" Pred\",linestyle=\"dashed\")\nplt.tight_layout(pad=4.5)\nplt.suptitle(\"Store 1 Item\");\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:26:31.728938Z","iopub.execute_input":"2022-07-19T03:26:31.729437Z","iopub.status.idle":"2022-07-19T03:26:45.483664Z","shell.execute_reply.started":"2022-07-19T03:26:31.729406Z","shell.execute_reply":"2022-07-19T03:26:45.482401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(4, 2, figsize = (20,20))\nfor axi in axes.flat:\n    axi.ticklabel_format(style=\"sci\", axis=\"y\", scilimits=(0,10))\n    axi.ticklabel_format(style=\"sci\", axis=\"x\", scilimits=(0,10))\n    axi.get_yaxis().set_major_formatter(plt.FuncFormatter(lambda x, loc: \"{:,}\".format(int(x))))\n    axi.get_xaxis().set_major_formatter(plt.FuncFormatter(lambda x, loc: \"{:,}\".format(int(x))))\n    \n(error.actual-error.pred).hist(ax = axes[0, 0], color = \"steelblue\", bins = 20)\nerror.error.hist(ax = axes[0,1], color = \"steelblue\", bins = 20)\nsr = error.copy()\nsr[\"StandardizedR\"] = (sr.error / (sr.actual-sr.pred).std())\nsr[\"StandardizedR2\"] = ((sr.error / (sr.actual-sr.pred).std())**2)\nsr.plot.scatter(x = \"pred\",y = \"StandardizedR\", color = \"red\", ax = axes[1,0])\nsr.plot.scatter(x = \"pred\",y = \"StandardizedR2\", color = \"red\", ax = axes[1,1])\nerror.actual.hist(ax = axes[2, 0], color = \"purple\", bins = 20)\nerror.pred.hist(ax = axes[2, 1], color = \"purple\", bins = 20)\nerror.plot.scatter(x = \"actual\",y = \"pred\", color = \"seagreen\", ax = axes[3,0]);\n# QQ Plot\nimport statsmodels.api as sm\nimport pylab\nsm.qqplot(sr.pred, ax = axes[3,1], c = \"seagreen\")\nplt.suptitle(\"ERROR ANALYSIS\", fontsize = 20)\naxes[0,0].set_title(\"Error Histogram\", fontsize = 15)\naxes[0,1].set_title(\"Absolute Error Histogram\", fontsize = 15)\naxes[1,0].set_title(\"Standardized Residuals & Fitted Values\", fontsize = 15)\naxes[1,1].set_title(\"Standardized Residuals^2 & Fitted Values\", fontsize = 15)\naxes[2,0].set_title(\"Actual Histogram\", fontsize = 15)\naxes[2,1].set_title(\"Pred Histogram\", fontsize = 15);\naxes[3,0].set_title(\"Actual Pred Relationship\", fontsize = 15);\naxes[3,1].set_title(\"QQ Plot\", fontsize = 15);\naxes[1,0].set_xlabel(\"Fitted Values (Pred)\", fontsize = 12)\naxes[1,1].set_xlabel(\"Fitted Values (Pred)\", fontsize = 12)\naxes[3,0].set_xlabel(\"Actual\", fontsize = 12)\naxes[1,0].set_ylabel(\"Standardized Residuals\", fontsize = 12)\naxes[1,1].set_ylabel(\"Standardized Residuals^2\", fontsize = 12)\naxes[3,0].set_ylabel(\"Pred\", fontsize = 12)\nfig.tight_layout(pad=3.0)\nplt.savefig(\"errors.png\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:26:45.485447Z","iopub.execute_input":"2022-07-19T03:26:45.485907Z","iopub.status.idle":"2022-07-19T03:26:49.774487Z","shell.execute_reply.started":"2022-07-19T03:26:45.485866Z","shell.execute_reply":"2022-07-19T03:26:49.772709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Next Model","metadata":{}},{"cell_type":"code","source":"# First model feature importance\ncols = feature_imp_df[feature_imp_df.gain > 0.01].feature.tolist()\nprint(\"Independent Variables:\", len(cols))\n\nsecond_model = lgb.LGBMRegressor(random_state=384).fit(\n    X_train[cols], Y_train, \n    eval_metric= lambda y_true, y_pred: [lgbm_smape(y_true, y_pred)])\n\nprint(\"TRAIN SMAPE:\", smape(Y_train, second_model.predict(X_train[cols])))\nprint(\"VALID SMAPE:\", smape(Y_val, second_model.predict(X_val[cols])))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:26:49.776387Z","iopub.execute_input":"2022-07-19T03:26:49.776871Z","iopub.status.idle":"2022-07-19T03:27:00.636583Z","shell.execute_reply.started":"2022-07-19T03:26:49.776827Z","shell.execute_reply":"2022-07-19T03:27:00.635691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain_final = df.loc[(df[\"date\"] < \"2018-01-01\"), :]\ntest_final = df.loc[(df[\"date\"] >= \"2018-01-01\"), :]\n\nX_train_final = train_final[cols]\nY_train_final = train_final.sales\nX_test_final = test_final[cols]\n\n\n#final_model = lgb.LGBMRegressor(**rsearch.best_params_, random_state=384, metric = \"custom\") # Tuned parameters\n# Best Params: {'num_leaves': 31, 'n_estimators': 15000, 'max_depth': 20}\nfinal_model = lgb.LGBMRegressor(num_leaves=31,\n    # , n_estimators=15000,\n     max_depth=20, random_state=384, metric = \"custom\")\nfinal_model.fit(X_train_final[cols], Y_train_final,\n                eval_metric= lambda y_true, y_pred: [lgbm_smape(y_true, y_pred)])","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:27:00.637785Z","iopub.execute_input":"2022-07-19T03:27:00.638327Z","iopub.status.idle":"2022-07-19T03:27:12.138848Z","shell.execute_reply.started":"2022-07-19T03:27:00.638294Z","shell.execute_reply":"2022-07-19T03:27:12.137723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Forecast","metadata":{}},{"cell_type":"code","source":"pred = pd.DataFrame({\n    \"id\":test_final.id.astype(int),\n    \"sales\":final_model.predict(X_test_final)\n})\npred.to_csv(\"submission.csv\", index = None)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:27:12.140185Z","iopub.execute_input":"2022-07-19T03:27:12.141279Z","iopub.status.idle":"2022-07-19T03:27:12.368805Z","shell.execute_reply.started":"2022-07-19T03:27:12.141229Z","shell.execute_reply":"2022-07-19T03:27:12.367532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nforecast = pd.DataFrame({\n    \"date\":test_final.date,\n    \"store\":test_final.store,\n    \"item\":test_final.item,\n    \"sales\":final_model.predict(X_test_final)\n})\n\nforecast[(forecast.store == 1) & (forecast.item == 1)].set_index(\"date\").sales.plot(color = \"orange\", figsize = (20,9),legend=True, label = \"Store 1 Item 1 Forecast\");","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:27:12.370045Z","iopub.execute_input":"2022-07-19T03:27:12.370383Z","iopub.status.idle":"2022-07-19T03:27:13.202330Z","shell.execute_reply.started":"2022-07-19T03:27:12.370356Z","shell.execute_reply":"2022-07-19T03:27:13.201173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_final[(train_final.store == 1) & (train_final.item == 17)].set_index(\"date\").sales.plot(figsize = (20,9),legend=True, label = \"Store 1 Item 1 Sales\")\nforecast[(forecast.store == 1) & (forecast.item == 17)].set_index(\"date\").sales.plot(legend=True, label = \"Store 1 Item 1 Forecast\");","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:27:13.203414Z","iopub.execute_input":"2022-07-19T03:27:13.203712Z","iopub.status.idle":"2022-07-19T03:27:13.643035Z","shell.execute_reply.started":"2022-07-19T03:27:13.203686Z","shell.execute_reply":"2022-07-19T03:27:13.641720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"store = 1\nsub = train[train.store == store].set_index(\"date\")\nforc = forecast[forecast.store == store].set_index(\"date\")\n\n\nfig, axes = plt.subplots(10, 5, figsize=(20, 35))\nfor i in range(1,51):\n    if i < 6:\n        sub[sub.item == i].sales.plot(ax=axes[0, i-1], legend=True, label = \"Item \"+str(i)+\" Sales\")\n        forc[forc.item == i].sales.plot(ax=axes[0, i-1], legend=True, label = \"Forecast\")\n    if i >= 6 and i<11:\n        sub[sub.item == i].sales.plot(ax=axes[1, i - 6], legend=True, label = \"Item \"+str(i)+\" Sales\")\n        forc[forc.item == i].sales.plot(ax=axes[1, i-6], legend=True, label = \"Forecast\")\n    if i >= 11 and i<16:\n        sub[sub.item == i].sales.plot(ax=axes[2, i - 11], legend=True, label = \"Item \"+str(i)+\" Sales\") \n        forc[forc.item == i].sales.plot(ax=axes[2, i-11], legend=True, label = \"Forecast\")\n    if i >= 16 and i<21:\n        sub[sub.item == i].sales.plot(ax=axes[3, i - 16], legend=True, label = \"Item \"+str(i)+\" Sales\")    \n        forc[forc.item == i].sales.plot(ax=axes[3, i-16], legend=True, label = \"Forecast\")\n    if i >= 21 and i<26:\n        sub[sub.item == i].sales.plot(ax=axes[4, i - 21], legend=True, label = \"Item \"+str(i)+\" Sales\") \n        forc[forc.item == i].sales.plot(ax=axes[4, i-21], legend=True, label = \"Forecast\")\n    if i >= 26 and i<31:\n        sub[sub.item == i].sales.plot(ax=axes[5, i - 26], legend=True, label = \"Item \"+str(i)+\" Sales\")\n        forc[forc.item == i].sales.plot(ax=axes[5, i-26], legend=True, label = \"Forecast\")\n    if i >= 31 and i<36:\n        sub[sub.item == i].sales.plot(ax=axes[6, i - 31], legend=True, label = \"Item \"+str(i)+\" Sales\")  \n        forc[forc.item == i].sales.plot(ax=axes[6, i-31], legend=True, label = \"Forecast\")\n    if i >= 36 and i<41:\n        sub[sub.item == i].sales.plot(ax=axes[7, i - 36], legend=True, label = \"Item \"+str(i)+\" Sales\")\n        forc[forc.item == i].sales.plot(ax=axes[7, i-36], legend=True, label = \"Forecast\")\n    if i >= 41 and i<46:\n        sub[sub.item == i].sales.plot(ax=axes[8, i - 41], legend=True, label = \"Item \"+str(i)+\" Sales\") \n        forc[forc.item == i].sales.plot(ax=axes[8, i-41], legend=True, label = \"Forecast\")\n    if i >= 46 and i<51:\n        sub[sub.item == i].sales.plot(ax=axes[9, i - 46], legend=True, label = \"Item \"+str(i)+\" Sales\") \n        forc[forc.item == i].sales.plot(ax=axes[9, i-46], legend=True, label = \"Forecast\")\nplt.tight_layout(pad=6.5)\nplt.suptitle(\"Store 1 Items Actual & Forecast\");","metadata":{"execution":{"iopub.status.busy":"2022-07-19T03:27:13.644538Z","iopub.execute_input":"2022-07-19T03:27:13.645398Z","iopub.status.idle":"2022-07-19T03:27:29.811298Z","shell.execute_reply.started":"2022-07-19T03:27:13.645362Z","shell.execute_reply":"2022-07-19T03:27:29.810467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}