{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**This notebook is an exercise in the [Time Series](https://www.kaggle.com/learn/time-series) course.  You can reference the tutorial at [this link](https://www.kaggle.com/ryanholbrook/linear-regression-with-time-series).**\n\n---\n","metadata":{}},{"cell_type":"markdown","source":"# ❤️️❤️️❤️️❤️️ Sayang Semangaaat ❤️️❤️️❤️️❤️️","metadata":{}},{"cell_type":"markdown","source":"# Introduction #\n\nRun this cell to set everything up!","metadata":{}},{"cell_type":"code","source":"# Setup feedback system\nfrom learntools.core import binder\nbinder.bind(globals())\nfrom learntools.time_series.ex1 import *\n\n# Setup notebook\nfrom pathlib import Path\nfrom learntools.time_series.style import *  # plot style settings\n\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport seaborn as sns\nfrom sklearn.linear_model import LinearRegression\n\n\ndata_dir = Path('../input/ts-course-data/')\ncomp_dir = Path('../input/store-sales-time-series-forecasting')\n\nbook_sales = pd.read_csv(\n    data_dir / 'book_sales.csv',\n    index_col='Date',\n    parse_dates=['Date'],\n).drop('Paperback', axis=1)\nbook_sales['Time'] = np.arange(len(book_sales.index))\nbook_sales['Lag_1'] = book_sales['Hardcover'].shift(1)\nbook_sales = book_sales.reindex(columns=['Hardcover', 'Time', 'Lag_1'])\n\nar = pd.read_csv(data_dir / 'ar.csv')\n\ndtype = {\n    'store_nbr': 'category',\n    'family': 'category',\n    'sales': 'float32',\n    'onpromotion': 'uint64',\n}\nstore_sales = pd.read_csv(\n    comp_dir / 'train.csv',\n    dtype=dtype,\n    parse_dates=['date'],\n    infer_datetime_format=True,\n)\nstore_sales = store_sales.set_index('date').to_period('D')\nstore_sales = store_sales.set_index(['store_nbr', 'family'], append=True)\naverage_sales = store_sales.groupby('date').mean()['sales']","metadata":{"execution":{"iopub.status.busy":"2022-08-09T02:36:27.051188Z","iopub.execute_input":"2022-08-09T02:36:27.052223Z","iopub.status.idle":"2022-08-09T02:36:34.837304Z","shell.execute_reply.started":"2022-08-09T02:36:27.052183Z","shell.execute_reply":"2022-08-09T02:36:34.836026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path  #ambil library Path dari pathlib\n\ndata_dir = Path('../input/ts-course-data/')   #kita menunjukkan lokasi folder yang mau kita pake lalu dimasukkan ke variabel data_dir\ncomp_dir = Path('../input/store-sales-time-series-forecasting')   #kita menunjukkan lokasi folder lalu dimasukkan ke variabel comp_dir\nprint(data_dir)  # menunjukkan path data_dir","metadata":{"execution":{"iopub.status.busy":"2022-08-09T02:14:27.453553Z","iopub.execute_input":"2022-08-09T02:14:27.453993Z","iopub.status.idle":"2022-08-09T02:14:27.461154Z","shell.execute_reply.started":"2022-08-09T02:14:27.453956Z","shell.execute_reply":"2022-08-09T02:14:27.459847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd #ambil library pandas sebagai pd. Jadinya nanti kalo manggil tinggal pake pd\n\nbook_sales = pd.read_csv(data_dir / 'book_sales.csv')    #bikin variabel book_sales, datanya diisi dari baca csv yang ada di folder data_dir dengan nama file 'book_sales.csv'\nbook_sales.head()  #menampilkan 5 data pertama dari book_sales","metadata":{"execution":{"iopub.status.busy":"2022-08-09T02:16:40.353761Z","iopub.execute_input":"2022-08-09T02:16:40.354197Z","iopub.status.idle":"2022-08-09T02:16:40.391358Z","shell.execute_reply.started":"2022-08-09T02:16:40.354163Z","shell.execute_reply":"2022-08-09T02:16:40.389625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"book_sales2 = pd.read_csv(\n    data_dir / 'book_sales.csv',\n    index_col = 'Date'     # menggunakan kolom 'Date' sebagai kolom index [index_col dilihat dari parameter fungsi]\n)\nbook_sales2.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T02:24:14.430712Z","iopub.execute_input":"2022-08-09T02:24:14.431183Z","iopub.status.idle":"2022-08-09T02:24:14.455088Z","shell.execute_reply.started":"2022-08-09T02:24:14.431146Z","shell.execute_reply":"2022-08-09T02:24:14.453883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"book_sales3 = pd.read_csv(\n    data_dir / 'book_sales.csv',\n    index_col = 'Date',\n    parse_dates = ['Date']     # membaca kolom 'Date' sebagai tanggal (di book_sales2 masih sebaai string biasa)\n)\nbook_sales3.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T02:24:16.739316Z","iopub.execute_input":"2022-08-09T02:24:16.740593Z","iopub.status.idle":"2022-08-09T02:24:16.759508Z","shell.execute_reply.started":"2022-08-09T02:24:16.740539Z","shell.execute_reply":"2022-08-09T02:24:16.758290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**INGAT INDEX DI PYTHON DIMULAI DARI 0,1,2,3.....**","metadata":{}},{"cell_type":"code","source":"book_sales4 = pd.read_csv(\n    data_dir / 'book_sales.csv',\n    index_col='Date',\n    parse_dates=['Date'],\n).drop('Paperback', axis=1)     # membuang(drop) kolom yang bernama 'Paperback' pada axis(nomor kolom) 1\nbook_sales4.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T02:24:19.666549Z","iopub.execute_input":"2022-08-09T02:24:19.667477Z","iopub.status.idle":"2022-08-09T02:24:19.686511Z","shell.execute_reply.started":"2022-08-09T02:24:19.667420Z","shell.execute_reply":"2022-08-09T02:24:19.685553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\nbook_sales4['Time'] = np.arange(len(book_sales4.index))   \n# membuat kolom 'Time' dengan isi arrange(urutan angka mulai dari 0 hingga x) dengan batasan x adalah len(book_sales4.index)\n#len(book_sales4.index) artinya length(panjang) dari book_sales4.index (index dari book_sales 4)\n# len tidak harus menggunakan kolom index. Bisa kolom yang lain. Namun sebaiknya tetap menggunakan index karena kolom index tidak mungkin tidak memilii nilai.\nbook_sales4.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T02:24:41.510283Z","iopub.execute_input":"2022-08-09T02:24:41.510669Z","iopub.status.idle":"2022-08-09T02:24:41.531444Z","shell.execute_reply.started":"2022-08-09T02:24:41.510639Z","shell.execute_reply":"2022-08-09T02:24:41.530258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"book_sales4['Lag_1'] = book_sales4['Hardcover'].shift(1)\n# membuat kolom 'Lag_1'\n# isi dari kolom adalah data yang ada pada kolom 'Hardcover' lalu digeser(shift) sebesar 1 satuan\nbook_sales4.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T02:25:56.645730Z","iopub.execute_input":"2022-08-09T02:25:56.646248Z","iopub.status.idle":"2022-08-09T02:25:56.661016Z","shell.execute_reply.started":"2022-08-09T02:25:56.646205Z","shell.execute_reply":"2022-08-09T02:25:56.659908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"book_sales4 = book_sales4.reindex(columns=['Time', 'Hardcover', 'Lag_1'])\n# reindex digunakan untuk melakukan penataan kolom / merubah tatanan kolom\n# disini kolom diubah dari [Hardcover - Time - Lag_1] menjadi [Time - Harcover - Lag_1]\nbook_sales4.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T02:27:13.359985Z","iopub.execute_input":"2022-08-09T02:27:13.360464Z","iopub.status.idle":"2022-08-09T02:27:13.375028Z","shell.execute_reply.started":"2022-08-09T02:27:13.360429Z","shell.execute_reply":"2022-08-09T02:27:13.374183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ❤️️❤️️❤️️❤️️ 🤗 Kita Belajar masih sampe sini sayang 🤗 ❤️️❤️️❤️️❤️️","metadata":{}},{"cell_type":"code","source":"#INI BELUM DIBAHAS YA CANTIK :) \nar = pd.read_csv(data_dir / 'ar.csv')\n\ndtype = {\n    'store_nbr': 'category',\n    'family': 'category',\n    'sales': 'float32',\n    'onpromotion': 'uint64',\n}\nstore_sales = pd.read_csv(\n    comp_dir / 'train.csv',\n    dtype=dtype,\n    parse_dates=['date'],\n    infer_datetime_format=True,\n)\nstore_sales = store_sales.set_index('date').to_period('D')\nstore_sales = store_sales.set_index(['store_nbr', 'family'], append=True)\naverage_sales = store_sales.groupby('date').mean()['sales']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"--------------------------------------------------------------------------------\n\nOne advantage linear regression has over more complicated algorithms is that the models it creates are *explainable* -- it's easy to interpret what contribution each feature makes to the predictions. In the model `target = weight * feature + bias`, the `weight` tells you by how much the `target` changes on average for each unit of change in the `feature`.\n\nRun the next cell to see a linear regression on *Hardcover Sales*.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots()\nax.plot('Time', 'Hardcover', data=book_sales, color='0.75')\nax = sns.regplot(x='Time', y='Hardcover', data=book_sales, ci=None, scatter_kws=dict(color='0.25'))\nax.set_title('Time Plot of Hardcover Sales');","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1) Interpret linear regression with the time dummy\n\nThe linear regression line has an equation of (approximately) `Hardcover = 3.33 * Time + 150.5`. Over 6 days how much on average would you expect hardcover sales to change? After you've thought about it, run the next cell.","metadata":{}},{"cell_type":"code","source":"# View the solution (Run this line to receive credit!)\nq_1.check()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Uncomment the next line for a hint\n#q_1.hint()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"-------------------------------------------------------------------------------\n\nInterpreting the regression coefficients can help us recognize serial dependence in a time plot. Consider the model `target = weight * lag_1 + error`, where `error` is random noise and `weight` is a number between -1 and 1. The `weight` in this case tells you how likely the next time step will have the same sign as the previous time step: a `weight` close to 1 means `target` will likely have the same sign as the previous step, while a `weight` close to -1 means `target` will likely have the opposite sign.\n\n# 2) Interpret linear regression with a lag feature\n\nRun the following cell to see two series generated according to the model just described.","metadata":{}},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(2, 1, figsize=(11, 5.5), sharex=True)\nax1.plot(ar['ar1'])\nax1.set_title('Series 1')\nax2.plot(ar['ar2'])\nax2.set_title('Series 2');","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"One of these series has the equation `target = 0.95 * lag_1 + error` and the other has the equation `target = -0.95 * lag_1 + error`, differing only by the sign on the lag feature. Can you tell which equation goes with each series?","metadata":{}},{"cell_type":"code","source":"# View the solution (Run this cell to receive credit!)\nq_2.check()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Uncomment the next line for a hint\n#q_2.hint()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"-------------------------------------------------------------------------------\n\nNow we'll get started with the *Store Sales - Time Series Forecasting* competition data. The entire dataset comprises almost 1800 series recording store sales across a variety of product families from 2013 into 2017. For this lesson, we'll just work with a single series (`average_sales`) of the average sales each day.\n\n# 3) Fit a time-step feature\n\nComplete the code below to create a linear regression model with a time-step feature on the series of average product sales. The target is in a column called `'sales'`.","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\n\ndf = average_sales.to_frame()\n\n# YOUR CODE HERE: Create a time dummy\ntime = ____\n\ndf['time'] = time \n\n# YOUR CODE HERE: Create training data\nX = ____  # features\ny = ____  # target\n\n# Train the model\nmodel = LinearRegression()\nmodel.fit(X, y)\n\n# Store the fitted values as a time series with the same time index as\n# the training data\ny_pred = pd.Series(model.predict(X), index=X.index)\n\n\n# Check your answer\nq_3.check()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lines below will give you a hint or solution code\n#q_3.hint()\n#q_3.solution()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Run this cell if you'd like to see a plot of the result.","metadata":{}},{"cell_type":"code","source":"ax = y.plot(**plot_params, alpha=0.5)\nax = y_pred.plot(ax=ax, linewidth=3)\nax.set_title('Time Plot of Total Store Sales');","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"-------------------------------------------------------------------------------\n\n# 4) Fit a lag feature to Store Sales\n\nComplete the code below to create a linear regression model with a lag feature on the series of average product sales. The target is in a column of `df` called `'sales'`.","metadata":{}},{"cell_type":"code","source":"df = average_sales.to_frame()\n\n# YOUR CODE HERE: Create a lag feature from the target 'sales'\nlag_1 = ____\n\ndf['lag_1'] = lag_1  # add to dataframe\n\nX = df.loc[:, ['lag_1']].dropna()  # features\ny = df.loc[:, 'sales']  # target\ny, X = y.align(X, join='inner')  # drop corresponding values in target\n\n# YOUR CODE HERE: Create a LinearRegression instance and fit it to X and y.\nmodel = ____\n\n# YOUR CODE HERE: Create Store the fitted values as a time series with\n# the same time index as the training data\ny_pred = ____\n\n\n# Check your answer\nq_4.check()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lines below will give you a hint or solution code\nq_4.hint()\nq_4.solution()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Run the next cell if you'd like to see the result.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots()\nax.plot(X['lag_1'], y, '.', color='0.25')\nax.plot(X['lag_1'], y_pred)\nax.set(aspect='equal', ylabel='sales', xlabel='lag_1', title='Lag Plot of Average Sales');","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Keep Going #\n\n[**Model trend**](https://www.kaggle.com/ryanholbrook/trend) in time series with moving average plots and the time dummy.","metadata":{}},{"cell_type":"markdown","source":"---\n\n\n\n\n*Have questions or comments? Visit the [course discussion forum](https://www.kaggle.com/learn/time-series/discussion) to chat with other learners.*","metadata":{}}]}