{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Hi everybody\n## In this notebook I will analyze and forecast the time series of store sales for Kaggle's competition.","metadata":{}},{"cell_type":"code","source":"# importing libraries.\nimport numpy as np \nimport pandas as pd \npd.set_option('display.float_format', '{:.2f}'.format)\npd.set_option('display.max_rows', 100)\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nsns.set_style('darkgrid')\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nimport plotly.express as px\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objs as go\n\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.impute import KNNImputer\nfrom sklearn.preprocessing import OneHotEncoder","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:05.264672Z","iopub.execute_input":"2022-08-03T11:24:05.265843Z","iopub.status.idle":"2022-08-03T11:24:09.017261Z","shell.execute_reply.started":"2022-08-03T11:24:05.265799Z","shell.execute_reply":"2022-08-03T11:24:09.015620Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Before reading the data we need to understand first what this data is about, fortunately, Kaggle has provided some documentaries about each dataset that we'll be dealing with.\n\n# File Descriptions and Data Field Information\n## train.csv\n### The training data, comprising time series of features store_nbr, family, and onpromotion as well as the target sales.\n### store_nbr identifies the store at which the products are sold.\n### family identifies the type of product sold.\n### sales gives the total sales for a product family at a particular store at a given date. Fractional values are possible since products can be sold in fractional units (1.5 kg of cheese, for instance, as opposed to 1 bag of chips).\n### onpromotion gives the total number of items in a product family that were being promoted at a store at a given date.\n## test.csv\n### The test data, having the same features as the training data. You will predict the target sales for the dates in this file.\n### The dates in the test data are for the 15 days after the last date in the training data.\n## sample_submission.csv\n### A sample submission file in the correct format.\n## oil.csv\n### Daily oil price. Includes values during both the train and test data timeframes. (Ecuador is an oil-dependent country and it's economical health is highly vulnerable to shocks in oil prices.)\n## holidays_events.csv\n### Holidays and Events, with metadata\n### NOTE: Pay special attention to the transferred column. A holiday that is transferred officially falls on that calendar day, but was moved to another date by the government. A transferred day is more like a normal day than a holiday. To find the day that it was actually celebrated, look for the corresponding row where type is Transfer. For example, the holiday Independencia de Guayaquil was transferred from 2012-10-09 to 2012-10-12, which means it was celebrated on 2012-10-12. Days that are type Bridge are extra days that are added to a holiday (e.g., to extend the break across a long weekend). These are frequently made up by the type Work Day which is a day not normally scheduled for work (e.g., Saturday) that is meant to payback the Bridge.\n### Additional holidays are days added a regular calendar holiday, for example, as typically happens around Christmas (making Christmas Eve a holiday).","metadata":{}},{"cell_type":"code","source":"# reading the data.\nholidays = pd.read_csv('../input/store-sales-time-series-forecasting/holidays_events.csv', header = 0)\noil = pd.read_csv('../input/store-sales-time-series-forecasting/oil.csv', header = 0)\nstores = pd.read_csv('../input/store-sales-time-series-forecasting/stores.csv', header = 0)\ntrans = pd.read_csv('../input/store-sales-time-series-forecasting/transactions.csv', header = 0)\n\ntrain = pd.read_csv('../input/store-sales-time-series-forecasting/train.csv', header = 0)\ntest = pd.read_csv('../input/store-sales-time-series-forecasting/test.csv', header = 0)\n\n# converting date column in each dataset to DateTime format.\nholidays['date'] = pd.to_datetime(holidays['date'], format = \"%Y-%m-%d\")\noil['date'] = pd.to_datetime(oil['date'], format = \"%Y-%m-%d\")\ntrans['date'] = pd.to_datetime(trans['date'], format = \"%Y-%m-%d\")\ntrain['date'] = pd.to_datetime(train['date'], format = \"%Y-%m-%d\")\ntest['date'] = pd.to_datetime(test['date'], format = \"%Y-%m-%d\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:09.019000Z","iopub.execute_input":"2022-08-03T11:24:09.019371Z","iopub.status.idle":"2022-08-03T11:24:12.796299Z","shell.execute_reply.started":"2022-08-03T11:24:09.019338Z","shell.execute_reply":"2022-08-03T11:24:12.794752Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Holidays","metadata":{}},{"cell_type":"code","source":"# showing holidays data.\nholidays","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:12.797840Z","iopub.execute_input":"2022-08-03T11:24:12.798335Z","iopub.status.idle":"2022-08-03T11:24:12.826297Z","shell.execute_reply.started":"2022-08-03T11:24:12.798291Z","shell.execute_reply":"2022-08-03T11:24:12.825124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transD = holidays.transferred.value_counts() # geting the number of transferred holidays.\ntypeD = holidays.type.value_counts() # geting the number of each holiday type.\nlocaleD = holidays.locale.value_counts() # geting the number of each holiday locale.\n\n# plotting the data.\nfig = make_subplots(rows=2, cols=3, specs=[[{\"type\": \"pie\"}, {\"type\": \"pie\"},{\"type\": \"pie\"}],\n                          [{\"colspan\": 3},None,None] ],\n                    vertical_spacing=0.2, horizontal_spacing=0.1,\n                    subplot_titles=('Percentage of Transferred Holidays', 'Percentage of Holidays\\' Typs',\n                                    \"Percentage of Holidays\\' Lacale States\" ,'Holidays\\' Names'))\n\nfig.append_trace(go.Pie(labels=transD.index, values=transD),\n              row=1, col=1)\n\nfig.add_trace(go.Pie(labels=typeD.index, values=typeD),\n              row=1, col=2)\n\nfig.add_trace(go.Pie(labels=localeD.index, values=localeD),\n              row=1, col=3)\n\nfig.add_trace(go.Histogram(x=holidays.locale_name),\n              row=2, col=1)\n\nfig.update_layout(height=800,showlegend=False)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:12.829543Z","iopub.execute_input":"2022-08-03T11:24:12.830838Z","iopub.status.idle":"2022-08-03T11:24:13.147795Z","shell.execute_reply.started":"2022-08-03T11:24:12.830788Z","shell.execute_reply":"2022-08-03T11:24:13.146546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# getting only what we need from holidays' data.\nholidays = holidays[holidays.transferred == False][['date','type','locale','locale_name']][(holidays.date >= '2013-01-01')&(holidays.date <= '2017-08-31')]\nholidays","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:13.149379Z","iopub.execute_input":"2022-08-03T11:24:13.149862Z","iopub.status.idle":"2022-08-03T11:24:13.178705Z","shell.execute_reply.started":"2022-08-03T11:24:13.149818Z","shell.execute_reply":"2022-08-03T11:24:13.177270Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Oil","metadata":{}},{"cell_type":"code","source":"# showing oil data.\noil","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:13.180255Z","iopub.execute_input":"2022-08-03T11:24:13.180695Z","iopub.status.idle":"2022-08-03T11:24:13.199673Z","shell.execute_reply.started":"2022-08-03T11:24:13.180657Z","shell.execute_reply":"2022-08-03T11:24:13.198333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting oil data.\nfig = go.Figure([go.Scatter(x=oil['date'], y=oil['dcoilwtico'],marker=dict(color= '#783242'))])\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:13.202024Z","iopub.execute_input":"2022-08-03T11:24:13.202549Z","iopub.status.idle":"2022-08-03T11:24:13.260565Z","shell.execute_reply.started":"2022-08-03T11:24:13.202500Z","shell.execute_reply":"2022-08-03T11:24:13.259144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oil = oil.set_index('date').fillna(method = 'bfill')\noil_moving_average = oil.rolling(\n    window=365,       \n    center=True,      \n    min_periods=183,  \n).mean()\n\nfig = make_subplots(rows=1, cols=1, vertical_spacing=0.08)\n\nfig.add_trace(go.Scatter(x=oil.index, y=oil['dcoilwtico'], mode='lines',\n                     marker=dict(color= '#783242'), name='365-Day Oil Moving Average'))\n\nfig.add_trace(go.Scatter(x=oil_moving_average.index,y=oil_moving_average.dcoilwtico,mode='lines',name='Trend'))\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:13.262226Z","iopub.execute_input":"2022-08-03T11:24:13.262620Z","iopub.status.idle":"2022-08-03T11:24:13.365955Z","shell.execute_reply.started":"2022-08-03T11:24:13.262583Z","shell.execute_reply":"2022-08-03T11:24:13.364530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's check if there is seasonality in the oil data!","metadata":{}},{"cell_type":"code","source":"def grouped(df, key, freq, col):\n    \"\"\" GROUP DATA WITH CERTAIN FREQUENCY \"\"\"\n    df_grouped = df.groupby([pd.Grouper(key=key, freq=freq)]).agg(mean = (col, 'mean'))\n    df_grouped = df_grouped.reset_index()\n    return df_grouped\n\ndef seasonal_plot(X, y, period, freq, ax=None):\n    if ax is None:\n        _, ax = plt.subplots()\n    palette = sns.color_palette(\"husl\", n_colors=X[period].nunique(),)\n    ax = sns.lineplot(x=X[freq], \n                      y=X[y],\n                      ax=ax, \n                      hue=X[period],\n                      palette=palette, \n                      legend=False)\n    ax.set_title(f\"Seasonal Plot ({period}/{freq})\")\n    for line, name in zip(ax.lines, X[period].unique()):\n        y_ = line.get_ydata()[-1]\n        ax.annotate(name, \n                    xy=(1, y_), \n                    xytext=(6, 0), \n                    color=line.get_color(), \n                    xycoords=ax.get_yaxis_transform(), \n                    textcoords=\"offset points\", \n                    size=14, \n                    va=\"center\")\n    return ax\n\ndef plot_periodogram(ts, detrend='linear', ax=None):\n    from scipy.signal import periodogram\n    fs = pd.Timedelta(\"365D\") / pd.Timedelta(\"1D\")\n    freqencies, spectrum = periodogram(ts, fs=fs, detrend=detrend, window=\"boxcar\", scaling='spectrum')\n    if ax is None:\n        _, ax = plt.subplots()\n    ax.step(freqencies, spectrum, color=\"purple\")\n    ax.set_xscale(\"log\")\n    ax.set_xticks([1, 2, 4, 6, 12, 26, 52, 104])\n    ax.set_xticklabels([\"Annual (1)\", \"Semiannual (2)\", \"Quarterly (4)\", \n                        \"Bimonthly (6)\", \"Monthly (12)\", \"Biweekly (26)\", \n                        \"Weekly (52)\", \"Semiweekly (104)\"], rotation=30)\n    ax.ticklabel_format(axis=\"y\", style=\"sci\", scilimits=(0, 0))\n    ax.set_ylabel(\"Variance\")\n    ax.set_title(\"Periodogram\")\n    return ax\n\ndef seasonality(df, key, freq, col):\n    df_grouped = grouped(df, key, freq, col)\n    df_grouped['date'] = pd.to_datetime(df_grouped['date'], format = \"%Y-%m-%d\")\n    df_grouped.index = df_grouped['date'] \n    df_grouped = df_grouped.drop(columns=['date'])\n    df_grouped.index.freq = freq # manually set the frequency of the index\n    \n    X = df_grouped.copy()\n    X.index = pd.to_datetime(X.index, format = \"%Y-%m-%d\") \n    X.index.freq = freq \n    # days within a week\n    X[\"day\"] = X.index.dayofweek   # the x-axis (freq)\n    X[\"week\"] = pd.Int64Index(X.index.isocalendar().week)  # the seasonal period (period)\n    # days within a year\n    X[\"dayofyear\"] = X.index.dayofyear\n    X[\"year\"] = X.index.year\n    fig, (ax0, ax1, ax2) = plt.subplots(3, 1, figsize=(20, 30))\n    seasonal_plot(X, y='mean', period=\"week\", freq=\"day\", ax=ax0)\n    seasonal_plot(X, y='mean', period=\"year\", freq=\"dayofyear\", ax=ax1)\n    X_new = (X['mean'].copy()).dropna()\n    plot_periodogram(X_new, ax=ax2)\n    \nseasonality(oil.reset_index(), 'date', 'D', 'dcoilwtico')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:13.367777Z","iopub.execute_input":"2022-08-03T11:24:13.368134Z","iopub.status.idle":"2022-08-03T11:24:21.282218Z","shell.execute_reply.started":"2022-08-03T11:24:13.368103Z","shell.execute_reply":"2022-08-03T11:24:21.280893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transactions","metadata":{}},{"cell_type":"code","source":"# showing transactions data.\ntrans","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:21.287538Z","iopub.execute_input":"2022-08-03T11:24:21.288053Z","iopub.status.idle":"2022-08-03T11:24:21.306227Z","shell.execute_reply.started":"2022-08-03T11:24:21.288004Z","shell.execute_reply":"2022-08-03T11:24:21.305004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting transactions data.\ntrans_date = trans.drop('store_nbr',axis=1).groupby('date').mean()\nstore_nbr_trans = trans.drop('date',axis=1).groupby('store_nbr').mean()\n\ntrans_date_moving_average = trans_date.rolling(\n    window=365,       \n    center=True,      \n    min_periods=183,  \n).mean()\n\nfig = make_subplots(rows=2, cols=1, vertical_spacing=0.09,subplot_titles=('Transactions Moving Average','Average Store Transactions'))\n\nfig.add_trace(go.Scatter(x=trans_date.index, y=trans_date['transactions'], mode='lines',\n              marker=dict(color= '#1fdbb9')),\n              row=1, col=1)\n\nfig.add_trace(go.Scatter(x=trans_date_moving_average.index,y=trans_date_moving_average.transactions,mode='lines',marker=dict(color= '#067863'),name='Trend'),\n              row=1, col=1)\n\nfig.add_trace(go.Bar(x=store_nbr_trans.index, y=store_nbr_trans.transactions),\n              row=2, col=1)\n\nfig.update_layout(height=800,showlegend=False)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:21.307982Z","iopub.execute_input":"2022-08-03T11:24:21.308480Z","iopub.status.idle":"2022-08-03T11:24:21.452597Z","shell.execute_reply.started":"2022-08-03T11:24:21.308413Z","shell.execute_reply":"2022-08-03T11:24:21.451127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the seasonality in transactions data.\nseasonality(trans, 'date', 'D', 'transactions')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:21.454200Z","iopub.execute_input":"2022-08-03T11:24:21.454591Z","iopub.status.idle":"2022-08-03T11:24:31.676861Z","shell.execute_reply.started":"2022-08-03T11:24:21.454557Z","shell.execute_reply":"2022-08-03T11:24:31.675790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Data","metadata":{}},{"cell_type":"code","source":"# showing train data.\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:31.678201Z","iopub.execute_input":"2022-08-03T11:24:31.679458Z","iopub.status.idle":"2022-08-03T11:24:31.703242Z","shell.execute_reply.started":"2022-08-03T11:24:31.679403Z","shell.execute_reply":"2022-08-03T11:24:31.702138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_modi = train.drop('id',axis=1) # droping the ID column.\n\n# identifying holidays dates.\ntrain_modi['holiday'] = 0\nfor date in holidays.date:\n    train_modi['holiday'][train_modi.date == date] = 1\n    \ntrain_modi = pd.merge(train_modi, oil, how='left', on = 'date') # joining oil data.\ntrain_modi = pd.merge(train_modi, trans, how='left', on = ['date','store_nbr']) # joining transactions data.\n\ntrain_modi.drop(['store_nbr','family'],axis=1,inplace=True) # droping unnecessary columns.\n\ntrain_modi = train_modi.groupby('date').mean() # grouping by date.\ntrain_modi.fillna(method = 'bfill',inplace=True) # filling nan values.\ntrain_modi","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:31.704559Z","iopub.execute_input":"2022-08-03T11:24:31.704913Z","iopub.status.idle":"2022-08-03T11:24:39.147196Z","shell.execute_reply.started":"2022-08-03T11:24:31.704881Z","shell.execute_reply":"2022-08-03T11:24:39.145811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sales = train_modi.sales # sales data.\nstore_sales = train[['store_nbr','sales']].groupby('store_nbr').mean() # getting the average sales for each store.\nfamily_sales = train[['family','sales']].groupby('family').mean() # getting the average sales for each family.\nfamily_onpromotions = train[['family','onpromotion']].groupby('family').mean() #  getting the average promotion for each family.\n\n# plotting.\ntrain_sales_moving_average = train_sales.rolling(\n    window=365,       \n    center=True,      \n    min_periods=183,  \n).mean()\n\nfig = make_subplots(rows=4, cols=1, vertical_spacing=0.09,\n                    subplot_titles=('Sales Moving Average', 'Store Sales',\n                                    \"Family Sales\" ,'Family Onpromotions'))\n\nfig.add_trace(go.Scatter(x=train_sales.index, y=train_sales, mode='lines',\n              marker=dict(color= '#dbd623')),\n              row=1, col=1)\n\nfig.add_trace(go.Scatter(x=train_sales_moving_average.index,y=train_sales_moving_average,mode='lines',marker=dict(color= '#eb8328'),name='Trend'),\n              row=1, col=1)\n\nfig.add_trace(go.Bar(x=store_sales.index, y=store_sales.sales),\n              row=2, col=1)\n\nfig.add_trace(go.Bar(x=family_sales.index, y=family_sales.sales),\n              row=3, col=1)\n\nfig.add_trace(go.Bar(x=family_onpromotions.index, y=family_onpromotions.onpromotion),\n              row=4, col=1)\n\nfig.update_layout(height=1750,showlegend=False)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:39.148552Z","iopub.execute_input":"2022-08-03T11:24:39.149143Z","iopub.status.idle":"2022-08-03T11:24:40.088140Z","shell.execute_reply.started":"2022-08-03T11:24:39.149110Z","shell.execute_reply":"2022-08-03T11:24:40.086784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the seasonality in train data.\nseasonality(train_modi.reset_index(), 'date', 'D', 'sales')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:40.089975Z","iopub.execute_input":"2022-08-03T11:24:40.090741Z","iopub.status.idle":"2022-08-03T11:24:50.400626Z","shell.execute_reply.started":"2022-08-03T11:24:40.090701Z","shell.execute_reply":"2022-08-03T11:24:50.399402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train Data Correlations","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 15))\nsns.heatmap(train_modi.corr(), annot=True, cmap=\"flare_r\", linewidths=0.1, annot_kws={\"fontsize\":10})\nplt.title(\"Correlation US accidents - return rate\");","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:50.402172Z","iopub.execute_input":"2022-08-03T11:24:50.402656Z","iopub.status.idle":"2022-08-03T11:24:50.814482Z","shell.execute_reply.started":"2022-08-03T11:24:50.402614Z","shell.execute_reply":"2022-08-03T11:24:50.813176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"markdown","source":"## Preprcessing","metadata":{}},{"cell_type":"code","source":"full_data = pd.concat([train, test]).drop('id',axis=1)\n\nfull_data['holiday'] = 0\nfor date in holidays.date:\n    full_data['holiday'][full_data.date == date] = 1\n    \nfull_data = pd.merge(full_data, oil, how='left', on = 'date')\n\nfull_data['year'] = full_data.date.dt.year\nfull_data['month'] = full_data.date.dt.month\nfull_data['monthday'] = full_data.date.dt.day\nfull_data['weekday'] = full_data.date.dt.dayofweek\n\ntrain = full_data.iloc[:len(train),:] \ntest = full_data.iloc[len(train):,:] ","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:50.816098Z","iopub.execute_input":"2022-08-03T11:24:50.816522Z","iopub.status.idle":"2022-08-03T11:24:58.746874Z","shell.execute_reply.started":"2022-08-03T11:24:50.816488Z","shell.execute_reply":"2022-08-03T11:24:58.745675Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:24:58.748725Z","iopub.execute_input":"2022-08-03T11:24:58.749492Z","iopub.status.idle":"2022-08-03T11:24:58.771747Z","shell.execute_reply.started":"2022-08-03T11:24:58.749444Z","shell.execute_reply":"2022-08-03T11:24:58.770601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:25:25.175388Z","iopub.execute_input":"2022-08-03T11:25:25.175822Z","iopub.status.idle":"2022-08-03T11:25:25.197306Z","shell.execute_reply.started":"2022-08-03T11:25:25.175785Z","shell.execute_reply":"2022-08-03T11:25:25.196503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.dropna(axis=0,subset = ['sales'], inplace = True) # droping null values from target.\n\nX_train_full = train.drop(['sales'], axis =1)\ny_train_full = train.sales\n\nX_train, X_valid, y_train, y_valid = train_test_split(X_train_full,y_train_full,test_size=0.2, random_state=42) # spliting the data.","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:30:41.631553Z","iopub.execute_input":"2022-08-03T11:30:41.632007Z","iopub.status.idle":"2022-08-03T11:30:43.436372Z","shell.execute_reply.started":"2022-08-03T11:30:41.631969Z","shell.execute_reply":"2022-08-03T11:30:43.434836Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# A categorical transformer.\ncat_trans = Pipeline(steps = [\n    ('imputer',SimpleImputer(strategy = 'most_frequent')),\n    ('ohe',OneHotEncoder(handle_unknown = 'ignore'))\n])\n\n# A preprocessor that combines the two previous transformers.\npreprocessor = ColumnTransformer(transformers = [\n    ('cat', cat_trans, ['family'])\n],\n    remainder = \"drop\")\n\nX_train_trans = preprocessor.fit_transform(X_train).toarray()\nX_valid_trans = preprocessor.transform(X_valid).toarray()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:32:01.482382Z","iopub.execute_input":"2022-08-03T11:32:01.482874Z","iopub.status.idle":"2022-08-03T11:32:04.226924Z","shell.execute_reply.started":"2022-08-03T11:32:01.482839Z","shell.execute_reply":"2022-08-03T11:32:04.226030Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training Model","metadata":{}},{"cell_type":"code","source":"LR_model = LinearRegression() # creating linear regression model.\n\nLR_model.fit(X_train_trans, y_train) # fitting the model with the training data.\npreds_LR = LR_model.predict(X_valid_trans) # predict the target.\n\nmean_absolute_error(y_valid, preds_LR) # geting the mean absolute error for predictions.","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:58:16.221338Z","iopub.execute_input":"2022-08-03T11:58:16.221871Z","iopub.status.idle":"2022-08-03T11:58:20.262328Z","shell.execute_reply.started":"2022-08-03T11:58:16.221829Z","shell.execute_reply":"2022-08-03T11:58:20.261103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Now let's make our submission!","metadata":{}},{"cell_type":"code","source":"# preprocessing whole train and test data.\ntrain_trans = preprocessor.fit_transform(X_train_full).toarray()\ntest_trans = preprocessor.transform(test.drop('sales',axis=1)).toarray()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:48:46.380169Z","iopub.execute_input":"2022-08-03T11:48:46.380690Z","iopub.status.idle":"2022-08-03T11:48:49.530799Z","shell.execute_reply.started":"2022-08-03T11:48:46.380649Z","shell.execute_reply":"2022-08-03T11:48:49.529464Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LR_model = LinearRegression()\n\nLR_model.fit(train_trans, y_train_full)\npreds_LR = LR_model.predict(test_trans)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:48:50.776642Z","iopub.execute_input":"2022-08-03T11:48:50.777066Z","iopub.status.idle":"2022-08-03T11:48:55.672716Z","shell.execute_reply.started":"2022-08-03T11:48:50.777033Z","shell.execute_reply":"2022-08-03T11:48:55.671461Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/store-sales-time-series-forecasting/sample_submission.csv') # reading the sample_submission dataset.\nsubmission['sales'] = preds_LR # assigning sales to our predictions.\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:52:36.607386Z","iopub.execute_input":"2022-08-03T11:52:36.607837Z","iopub.status.idle":"2022-08-03T11:52:36.631302Z","shell.execute_reply.started":"2022-08-03T11:52:36.607804Z","shell.execute_reply":"2022-08-03T11:52:36.630111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False, header=True) # submitting.","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:54:49.199818Z","iopub.execute_input":"2022-08-03T11:54:49.200315Z","iopub.status.idle":"2022-08-03T11:54:49.267625Z","shell.execute_reply.started":"2022-08-03T11:54:49.200278Z","shell.execute_reply":"2022-08-03T11:54:49.266491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# All Done!","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}