{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-14T00:51:17.858065Z","iopub.execute_input":"2022-07-14T00:51:17.859047Z","iopub.status.idle":"2022-07-14T00:51:17.894778Z","shell.execute_reply.started":"2022-07-14T00:51:17.858912Z","shell.execute_reply":"2022-07-14T00:51:17.893839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import datetime\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport catboost\nfrom catboost import Pool\nfrom catboost import CatBoostRegressor\nfrom xgboost import XGBRegressor\nfrom xgboost import plot_importance\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler\n\n%matplotlib inline\nsns.set(style=\"darkgrid\")\npd.set_option('display.float_format', lambda x: '%.2f' % x)\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:17.896655Z","iopub.execute_input":"2022-07-14T00:51:17.897292Z","iopub.status.idle":"2022-07-14T00:51:19.911684Z","shell.execute_reply.started":"2022-07-14T00:51:17.897258Z","shell.execute_reply":"2022-07-14T00:51:19.910838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from statsmodels.tsa.deterministic import CalendarFourier, DeterministicProcess\n\n\n\n# Set Matplotlib defaults\nplt.style.use(\"seaborn-whitegrid\")\nplt.rc(\"figure\", autolayout=True, figsize=(11, 5))\nplt.rc(\n    \"axes\",\n    labelweight=\"bold\",\n    labelsize=\"large\",\n    titleweight=\"bold\",\n    titlesize=16,\n    titlepad=10,\n)\nplot_params = dict(\n    color=\"0.75\",\n    style=\".-\",\n    markeredgecolor=\"0.25\",\n    markerfacecolor=\"0.25\",\n    legend=False,\n)\n%config InlineBackend.figure_format = 'retina'\n\n\n# annotations: https://stackoverflow.com/a/49238256/5769929\ndef seasonal_plot(X, y, period, freq, ax=None):\n    if ax is None:\n        _, ax = plt.subplots()\n    palette = sns.color_palette(\"husl\", n_colors=X[period].nunique(),)\n    ax = sns.lineplot(\n        x=freq,\n        y=y,\n        hue=period,\n        data=X,\n        ci=False,\n        ax=ax,\n        palette=palette,\n        legend=False,\n    )\n    ax.set_title(f\"Seasonal Plot ({period}/{freq})\")\n    for line, name in zip(ax.lines, X[period].unique()):\n        y_ = line.get_ydata()[-1]\n        ax.annotate(\n            name,\n            xy=(1, y_),\n            xytext=(6, 0),\n            color=line.get_color(),\n            xycoords=ax.get_yaxis_transform(),\n            textcoords=\"offset points\",\n            size=14,\n            va=\"center\",\n        )\n    return ax\n\n\ndef plot_periodogram(ts, detrend='linear', ax=None):\n    from scipy.signal import periodogram\n    fs = pd.Timedelta(\"1Y\") / pd.Timedelta(\"1D\")\n    freqencies, spectrum = periodogram(\n        ts,\n        fs=fs,\n        detrend=detrend,\n        window=\"boxcar\",\n        scaling='spectrum',\n    )\n    if ax is None:\n        _, ax = plt.subplots()\n    ax.step(freqencies, spectrum, color=\"purple\")\n    ax.set_xscale(\"log\")\n    ax.set_xticks([1, 2, 4, 6, 12, 26, 52, 104])\n    ax.set_xticklabels(\n        [\n            \"Annual (1)\",\n            \"Semiannual (2)\",\n            \"Quarterly (4)\",\n            \"Bimonthly (6)\",\n            \"Monthly (12)\",\n            \"Biweekly (26)\",\n            \"Weekly (52)\",\n            \"Semiweekly (104)\",\n        ],\n        rotation=30,\n    )\n    ax.ticklabel_format(axis=\"y\", style=\"sci\", scilimits=(0, 0))\n    ax.set_ylabel(\"Variance\")\n    ax.set_title(\"Periodogram\")\n    return ax","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:19.913507Z","iopub.execute_input":"2022-07-14T00:51:19.914195Z","iopub.status.idle":"2022-07-14T00:51:20.020297Z","shell.execute_reply.started":"2022-07-14T00:51:19.914148Z","shell.execute_reply":"2022-07-14T00:51:20.018639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/tabular-playground-series-jan-2022/train.csv',parse_dates=['date'])\ntest = pd.read_csv('/kaggle/input/tabular-playground-series-jan-2022/test.csv',parse_dates=['date'])","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:20.032316Z","iopub.execute_input":"2022-07-14T00:51:20.037629Z","iopub.status.idle":"2022-07-14T00:51:20.178121Z","shell.execute_reply.started":"2022-07-14T00:51:20.037555Z","shell.execute_reply":"2022-07-14T00:51:20.177179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/tabular-playground-series-jan-2022/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:20.180560Z","iopub.execute_input":"2022-07-14T00:51:20.180890Z","iopub.status.idle":"2022-07-14T00:51:20.195695Z","shell.execute_reply.started":"2022-07-14T00:51:20.180860Z","shell.execute_reply":"2022-07-14T00:51:20.194261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:20.197400Z","iopub.execute_input":"2022-07-14T00:51:20.198241Z","iopub.status.idle":"2022-07-14T00:51:20.215947Z","shell.execute_reply.started":"2022-07-14T00:51:20.198202Z","shell.execute_reply":"2022-07-14T00:51:20.214764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:20.217449Z","iopub.execute_input":"2022-07-14T00:51:20.218247Z","iopub.status.idle":"2022-07-14T00:51:20.232415Z","shell.execute_reply.started":"2022-07-14T00:51:20.218212Z","shell.execute_reply":"2022-07-14T00:51:20.231053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Min date from train set:' , test['date'].min().date())\nprint('Max date from train set:' , test['date'].max().date())","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:20.234218Z","iopub.execute_input":"2022-07-14T00:51:20.234876Z","iopub.status.idle":"2022-07-14T00:51:20.243349Z","shell.execute_reply.started":"2022-07-14T00:51:20.234845Z","shell.execute_reply":"2022-07-14T00:51:20.242181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head().abs","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:20.245123Z","iopub.execute_input":"2022-07-14T00:51:20.245934Z","iopub.status.idle":"2022-07-14T00:51:20.261199Z","shell.execute_reply.started":"2022-07-14T00:51:20.245890Z","shell.execute_reply":"2022-07-14T00:51:20.259908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Min date from train set:' , train['date'].min().date())\nprint('Max date from train set:' , train['date'].max().date())","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:20.266830Z","iopub.execute_input":"2022-07-14T00:51:20.269455Z","iopub.status.idle":"2022-07-14T00:51:20.280285Z","shell.execute_reply.started":"2022-07-14T00:51:20.269402Z","shell.execute_reply":"2022-07-14T00:51:20.279230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:20.281521Z","iopub.execute_input":"2022-07-14T00:51:20.282353Z","iopub.status.idle":"2022-07-14T00:51:20.302846Z","shell.execute_reply.started":"2022-07-14T00:51:20.282276Z","shell.execute_reply":"2022-07-14T00:51:20.301658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:20.304262Z","iopub.execute_input":"2022-07-14T00:51:20.304832Z","iopub.status.idle":"2022-07-14T00:51:20.337987Z","shell.execute_reply.started":"2022-07-14T00:51:20.304801Z","shell.execute_reply":"2022-07-14T00:51:20.336806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\npx.box(train, x = 'num_sold')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:20.343652Z","iopub.execute_input":"2022-07-14T00:51:20.346670Z","iopub.status.idle":"2022-07-14T00:51:23.079598Z","shell.execute_reply.started":"2022-07-14T00:51:20.346618Z","shell.execute_reply":"2022-07-14T00:51:23.078225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_date_features(df):\n    df['month'] = df.date.dt.month.astype(\"int8\")\n    #df['day_of_month'] = df.date.dt.day.astype(\"int8\")\n    df['day_of_year'] = df.date.dt.dayofyear.astype(\"int16\")\n    #df['week_of_month'] = (df.date.apply(lambda d: (d.day-1) // 7 + 1)).astype(\"int8\")\n    df['week_of_year'] = (df.date.dt.weekofyear).astype(\"int8\")\n    df['day_of_week'] = (df.date.dt.dayofweek + 1).astype(\"int8\")\n    df['year'] = df.date.dt.year.astype(\"int32\")\n    df[\"is_wknd\"] = (df.date.dt.weekday // 4).astype(\"int8\")\n    #df[\"quarter\"] = df.date.dt.quarter.astype(\"int8\")\n    #df['is_month_start'] = df.date.dt.is_month_start.astype(\"int8\")\n    #df['is_month_end'] = df.date.dt.is_month_end.astype(\"int8\")\n    #df['is_quarter_start'] = df.date.dt.is_quarter_start.astype(\"int8\")\n    #df['is_quarter_end'] = df.date.dt.is_quarter_end.astype(\"int8\")\n    #df['is_year_start'] = df.date.dt.is_year_start.astype(\"int8\")\n    #df['is_year_end'] = df.date.dt.is_year_end.astype(\"int8\")\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:53:31.544086Z","iopub.execute_input":"2022-07-14T00:53:31.544470Z","iopub.status.idle":"2022-07-14T00:53:31.555531Z","shell.execute_reply.started":"2022-07-14T00:53:31.544439Z","shell.execute_reply":"2022-07-14T00:53:31.554565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train.copy()\ncreate_date_features(train_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:56:50.206669Z","iopub.execute_input":"2022-07-14T00:56:50.207505Z","iopub.status.idle":"2022-07-14T00:56:50.420566Z","shell.execute_reply.started":"2022-07-14T00:56:50.207460Z","shell.execute_reply":"2022-07-14T00:56:50.419124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:56:52.062658Z","iopub.execute_input":"2022-07-14T00:56:52.063073Z","iopub.status.idle":"2022-07-14T00:56:52.103218Z","shell.execute_reply.started":"2022-07-14T00:56:52.063040Z","shell.execute_reply":"2022-07-14T00:56:52.101277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:59:16.462785Z","iopub.execute_input":"2022-07-14T00:59:16.463187Z","iopub.status.idle":"2022-07-14T00:59:16.472088Z","shell.execute_reply.started":"2022-07-14T00:59:16.463154Z","shell.execute_reply":"2022-07-14T00:59:16.470865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-07-14T01:00:23.582881Z","iopub.execute_input":"2022-07-14T01:00:23.583277Z","iopub.status.idle":"2022-07-14T01:00:23.605571Z","shell.execute_reply.started":"2022-07-14T01:00:23.583244Z","shell.execute_reply":"2022-07-14T01:00:23.604444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T01:02:49.286431Z","iopub.execute_input":"2022-07-14T01:02:49.286828Z","iopub.status.idle":"2022-07-14T01:02:49.309590Z","shell.execute_reply.started":"2022-07-14T01:02:49.286793Z","shell.execute_reply":"2022-07-14T01:02:49.308418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, axes = plt.subplots(2, 1, figsize=(15,10), sharex=True)\nsns.barplot(data=train_df[train_df['store'] == 'KaggleRama'], x='country',y='num_sold').set_title('KaggleRama')\nsns.barplot(data=train_df[train_df['store'] == 'KaggleMart'], x='country',y='num_sold', ax=axes[0]).set_title('KaggleMart')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T01:00:50.775298Z","iopub.execute_input":"2022-07-14T01:00:50.775719Z","iopub.status.idle":"2022-07-14T01:00:51.850994Z","shell.execute_reply.started":"2022-07-14T01:00:50.775668Z","shell.execute_reply":"2022-07-14T01:00:51.849786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Check for outliers\nsns.jointplot(x='num_sold', y='date', data=train_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T01:01:25.649504Z","iopub.execute_input":"2022-07-14T01:01:25.649892Z","iopub.status.idle":"2022-07-14T01:01:27.046071Z","shell.execute_reply.started":"2022-07-14T01:01:25.649860Z","shell.execute_reply":"2022-07-14T01:01:27.044954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(x='date',y='num_sold', data=train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T01:01:38.842963Z","iopub.execute_input":"2022-07-14T01:01:38.843497Z","iopub.status.idle":"2022-07-14T01:02:19.947000Z","shell.execute_reply.started":"2022-07-14T01:01:38.843453Z","shell.execute_reply":"2022-07-14T01:02:19.945781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.query('num_sold >=1500')['day_of_year'].nunique","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.101821Z","iopub.status.idle":"2022-07-14T00:51:23.102461Z","shell.execute_reply.started":"2022-07-14T00:51:23.102230Z","shell.execute_reply":"2022-07-14T00:51:23.102260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from statsmodels.graphics.tsaplots import plot_acf, plot_pacf","metadata":{"execution":{"iopub.status.busy":"2022-07-14T01:06:17.559587Z","iopub.execute_input":"2022-07-14T01:06:17.560016Z","iopub.status.idle":"2022-07-14T01:06:17.721643Z","shell.execute_reply.started":"2022-07-14T01:06:17.559981Z","shell.execute_reply":"2022-07-14T01:06:17.720160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def smape_loss(y_true, y_pred):\n    \"\"\"SMAPE Loss\"\"\"\n    return np.abs(y_true - y_pred) / (y_true + np.abs(y_pred)) * 200","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.103591Z","iopub.status.idle":"2022-07-14T00:51:23.104017Z","shell.execute_reply.started":"2022-07-14T00:51:23.103807Z","shell.execute_reply":"2022-07-14T00:51:23.103826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\nOneHot_enc = OneHotEncoder()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.105190Z","iopub.status.idle":"2022-07-14T00:51:23.106004Z","shell.execute_reply.started":"2022-07-14T00:51:23.105767Z","shell.execute_reply":"2022-07-14T00:51:23.105790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_df = OneHot_enc.fit_transform(train_df[['country','product','store']])","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.107105Z","iopub.status.idle":"2022-07-14T00:51:23.107765Z","shell.execute_reply.started":"2022-07-14T00:51:23.107498Z","shell.execute_reply":"2022-07-14T00:51:23.107525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"column_name = OneHot_enc.get_feature_names_out(['country', 'product','store'])","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.108916Z","iopub.status.idle":"2022-07-14T00:51:23.109531Z","shell.execute_reply.started":"2022-07-14T00:51:23.109324Z","shell.execute_reply":"2022-07-14T00:51:23.109351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"one_hot_encoded_frame = pd.DataFrame.sparse.from_spmatrix(temp_df, columns=column_name)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.110684Z","iopub.status.idle":"2022-07-14T00:51:23.111339Z","shell.execute_reply.started":"2022-07-14T00:51:23.111113Z","shell.execute_reply":"2022-07-14T00:51:23.111140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"one_hot_encoded_frame = train_df.join(one_hot_encoded_frame)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.112606Z","iopub.status.idle":"2022-07-14T00:51:23.113334Z","shell.execute_reply.started":"2022-07-14T00:51:23.113048Z","shell.execute_reply":"2022-07-14T00:51:23.113090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"one_hot_encoded_frame.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.114358Z","iopub.status.idle":"2022-07-14T00:51:23.115100Z","shell.execute_reply.started":"2022-07-14T00:51:23.114877Z","shell.execute_reply":"2022-07-14T00:51:23.114900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoded_df = one_hot_encoded_frame.drop('country', axis=1).drop('store',axis=1).drop('product',axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.116630Z","iopub.status.idle":"2022-07-14T00:51:23.117305Z","shell.execute_reply.started":"2022-07-14T00:51:23.117095Z","shell.execute_reply":"2022-07-14T00:51:23.117116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoded_df","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.118548Z","iopub.status.idle":"2022-07-14T00:51:23.119256Z","shell.execute_reply.started":"2022-07-14T00:51:23.119033Z","shell.execute_reply":"2022-07-14T00:51:23.119054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"date_df = encoded_df.set_index('date')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.120480Z","iopub.status.idle":"2022-07-14T00:51:23.121151Z","shell.execute_reply.started":"2022-07-14T00:51:23.120920Z","shell.execute_reply":"2022-07-14T00:51:23.120941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"date_df['Time'] = np.arange(len(date_df.index))\ndate_df['Lag_1'] = date_df['num_sold'].shift(1)\ndate_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.122412Z","iopub.status.idle":"2022-07-14T00:51:23.123093Z","shell.execute_reply.started":"2022-07-14T00:51:23.122869Z","shell.execute_reply":"2022-07-14T00:51:23.122893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"date_df = date_df.fillna(method='bfill')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.124210Z","iopub.status.idle":"2022-07-14T00:51:23.124599Z","shell.execute_reply.started":"2022-07-14T00:51:23.124412Z","shell.execute_reply":"2022-07-14T00:51:23.124430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_periodogram(date_df.num_sold)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.126165Z","iopub.status.idle":"2022-07-14T00:51:23.126554Z","shell.execute_reply.started":"2022-07-14T00:51:23.126364Z","shell.execute_reply":"2022-07-14T00:51:23.126383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.127945Z","iopub.status.idle":"2022-07-14T00:51:23.128581Z","shell.execute_reply.started":"2022-07-14T00:51:23.128380Z","shell.execute_reply":"2022-07-14T00:51:23.128401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = encoded_df.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.129809Z","iopub.status.idle":"2022-07-14T00:51:23.130469Z","shell.execute_reply.started":"2022-07-14T00:51:23.130257Z","shell.execute_reply":"2022-07-14T00:51:23.130279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = 10\nbatch = 256\nlr = 0.0003\nadam = optimizers.Adam(lr)\n\nmodel_cnn = Sequential()\nmodel_cnn.add(Conv1D(filters=64, kernel_size=2, activation='relu', input_shape=(X_train_series.shape[1], X_train_series.shape[2])))\nmodel_cnn.add(MaxPooling1D(pool_size=2))\nmodel_cnn.add(Flatten())\nmodel_cnn.add(Dense(50, activation='relu'))\nmodel_cnn.add(Dense(1))\nmodel_cnn.compile(loss='mse', optimizer=adam)\nmodel_cnn.summary()\n\ncnn_history = model_cnn.fit(X_train_series, Y_train, validation_data=(X_valid_series, Y_valid), epochs=epochs, verbose=2)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.131624Z","iopub.status.idle":"2022-07-14T00:51:23.132299Z","shell.execute_reply.started":"2022-07-14T00:51:23.132086Z","shell.execute_reply":"2022-07-14T00:51:23.132108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"original_train_df = pd.read_csv('../input/tabular-playground-series-jan-2022/train.csv')\noriginal_test_df = pd.read_csv('../input/tabular-playground-series-jan-2022/test.csv')\n\n# The dates are read as strings and must be converted\nfor df in [original_train_df, original_test_df]:\n    df['date'] = pd.to_datetime(df.date)\noriginal_train_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.133480Z","iopub.status.idle":"2022-07-14T00:51:23.134175Z","shell.execute_reply.started":"2022-07-14T00:51:23.133972Z","shell.execute_reply":"2022-07-14T00:51:23.133995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\ndef engineer(df):\n    \"\"\"Return a new dataframe with the engineered features\"\"\"\n    new_df = pd.DataFrame({\n                           'wd4': df.date.dt.weekday == 4, # Friday\n                           'wd56': df.date.dt.weekday >= 5, # Saturday and Sunday\n                          })\n    # One-hot encoding (no need to encode the last categories)\n    for country in ['Finland', 'Norway']:\n        new_df[country] = df.country == country\n    new_df['KaggleRama'] = df.store == 'KaggleRama'\n    for product in ['Kaggle Mug', 'Kaggle Hat']:\n        new_df[product] = df['product'] == product\n        \n    # Seasonal variations (Fourier series)\n    # The three products have different seasonal patterns\n    dayofyear = df.date.dt.dayofyear\n    for k in range(1, 3):\n        new_df[f'sin{k}'] = np.sin(dayofyear / 365 * 2 * math.pi * k)\n        new_df[f'cos{k}'] = np.cos(dayofyear / 365 * 2 * math.pi * k)\n        new_df[f'mug_sin{k}'] = new_df[f'sin{k}'] * new_df['Kaggle Mug']\n        new_df[f'mug_cos{k}'] = new_df[f'cos{k}'] * new_df['Kaggle Mug']\n        new_df[f'hat_sin{k}'] = new_df[f'sin{k}'] * new_df['Kaggle Hat']\n        new_df[f'hat_cos{k}'] = new_df[f'cos{k}'] * new_df['Kaggle Hat']\n\n    return new_df\n\ntrain_df = engineer(original_train_df)\ntrain_df['date'] = original_train_df.date\ntrain_df['num_sold'] = original_train_df.num_sold.astype(np.float32)\ntest_df = engineer(original_test_df)\n\nfeatures = test_df.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.135365Z","iopub.status.idle":"2022-07-14T00:51:23.136037Z","shell.execute_reply.started":"2022-07-14T00:51:23.135834Z","shell.execute_reply":"2022-07-14T00:51:23.135856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.137161Z","iopub.status.idle":"2022-07-14T00:51:23.137548Z","shell.execute_reply.started":"2022-07-14T00:51:23.137361Z","shell.execute_reply":"2022-07-14T00:51:23.137379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[features]","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.138898Z","iopub.status.idle":"2022-07-14T00:51:23.139301Z","shell.execute_reply.started":"2022-07-14T00:51:23.139086Z","shell.execute_reply":"2022-07-14T00:51:23.139103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fit_model(X_tr, X_va=None, outliers=False):\n    \"\"\"Scale the data, fit a model, plot the training history and validate the model\"\"\"\n    \n\n    # Preprocess the data\n    X_tr_f = X_tr[features]\n    preproc = StandardScaler()\n    X_tr_f = preproc.fit_transform(X_tr_f)\n    y_tr = X_tr.num_sold.values.reshape(-1, 1)\n    \n    # Train the model\n    model = LinearRegression()\n    #model = HuberRegressor(epsilon=1.20, max_iter=500)\n    #model = Ridge()\n    model.fit(X_tr_f, np.log(y_tr).ravel())\n\n    if X_va is not None:\n        # Preprocess the validation data\n        X_va_f = X_va[features]\n        X_va_f = preproc.transform(X_va_f)\n        y_va = X_va.num_sold.values.reshape(-1, 1)\n\n        # Inference for validation\n        y_va_pred = np.exp(model.predict(X_va_f)).reshape(-1, 1)\n        oof.update(pd.Series(y_va_pred.ravel(), index=X_va.index))\n        \n        # Evaluation: Execution time and SMAPE\n        smape_before_correction = np.mean(smape_loss(y_va, y_va_pred))\n        #y_va_pred *= LOSS_CORRECTION\n        smape = np.mean(smape_loss(y_va, y_va_pred))\n        print(f\"Fold {run}.{fold} | {str(datetime.now() - start_time)[-12:-7]}\"\n              f\" | SMAPE: {smape:.5f}   (before correction: {smape_before_correction:.5f})\")\n        score_list.append(smape)\n        \n        # Plot y_true vs. y_pred\n        if fold == 0:\n            plt.figure(figsize=(10, 10))\n            plt.scatter(y_va, y_va_pred, s=1, color='r')\n            #plt.scatter(np.log(y_va), np.log(y_va_pred), s=1, color='g')\n            plt.plot([plt.xlim()[0], plt.xlim()[1]], [plt.xlim()[0], plt.xlim()[1]], '--', color='k')\n            plt.gca().set_aspect('equal')\n            plt.xlabel('y_true')\n            plt.ylabel('y_pred')\n            plt.title('OOF Predictions')\n            plt.show()\n        \n    return preproc, model\n\npreproc, model = fit_model(train_df)\n\n# Plot all num_sold_true and num_sold_pred (five years) for one country-store-product combination\ndef plot_five_years_combination(engineer, country='Norway', store='KaggleMart', product='Kaggle Hat'):\n    demo_df = pd.DataFrame({'row_id': 0,\n                            'date': pd.date_range('2015-01-01', '2019-12-31', freq='D'),\n                            'country': country,\n                            'store': store,\n                            'product': product})\n    demo_df.set_index('date', inplace=True, drop=False)\n    demo_df = engineer(demo_df)\n    demo_df['num_sold'] = np.exp(model.predict(preproc.transform(demo_df[features])))\n    plt.figure(figsize=(20, 6))\n    plt.plot(np.arange(len(demo_df)), demo_df.num_sold, label='prediction')\n    train_subset = train_df[(original_train_df.country == country) & (original_train_df.store == store) & (original_train_df['product'] == product)]\n    plt.scatter(np.arange(len(train_subset)), train_subset.num_sold, label='true', alpha=0.5, color='red', s=3)\n    plt.legend()\n    plt.title('Predictions and true num_sold for five years')\n    plt.show()\n\nplot_five_years_combination(engineer)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.141402Z","iopub.status.idle":"2022-07-14T00:51:23.141867Z","shell.execute_reply.started":"2022-07-14T00:51:23.141612Z","shell.execute_reply":"2022-07-14T00:51:23.141630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['pred'] = np.exp(model.predict(preproc.transform(train_df[features])))\nresiduals = np.log(train_df.pred) - np.log(train_df.num_sold)\nplt.figure(figsize=(18, 4))\nplt.scatter(np.arange(len(residuals)), residuals, s=1)\nplt.title('All residuals by row number')\nplt.ylabel('residual')\nplt.show()\nplt.figure(figsize=(18, 4))\nplt.hist(residuals, bins=200)\nplt.title('Histogram of all residuals')\nplt.show()\nprint(f\"Standard deviation of log residuals: {residuals.std():.3f}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-14T00:51:23.143163Z","iopub.status.idle":"2022-07-14T00:51:23.144051Z","shell.execute_reply.started":"2022-07-14T00:51:23.143816Z","shell.execute_reply":"2022-07-14T00:51:23.143837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}