{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-25T08:22:15.582547Z","iopub.execute_input":"2022-07-25T08:22:15.583665Z","iopub.status.idle":"2022-07-25T08:22:15.615205Z","shell.execute_reply.started":"2022-07-25T08:22:15.583557Z","shell.execute_reply":"2022-07-25T08:22:15.613999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Import helpful Libraries\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:23:36.819441Z","iopub.execute_input":"2022-07-25T08:23:36.819841Z","iopub.status.idle":"2022-07-25T08:23:37.595183Z","shell.execute_reply.started":"2022-07-25T08:23:36.819809Z","shell.execute_reply":"2022-07-25T08:23:37.593723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #Load data\n# holidays=pd.read_csv('../input/store-sales-time-series-forecasting/holidays_events.csv',index_col=None, header=0, parse_dates=['date'])\n# oil=pd.read_csv(\"../input/store-sales-time-series-forecasting/oil.csv\",index_col=None, header=0, parse_dates=['date'])\n# stores=pd.read_csv(\"../input/store-sales-time-series-forecasting/stores.csv\")\n# transactions=pd.read_csv(\"../input/store-sales-time-series-forecasting/transactions.csv\",index_col=None, header=0, parse_dates=['date'])\n# train=pd.read_csv(\"../input/store-sales-time-series-forecasting/train.csv\",index_col=None, header=0, parse_dates=['date'])\n# data=train.merge(holidays, on=['date'], how='inner').merge(oil,on=['date'],how='inner').merge(stores,on=['store_nbr'],how='inner').merge(transactions,on=['date'],how='inner')\n# data.tail(15)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:24:14.588036Z","iopub.execute_input":"2022-07-25T08:24:14.589224Z","iopub.status.idle":"2022-07-25T08:24:22.250074Z","shell.execute_reply.started":"2022-07-25T08:24:14.589168Z","shell.execute_reply":"2022-07-25T08:24:22.248987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the data\nX_full = pd.read_csv('../input/store-sales-time-series-forecasting/train.csv',index_col=\"id\")\nX_test_full = pd.read_csv('../input/store-sales-time-series-forecasting/test.csv', index_col='id')\n\n# Remove rows with missing target, separate target from predictors\nX_full.dropna(axis=0, subset=['sales'], inplace=True)\ny = X_full.sales\nX_full.drop(['sales'], axis=1, inplace=True)\n\n# Break off validation set from training data\nX_train_full, X_valid_full, y_train, y_valid = train_test_split(X_full, y, \n                                                                train_size=0.8, test_size=0.2,\n                                                                random_state=0)\n\n# \"Cardinality\" means the number of unique values in a column\n# Select categorical columns with relatively low cardinality (convenient but arbitrary)\ncategorical_cols = [cname for cname in X_train_full.columns if\n                    X_train_full[cname].nunique() < 20 and \n                    X_train_full[cname].dtype == \"object\"]\n\n# Select numerical columns\nnumerical_cols = [cname for cname in X_train_full.columns if \n                X_train_full[cname].dtype in ['int64', 'float64']]\n\n# Keep selected columns only\nmy_cols = categorical_cols + numerical_cols\nX_train = X_train_full[my_cols].copy()\nX_valid = X_valid_full[my_cols].copy()\nX_test = X_test_full[my_cols].copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:53:15.013671Z","iopub.execute_input":"2022-07-25T08:53:15.014839Z","iopub.status.idle":"2022-07-25T08:53:23.512683Z","shell.execute_reply.started":"2022-07-25T08:53:15.014791Z","shell.execute_reply":"2022-07-25T08:53:23.511879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:53:25.680151Z","iopub.execute_input":"2022-07-25T08:53:25.680582Z","iopub.status.idle":"2022-07-25T08:53:25.691439Z","shell.execute_reply.started":"2022-07-25T08:53:25.680551Z","shell.execute_reply":"2022-07-25T08:53:25.690191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_absolute_error\n\n# Preprocessing for numerical data\nnumerical_transformer = SimpleImputer(strategy='constant')\n\n# Preprocessing for categorical data\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n\n# Bundle preprocessing for numerical and categorical data\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('cat', categorical_transformer, categorical_cols)\n    ])","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:53:30.260192Z","iopub.execute_input":"2022-07-25T08:53:30.260588Z","iopub.status.idle":"2022-07-25T08:53:30.269642Z","shell.execute_reply.started":"2022-07-25T08:53:30.260556Z","shell.execute_reply":"2022-07-25T08:53:30.268657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define model\nmodel = RandomForestRegressor(n_estimators=50, random_state=0)\n\n# Bundle preprocessing and modeling code in a pipeline\nclf = Pipeline(steps=[('preprocessor', preprocessor),\n                      ('model', model)\n                     ])\n\n# Preprocessing of training data, fit model \nclf.fit(X_train, y_train)\n\n# Preprocessing of validation data, get predictions\npreds = clf.predict(X_valid)\n\nprint('MAE:', mean_absolute_error(y_valid, preds))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:53:39.119864Z","iopub.execute_input":"2022-07-25T08:53:39.120249Z","iopub.status.idle":"2022-07-25T08:55:03.051225Z","shell.execute_reply.started":"2022-07-25T08:53:39.120221Z","shell.execute_reply":"2022-07-25T08:55:03.050079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Bundle preprocessing and modeling code in a pipeline\nmy_pipeline = Pipeline(steps=[('preprocessor', preprocessor),\n                              ('model', model)\n                             ])\n\n# Preprocessing of training data, fit model \nmy_pipeline.fit(X_train, y_train)\n\n# Preprocessing of validation data, get predictions\npreds = my_pipeline.predict(X_valid)\n\n# Evaluate the model\nscore = mean_absolute_error(y_valid, preds)\nprint('MAE:', score)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:57:35.582603Z","iopub.execute_input":"2022-07-25T08:57:35.583070Z","iopub.status.idle":"2022-07-25T08:58:58.788942Z","shell.execute_reply.started":"2022-07-25T08:57:35.583037Z","shell.execute_reply":"2022-07-25T08:58:58.787839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocessing of test data, fit model\npreds_test = my_pipeline.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:59:06.316102Z","iopub.execute_input":"2022-07-25T08:59:06.316520Z","iopub.status.idle":"2022-07-25T08:59:06.470809Z","shell.execute_reply.started":"2022-07-25T08:59:06.316487Z","shell.execute_reply":"2022-07-25T08:59:06.469538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save test predictions to file\noutput = pd.DataFrame({'Id': X_test.index,\n                       'sales': preds_test})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:59:11.071511Z","iopub.execute_input":"2022-07-25T08:59:11.071866Z","iopub.status.idle":"2022-07-25T08:59:11.180566Z","shell.execute_reply.started":"2022-07-25T08:59:11.071838Z","shell.execute_reply":"2022-07-25T08:59:11.179415Z"},"trusted":true},"execution_count":null,"outputs":[]}]}