{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-22T07:05:02.938521Z","iopub.execute_input":"2022-07-22T07:05:02.940873Z","iopub.status.idle":"2022-07-22T07:05:02.991451Z","shell.execute_reply.started":"2022-07-22T07:05:02.940663Z","shell.execute_reply":"2022-07-22T07:05:02.990117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#set up code checking\nfrom learntools.core import binder\nbinder.bind(globals())\nfrom learntools.machine_learning.ex7 import *","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:05:02.993810Z","iopub.execute_input":"2022-07-22T07:05:02.994182Z","iopub.status.idle":"2022-07-22T07:05:03.043636Z","shell.execute_reply.started":"2022-07-22T07:05:02.994149Z","shell.execute_reply":"2022-07-22T07:05:03.042552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Import helpful Libraries\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:05:03.044948Z","iopub.execute_input":"2022-07-22T07:05:03.045656Z","iopub.status.idle":"2022-07-22T07:05:04.649791Z","shell.execute_reply.started":"2022-07-22T07:05:03.045613Z","shell.execute_reply":"2022-07-22T07:05:04.648533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Load data\nholidays=pd.read_csv('../input/store-sales-time-series-forecasting/holidays_events.csv',index_col=None, header=0, parse_dates=['date'])\noil=pd.read_csv(\"../input/store-sales-time-series-forecasting/oil.csv\",index_col=None, header=0, parse_dates=['date'])\nstores=pd.read_csv(\"../input/store-sales-time-series-forecasting/stores.csv\")\ntransactions=pd.read_csv(\"../input/store-sales-time-series-forecasting/transactions.csv\",index_col=None, header=0, parse_dates=['date'])\ntrain=pd.read_csv(\"../input/store-sales-time-series-forecasting/train.csv\",index_col=None, header=0, parse_dates=['date'])\ndata=train.merge(holidays, on=['date'], how='inner').merge(oil,on=['date'],how='inner').merge(stores,on=['store_nbr'],how='inner').merge(transactions,on=['date'],how='inner')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:05:04.652457Z","iopub.execute_input":"2022-07-22T07:05:04.652901Z","iopub.status.idle":"2022-07-22T07:05:14.967390Z","shell.execute_reply.started":"2022-07-22T07:05:04.652862Z","shell.execute_reply":"2022-07-22T07:05:14.966285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create X and y\nX=data.drop(['date'],axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:05:14.969008Z","iopub.execute_input":"2022-07-22T07:05:14.969484Z","iopub.status.idle":"2022-07-22T07:05:24.857274Z","shell.execute_reply.started":"2022-07-22T07:05:14.969438Z","shell.execute_reply":"2022-07-22T07:05:24.856035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fill missing value using Extension to Imutation approach\nfrom sklearn.impute import SimpleImputer\n#Make copy to avoid changine original data\nX_plus=X.copy()\nX_obj=X_plus[['type_x','type_y','locale','locale_name','description','transferred','dcoilwtico',\n                       'city','state','family']]\nX_plus_num=X_plus.drop(X_obj,axis=1)\n\n#Make new columns indicating what will be imputed\ncols_with_missing=[col for col in X_plus_num.columns\n                   if X_plus_num[col].isnull().any()]\nfor col in cols_with_missing:\n    X_plus_num[col+'_was_missing']=X_plus_num[col].isnull()\n  \n\n#Imputation\nmy_imputer=SimpleImputer()\nX_imputed=pd.DataFrame(my_imputer.fit_transform(X_plus_num))\n\n#Add object columns\nX_new=pd.concat([X_plus,X_obj])\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:05:24.858690Z","iopub.execute_input":"2022-07-22T07:05:24.859037Z","iopub.status.idle":"2022-07-22T07:05:46.309436Z","shell.execute_reply.started":"2022-07-22T07:05:24.859005Z","shell.execute_reply":"2022-07-22T07:05:46.308302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Get list of categorical variables\ns=(X_new.dtypes=='object')\nobject_cols=list(s[s].index)\nprint(\"Categorical variables:\")\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:05:46.311069Z","iopub.execute_input":"2022-07-22T07:05:46.312319Z","iopub.status.idle":"2022-07-22T07:05:46.321658Z","shell.execute_reply.started":"2022-07-22T07:05:46.312270Z","shell.execute_reply":"2022-07-22T07:05:46.320255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Keep columns with few category\nlow_cardinality_cols = [cname for cname in X_new.columns if X_new[cname].nunique() < 10 and \n                        X_new[cname].dtype == \"object\"]\nnumerical_cols = [cname for cname in X_new.columns if X_new[cname].dtype in ['int64', 'float64']]\nmy_cols = low_cardinality_cols + numerical_cols\nX_new_1=X_new[my_cols].copy()\nX_new_1.tail(15)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:05:46.323335Z","iopub.execute_input":"2022-07-22T07:05:46.323756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Above DataFrame does not contain column name. Return Column names\ns=(X_new_1.dtypes=='object')\nobject_cols=list(s[s].index)\nprint('Categorical varialbes:')\nprint(object_cols)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Change Categorical Variables\nX_new_2=X_new_1.copy()\nfrom sklearn.preprocessing import OneHotEncoder\nOH_encoder=OneHotEncoder(handle_unknown='ignore',sparse=False)\nOH_cols_X=pd.DataFrame(OH_encoder.fit_transform(X_new_1[object_cols]))\n#One-Hot encoding removes index. Put it back.\nOH_cols_X.index = X_new_1.index\n#Remove categorical columns(later replace with one-hot encoding)\nnum_X=X_new_1.drop(object_cols,axis=1)\n#Add one-hot encoded columns to numerical features\nX_data=pd.concat([num_X,OH_cols_X],axis=1)\nX_data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Split into validation and trainging data\ny=X_data.sales\nX_1=X_data.drop(['sales'],axis=1)\ntrain_X,val_X,train_y,val_y=train_test_split(X_1,y,train_size=0.8,test_size=0.2,random_state=1)\ntrain_X.head()\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Define Models\nfrom sklearn.ensemble import RandomForestRegressor\nmodel_1=RandomForestRegressor(n_estimators=50,random_state=0)\nmodel_2=RandomForestRegressor(n_estimators=100,random_state=0)\nmodel_3=RandomForestRegressor(n_estimators=100,criterion='absolute_error',random_state=0)\nmodel_4=RandomForestRegressor(n_estimators=200,min_samples_split=20,random_state=0)\nmodel_5=RandomForestRegressor(n_estimators=100,max_depth=7,random_state=0)\nmodels=[model_1,model_2,model_3,model_4,model_5]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#To Select best model\nfrom sklearn.metrics import mean_absolute_error\n#function for comparing different models\ndef score_model(model,X_t=train_X, X_v=val_X,y_t=train_y,y_v=val_y):\n    model.fit(X_t,y_t)\n    preds=model.predict(X_v)\n    return mean_absolute_error(y_v,preds)\nfor i in range(0,len(models)):\n    mae=score_model(models[i])\n    print(\"Model%dMAE%d\"%(i+1,mae))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Choose the Model\nmy_model=RandomForestRegressor(n_estimators=100,criterion='absolute_error',random_state=0)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test=pd.read_csv('../input/store-sales-time-series-forecasting/test.csv').drop(['date'],axis=1)\nX_final=X_data(['id','date','store_nbr','family','onpromotion'])","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:09:16.139253Z","iopub.execute_input":"2022-07-22T07:09:16.139769Z","iopub.status.idle":"2022-07-22T07:09:16.240186Z","shell.execute_reply.started":"2022-07-22T07:09:16.139647Z","shell.execute_reply":"2022-07-22T07:09:16.239000Z"},"trusted":true},"execution_count":null,"outputs":[]}]}