{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-07T22:40:47.291359Z","iopub.execute_input":"2022-07-07T22:40:47.292537Z","iopub.status.idle":"2022-07-07T22:40:47.313115Z","shell.execute_reply.started":"2022-07-07T22:40:47.292472Z","shell.execute_reply":"2022-07-07T22:40:47.312058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nfrom xgboost import XGBClassifier\nfrom catboost import CatBoostRegressor\nfrom lightgbm import LGBMRegressor\nplt.style.use('ggplot')\nfrom sklearn import preprocessing\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neighbors import KNeighborsClassifier","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:47.546590Z","iopub.execute_input":"2022-07-07T22:40:47.547297Z","iopub.status.idle":"2022-07-07T22:40:47.555358Z","shell.execute_reply.started":"2022-07-07T22:40:47.547260Z","shell.execute_reply":"2022-07-07T22:40:47.554005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oil_data = pd.read_csv('../input/store-sales-time-series-forecasting/oil.csv')\noil_data","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:47.743726Z","iopub.execute_input":"2022-07-07T22:40:47.744130Z","iopub.status.idle":"2022-07-07T22:40:47.761110Z","shell.execute_reply.started":"2022-07-07T22:40:47.744099Z","shell.execute_reply":"2022-07-07T22:40:47.760069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oil_data.rename(columns={'dcoilwtico':'oilPrice'}, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:47.970180Z","iopub.execute_input":"2022-07-07T22:40:47.970627Z","iopub.status.idle":"2022-07-07T22:40:47.977472Z","shell.execute_reply.started":"2022-07-07T22:40:47.970588Z","shell.execute_reply":"2022-07-07T22:40:47.976165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oil_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:48.181337Z","iopub.execute_input":"2022-07-07T22:40:48.182532Z","iopub.status.idle":"2022-07-07T22:40:48.193784Z","shell.execute_reply.started":"2022-07-07T22:40:48.182485Z","shell.execute_reply":"2022-07-07T22:40:48.192900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oil_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:48.388394Z","iopub.execute_input":"2022-07-07T22:40:48.389645Z","iopub.status.idle":"2022-07-07T22:40:48.404407Z","shell.execute_reply.started":"2022-07-07T22:40:48.389587Z","shell.execute_reply":"2022-07-07T22:40:48.403366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Number of rows: {oil_data.shape[0]};  Number of columns: {oil_data.shape[1]}; No of missing values: {sum(oil_data.isna().sum())}')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:48.622344Z","iopub.execute_input":"2022-07-07T22:40:48.623650Z","iopub.status.idle":"2022-07-07T22:40:48.631935Z","shell.execute_reply.started":"2022-07-07T22:40:48.623577Z","shell.execute_reply":"2022-07-07T22:40:48.630822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number of missing Values in every column:')\nprint(oil_data.isna().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:48.930555Z","iopub.execute_input":"2022-07-07T22:40:48.931805Z","iopub.status.idle":"2022-07-07T22:40:48.941212Z","shell.execute_reply.started":"2022-07-07T22:40:48.931740Z","shell.execute_reply":"2022-07-07T22:40:48.940098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oil_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:49.268529Z","iopub.execute_input":"2022-07-07T22:40:49.269311Z","iopub.status.idle":"2022-07-07T22:40:49.283761Z","shell.execute_reply.started":"2022-07-07T22:40:49.269274Z","shell.execute_reply":"2022-07-07T22:40:49.282737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oil_data['oilPrice'] = oil_data['oilPrice'].replace(np.NaN, oil_data['oilPrice'].std())\nprint(oil_data['oilPrice'])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:49.707287Z","iopub.execute_input":"2022-07-07T22:40:49.707803Z","iopub.status.idle":"2022-07-07T22:40:49.720001Z","shell.execute_reply.started":"2022-07-07T22:40:49.707684Z","shell.execute_reply":"2022-07-07T22:40:49.718917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv('../input/store-sales-time-series-forecasting/sample_submission.csv')\nsample_submission","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:50.026247Z","iopub.execute_input":"2022-07-07T22:40:50.026662Z","iopub.status.idle":"2022-07-07T22:40:50.049130Z","shell.execute_reply.started":"2022-07-07T22:40:50.026631Z","shell.execute_reply":"2022-07-07T22:40:50.047911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:50.414499Z","iopub.execute_input":"2022-07-07T22:40:50.414948Z","iopub.status.idle":"2022-07-07T22:40:50.427468Z","shell.execute_reply.started":"2022-07-07T22:40:50.414911Z","shell.execute_reply":"2022-07-07T22:40:50.426487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Number of rows: {sample_submission.shape[0]};  Number of columns: {sample_submission.shape[1]}; No of missing values: {sum(sample_submission.isna().sum())}')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:50.648535Z","iopub.execute_input":"2022-07-07T22:40:50.649769Z","iopub.status.idle":"2022-07-07T22:40:50.656687Z","shell.execute_reply.started":"2022-07-07T22:40:50.649725Z","shell.execute_reply":"2022-07-07T22:40:50.655516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:50.847471Z","iopub.execute_input":"2022-07-07T22:40:50.848512Z","iopub.status.idle":"2022-07-07T22:40:50.868549Z","shell.execute_reply.started":"2022-07-07T22:40:50.848468Z","shell.execute_reply":"2022-07-07T22:40:50.867510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"holidays_events = pd.read_csv('../input/store-sales-time-series-forecasting/holidays_events.csv')\nholidays_events","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:51.139778Z","iopub.execute_input":"2022-07-07T22:40:51.140223Z","iopub.status.idle":"2022-07-07T22:40:51.161996Z","shell.execute_reply.started":"2022-07-07T22:40:51.140189Z","shell.execute_reply":"2022-07-07T22:40:51.161103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"holidays_events.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:51.255437Z","iopub.execute_input":"2022-07-07T22:40:51.255882Z","iopub.status.idle":"2022-07-07T22:40:51.273577Z","shell.execute_reply.started":"2022-07-07T22:40:51.255842Z","shell.execute_reply":"2022-07-07T22:40:51.272028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"holidays_events.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:51.460645Z","iopub.execute_input":"2022-07-07T22:40:51.461655Z","iopub.status.idle":"2022-07-07T22:40:51.475936Z","shell.execute_reply.started":"2022-07-07T22:40:51.461610Z","shell.execute_reply":"2022-07-07T22:40:51.475032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Number of rows: {holidays_events.shape[0]};  Number of columns: {holidays_events.shape[1]}; No of missing values: {sum(holidays_events.isna().sum())}')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:51.684587Z","iopub.execute_input":"2022-07-07T22:40:51.685909Z","iopub.status.idle":"2022-07-07T22:40:51.693434Z","shell.execute_reply.started":"2022-07-07T22:40:51.685864Z","shell.execute_reply":"2022-07-07T22:40:51.692209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number of missing Values in every column:')\nprint(holidays_events.isna().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:51.884537Z","iopub.execute_input":"2022-07-07T22:40:51.884991Z","iopub.status.idle":"2022-07-07T22:40:51.893549Z","shell.execute_reply.started":"2022-07-07T22:40:51.884956Z","shell.execute_reply":"2022-07-07T22:40:51.892608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"holidays_events.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:52.098729Z","iopub.execute_input":"2022-07-07T22:40:52.099191Z","iopub.status.idle":"2022-07-07T22:40:52.124984Z","shell.execute_reply.started":"2022-07-07T22:40:52.099140Z","shell.execute_reply":"2022-07-07T22:40:52.123704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stores_data = pd.read_csv('../input/store-sales-time-series-forecasting/stores.csv')\nstores_data","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:52.331371Z","iopub.execute_input":"2022-07-07T22:40:52.332257Z","iopub.status.idle":"2022-07-07T22:40:52.354264Z","shell.execute_reply.started":"2022-07-07T22:40:52.332211Z","shell.execute_reply":"2022-07-07T22:40:52.353019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stores_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:52.533375Z","iopub.execute_input":"2022-07-07T22:40:52.533806Z","iopub.status.idle":"2022-07-07T22:40:52.547444Z","shell.execute_reply.started":"2022-07-07T22:40:52.533771Z","shell.execute_reply":"2022-07-07T22:40:52.546221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stores_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:52.778445Z","iopub.execute_input":"2022-07-07T22:40:52.779484Z","iopub.status.idle":"2022-07-07T22:40:52.793323Z","shell.execute_reply.started":"2022-07-07T22:40:52.779434Z","shell.execute_reply":"2022-07-07T22:40:52.792015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Number of rows: {stores_data.shape[0]};  Number of columns: {stores_data.shape[1]}; No of missing values: {sum(stores_data.isna().sum())}')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:52.976526Z","iopub.execute_input":"2022-07-07T22:40:52.976943Z","iopub.status.idle":"2022-07-07T22:40:52.984879Z","shell.execute_reply.started":"2022-07-07T22:40:52.976912Z","shell.execute_reply":"2022-07-07T22:40:52.983693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stores_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:53.187637Z","iopub.execute_input":"2022-07-07T22:40:53.188628Z","iopub.status.idle":"2022-07-07T22:40:53.207560Z","shell.execute_reply.started":"2022-07-07T22:40:53.188575Z","shell.execute_reply":"2022-07-07T22:40:53.206407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv('../input/store-sales-time-series-forecasting/train.csv')\ntrain_data","metadata":{"execution":{"iopub.status.busy":"2022-07-07T22:40:53.450367Z","iopub.execute_input":"2022-07-07T22:40:53.451514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Number of rows: {train_data.shape[0]};  Number of columns: {train_data.shape[1]}; No of missing values: {sum(train_data.isna().sum())}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number of missing Values in every column:')\nprint(train_data.isna().sum())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv('../input/store-sales-time-series-forecasting/test.csv')\ntest_data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Number of rows: {test_data.shape[0]};  Number of columns: {test_data.shape[1]}; No of missing values: {sum(test_data.isna().sum())}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number of missing Values in every column:')\nprint(test_data.isna().sum())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transaction_data = pd.read_csv('../input/store-sales-time-series-forecasting/transactions.csv')\ntransaction_data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transaction_data.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transaction_data.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Number of rows: {transaction_data.shape[0]};  Number of columns: {transaction_data.shape[1]}; No of missing values: {sum(transaction_data.isna().sum())}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number of missing Values in every column:')\nprint(transaction_data.isna().sum())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transaction_data.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1 = pd.merge(holidays_events, train_data, \n                   on='date',\n                   how='inner')\n  \n# displaying result\nprint(train1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# using merge function by setting how='inner'\ntrain2 = pd.merge(stores_data,train1,\n                  on ='store_nbr',\n                  how='inner'\n                 )\nprint(train2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# using merge function by setting how='inner'\ntrain = pd.merge(train2, oil_data,\n                on='date',\n                how='inner')\n  \n# displaying result\nprint(train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(['locale','locale_name','description','transferred','type_y'], axis=1, inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['date'] = train['date'].str.replace('-','')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['date'] = train['date'].str.replace(' '' ','')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#label encoding\nfrom sklearn import preprocessing\n# label_encoder object knows how to understand word labels.\nlabel_encoder = preprocessing.LabelEncoder()\n  \n# Encode labels in column 'species'.\ntrain['date']= label_encoder.fit_transform(train['date'])\n  \ntrain['date'].unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#label encoding\nfrom sklearn import preprocessing\n# label_encoder object knows how to understand word labels.\nlabel_encoder = preprocessing.LabelEncoder()\n  \n# Encode labels in column 'species'.\ntrain['family']= label_encoder.fit_transform(train['family'])\n  \ntrain['family'].unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#label encoding\nfrom sklearn import preprocessing\n# label_encoder object knows how to understand word labels.\nlabel_encoder = preprocessing.LabelEncoder()\n  \n# Encode labels in column 'species'.\ntrain['city']= label_encoder.fit_transform(train['city'])\n  \ntrain['city'].unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#label encoding\nfrom sklearn import preprocessing\n# label_encoder object knows how to understand word labels.\nlabel_encoder = preprocessing.LabelEncoder()\n  \n# Encode labels in column 'species'.\ntrain['type_x']= label_encoder.fit_transform(train['type_x'])\n  \ntrain['type_x'].unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#label encoding\nfrom sklearn import preprocessing\n# label_encoder object knows how to understand word labels.\nlabel_encoder = preprocessing.LabelEncoder()\n  \n# Encode labels in column 'species'.\ntrain['state']= label_encoder.fit_transform(train['state'])\n  \ntrain['state'].unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#heat map\n# prints data that will be plotted\n# columns shown here are selected by corr() since\n# they are ideal for the plot\nprint(train.corr())\n  \n# plotting correlation heatmap\ndataplot = sns.heatmap(train.corr(), cmap=\"YlGnBu\", annot=True)\n  \n# displaying heatmap\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#output vector\ny = train.date\n \n#input vector\nx=train.drop('date',axis=1)\n \n#split\nx_train,x_test,y_train,y_test=train_test_split(x,y,test_size=0.2)\nfrom sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\ny_test = le.fit_transform(y_test)\n#verify\nprint(\"shape of original dataset :\", train.shape)\nprint(\"shape of input - training set\", x_train.shape)\nprint(\"shape of output - training set\", y_train.shape)\nprint(\"shape of input - testing set\", x_test.shape)\nprint(\"shape of output - testing set\", y_test.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBClassifier\nmodel_xgb_train = XGBClassifier()\nmodel_xgb_train.fit(x_train, y_train)\nmodel_xgb_train.score(x_test, y_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostRegressor\nmodel_cat_train = CatBoostRegressor(verbose=0, n_estimators=100)\nmodel_cat_train.fit(x_train, y_train)\nmodel_cat_train.score(x_test, y_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightgbm import LGBMRegressor\nmodel_lgb_train = LGBMRegressor()\nmodel_lgb_train.fit(x_train, y_train)\nmodel_lgb_train.score(x_test, y_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nmodel_RFC_train = RandomForestClassifier()\nmodel_RFC_train.fit(x_train, y_train)\nmodel_RFC_train.score(x_test, y_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\nmodel_DTC_train = DecisionTreeClassifier()\nmodel_DTC_train.fit(x_train, y_train)\nmodel_DTC_train.score(x_test, y_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nmodel_KN_train = KNeighborsClassifier()\nmodel_KN_train.fit(x_train, y_train)\nmodel_KN_train.score(x_test, y_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del holidays_events[\"date\"]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# using merge function by setting how='inner'\ntest1 = pd.merge(oil_data,test_data,\n                  on ='date',\n                  how='inner'\n                 )\nprint(test1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# using merge function by setting how='inner'\ntest2 = pd.merge(test1,stores_data,\n                  on ='store_nbr',\n                  how='inner'\n                 )\nprint(test2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.concat([test2, holidays_events])\nprint(test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.drop(['locale','locale_name','description'], axis=1, inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#label encoding\nfrom sklearn import preprocessing\n# label_encoder object knows how to understand word labels.\nlabel_encoder = preprocessing.LabelEncoder()\n  \n# Encode labels in column 'species'.\ntest['type']= label_encoder.fit_transform(test['type'])\n  \ntest['type'].unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#label encoding\nfrom sklearn import preprocessing\n# label_encoder object knows how to understand word labels.\nlabel_encoder = preprocessing.LabelEncoder()\n  \n# Encode labels in column 'species'.\ntest['family']= label_encoder.fit_transform(test['family'])\n  \ntest['family'].unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#label encoding\nfrom sklearn import preprocessing\n# label_encoder object knows how to understand word labels.\nlabel_encoder = preprocessing.LabelEncoder()\n  \n# Encode labels in column 'species'.\ntest['city']= label_encoder.fit_transform(test['city'])\n  \ntest['city'].unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#label encoding\nfrom sklearn import preprocessing\n# label_encoder object knows how to understand word labels.\nlabel_encoder = preprocessing.LabelEncoder()\n  \n# Encode labels in column 'species'.\ntest['state']= label_encoder.fit_transform(test['state'])\n  \ntest['state'].unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#label encoding\nfrom sklearn import preprocessing\n# label_encoder object knows how to understand word labels.\nlabel_encoder = preprocessing.LabelEncoder()\n  \n# Encode labels in column 'species'.\ntest['date']= label_encoder.fit_transform(test['date'])\n  \ntest['date'].unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#label encoding\nfrom sklearn import preprocessing\n# label_encoder object knows how to understand word labels.\nlabel_encoder = preprocessing.LabelEncoder()\n  \n# Encode labels in column 'species'.\ntest['transferred']= label_encoder.fit_transform(test['transferred'])\n  \ntest['transferred'].unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number of missing Values in every column:')\nprint(test.isna().sum())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['oilPrice'] = test['oilPrice'].replace(np.NaN, test['oilPrice'].std())\nprint(test['oilPrice'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['id'] = test['id'].replace(np.NaN, test['id'].std())\nprint(test['id'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['store_nbr'] = test['store_nbr'].replace(np.NaN, test['store_nbr'].std())\nprint(test['store_nbr'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['onpromotion'] = test['onpromotion'].replace(np.NaN, test['onpromotion'].std())\nprint(test['onpromotion'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['cluster'] = test['cluster'].replace(np.NaN, test['cluster'].std())\nprint(test['cluster'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#output vector\np = test.date.values\n \n#input vector\nq = test.drop('date',axis=1).values\n \n#split\np_train,p_test,q_train,q_test=train_test_split(p,q,test_size=0.2)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(p_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(q_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(p_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(q_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.family = le.fit_transform(test.family)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = test.drop([\"date\"], axis=1).values\nX.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X[0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model_xgb_train.predict(X)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['sales'] = predictions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model_cat_train.predict(X)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['sales'] = predictions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model_lgb_train.predict(X)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['sales'] = predictions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model_RFC_train.predict(X)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['sales'] = predictions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model_DTC_train.predict(X)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['sales'] = predictions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model_KN_train.predict(X)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['sales'] = predictions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}