{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Importing neccasary Libraries \nimport pandas as pd\nimport numpy as np \n%pip install upgini ","metadata":{"execution":{"iopub.status.busy":"2022-08-09T13:40:31.859903Z","iopub.execute_input":"2022-08-09T13:40:31.860697Z","iopub.status.idle":"2022-08-09T13:40:42.684572Z","shell.execute_reply.started":"2022-08-09T13:40:31.860613Z","shell.execute_reply":"2022-08-09T13:40:42.683606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf = pd.read_csv('../input/demand-forecasting-kernels-only/train.csv')\ndf = df.sample(n = 19_000, random_state = 0)\n\ndf.info()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-09T13:40:42.689790Z","iopub.execute_input":"2022-08-09T13:40:42.690100Z","iopub.status.idle":"2022-08-09T13:40:43.038693Z","shell.execute_reply.started":"2022-08-09T13:40:42.690063Z","shell.execute_reply":"2022-08-09T13:40:43.037506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['store'] = df['store'].astype(str)\ndf['item'] = df['item'].astype(str)\ndf['date'] = pd.to_datetime(df['date'])\n\ndf.sort_values('date', inplace =True)\ndf.reset_index(inplace = True, drop = True)\ndf.head() ","metadata":{"execution":{"iopub.status.busy":"2022-08-09T13:40:43.040269Z","iopub.execute_input":"2022-08-09T13:40:43.040745Z","iopub.status.idle":"2022-08-09T13:40:43.097747Z","shell.execute_reply.started":"2022-08-09T13:40:43.040694Z","shell.execute_reply":"2022-08-09T13:40:43.096596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T13:40:43.100646Z","iopub.execute_input":"2022-08-09T13:40:43.101010Z","iopub.status.idle":"2022-08-09T13:40:43.120215Z","shell.execute_reply.started":"2022-08-09T13:40:43.100978Z","shell.execute_reply":"2022-08-09T13:40:43.118448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# putting data from 2013 till 2016 in the training dataset then 2017 data in the test data set \ntrain = df[df['date'] < '2017-1-1']\ntest = df[df['date'] > '2016-12-31']\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-08-09T13:40:43.121644Z","iopub.execute_input":"2022-08-09T13:40:43.121987Z","iopub.status.idle":"2022-08-09T13:40:43.141544Z","shell.execute_reply.started":"2022-08-09T13:40:43.121956Z","shell.execute_reply":"2022-08-09T13:40:43.140726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Splitting test and train dataset into feautures and labels \n\ntrain_features = train.drop(columns = ['sales'], axis = 1)\ntrain_target = train['sales']\n\ntest_features = test.drop(columns = ['sales'], axis = 1)\ntest_target = test['sales']","metadata":{"execution":{"iopub.status.busy":"2022-08-09T13:40:43.142812Z","iopub.execute_input":"2022-08-09T13:40:43.143293Z","iopub.status.idle":"2022-08-09T13:40:43.151419Z","shell.execute_reply.started":"2022-08-09T13:40:43.143261Z","shell.execute_reply":"2022-08-09T13:40:43.150073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using upgini python library that is used to enrivh yout Dataset by searching through \n# thousands of public and community data sources and then creates features that enrich\n# the capabilities of our models. \n\nfrom upgini import  FeaturesEnricher, SearchKey \nfrom upgini.metadata import CVType\n\nenricher = FeaturesEnricher(\n        search_keys = {\n        'date' : SearchKey.DATE,\n        },\n    cv = CVType.time_series \n)\n\nenricher.fit(train_features,\n            train_target, \n            eval_set = [(test_features, test_target)])","metadata":{"execution":{"iopub.status.busy":"2022-08-09T14:43:19.697989Z","iopub.execute_input":"2022-08-09T14:43:19.698464Z","iopub.status.idle":"2022-08-09T14:45:47.387275Z","shell.execute_reply.started":"2022-08-09T14:43:19.698422Z","shell.execute_reply":"2022-08-09T14:45:47.385909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using CatBOost Regressor as our model and comparing the error for the noraml \n# Dataset to the enriched Dataset \n\nfrom catboost import CatBoostRegressor\nfrom catboost.utils import eval_metric\n\nmodel = CatBoostRegressor(verbose = False, allow_writing_files = False, random_state = 0)\n\nenricher.calculate_metrics(\n    train_features, train_target, \n    eval_set = [( test_features, test_target)],\n    estimator = model,\n    scoring = 'mean_absolute_percentage_error'\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T14:46:20.495641Z","iopub.execute_input":"2022-08-09T14:46:20.496800Z","iopub.status.idle":"2022-08-09T14:47:28.628202Z","shell.execute_reply.started":"2022-08-09T14:46:20.496719Z","shell.execute_reply":"2022-08-09T14:47:28.626826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adding the enriched Dataset to the normal Dataset \n\nenriched_train_features = enricher.transform(train_features, keep_input = True)\nenriched_test_features = enricher.transform(test_features, keep_input = True)\n\nenriched_train_features.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T14:48:46.560048Z","iopub.execute_input":"2022-08-09T14:48:46.560605Z","iopub.status.idle":"2022-08-09T14:50:14.453221Z","shell.execute_reply.started":"2022-08-09T14:48:46.560556Z","shell.execute_reply":"2022-08-09T14:50:14.452020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training models on the normal dataset to find its accuracy using the SMAPE metric\n# Symmetric mean absolute percentage error ( SMAPE )\n\n\nmodel.fit(train_features, train_target)\npreds = model.predict(test_features)\n\neval_metric(test_target.values, preds, 'SMAPE')","metadata":{"execution":{"iopub.status.busy":"2022-08-09T14:51:51.165308Z","iopub.execute_input":"2022-08-09T14:51:51.165821Z","iopub.status.idle":"2022-08-09T14:51:53.667558Z","shell.execute_reply.started":"2022-08-09T14:51:51.165781Z","shell.execute_reply":"2022-08-09T14:51:53.666252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training our enriched Data into the model to find its accuracy using the SMAPE metric  \nmodel.fit(enriched_train_features,train_target)\nenriched_preds = model.predict(enriched_test_features)\n\neval_metric(test_target.values, enriched_preds, 'SMAPE')","metadata":{"execution":{"iopub.status.busy":"2022-08-09T14:51:56.727263Z","iopub.execute_input":"2022-08-09T14:51:56.727783Z","iopub.status.idle":"2022-08-09T14:52:02.924876Z","shell.execute_reply.started":"2022-08-09T14:51:56.727742Z","shell.execute_reply":"2022-08-09T14:52:02.923592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A significant Decrese in the error metric from 37.6% to 14.7% (More can be done too)","metadata":{}}]}