{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport scipy.stats\nfrom sklearn.preprocessing import LabelEncoder \nlabel_encoder= LabelEncoder()\n\nimport seaborn as sns\n\nfrom matplotlib import cm\nfrom matplotlib.colors import ListedColormap, LinearSegmentedColormap\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-30T13:59:53.874703Z","iopub.execute_input":"2022-07-30T13:59:53.875296Z","iopub.status.idle":"2022-07-30T13:59:55.145349Z","shell.execute_reply.started":"2022-07-30T13:59:53.875212Z","shell.execute_reply":"2022-07-30T13:59:55.144318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Explore Data from item_categories.csv, and let's see what informations we will gain****","metadata":{}},{"cell_type":"code","source":"item_categ = pd.read_csv(\"../input/competitive-data-science-predict-future-sales/item_categories.csv\")\nitem_categ.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:59:55.146871Z","iopub.execute_input":"2022-07-30T13:59:55.147107Z","iopub.status.idle":"2022-07-30T13:59:55.178891Z","shell.execute_reply.started":"2022-07-30T13:59:55.147084Z","shell.execute_reply":"2022-07-30T13:59:55.177923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item_categ['item_category_name'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:59:55.180371Z","iopub.execute_input":"2022-07-30T13:59:55.181467Z","iopub.status.idle":"2022-07-30T13:59:55.195371Z","shell.execute_reply.started":"2022-07-30T13:59:55.181436Z","shell.execute_reply":"2022-07-30T13:59:55.194404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item_categ['item_category_name'].nunique() #let's see how many kind of items in dataset","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:59:55.197794Z","iopub.execute_input":"2022-07-30T13:59:55.198083Z","iopub.status.idle":"2022-07-30T13:59:55.208009Z","shell.execute_reply.started":"2022-07-30T13:59:55.198047Z","shell.execute_reply":"2022-07-30T13:59:55.206995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#since the item names are not in English, so I transform them into numerical category using label encoder\nitem_categ['item_category_name']= label_encoder.fit_transform(item_categ['item_category_name'])","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:59:55.209367Z","iopub.execute_input":"2022-07-30T13:59:55.209764Z","iopub.status.idle":"2022-07-30T13:59:55.218746Z","shell.execute_reply.started":"2022-07-30T13:59:55.209729Z","shell.execute_reply":"2022-07-30T13:59:55.217608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X= item_categ.iloc[:,[0,1]].values","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:59:55.219839Z","iopub.execute_input":"2022-07-30T13:59:55.220099Z","iopub.status.idle":"2022-07-30T13:59:55.232804Z","shell.execute_reply.started":"2022-07-30T13:59:55.220072Z","shell.execute_reply":"2022-07-30T13:59:55.232064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ****EDA for items.csv****","metadata":{}},{"cell_type":"code","source":"item = pd.read_csv(\"../input/competitive-data-science-predict-future-sales/items.csv\")\nitem.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:59:55.234105Z","iopub.execute_input":"2022-07-30T13:59:55.234807Z","iopub.status.idle":"2022-07-30T13:59:55.303960Z","shell.execute_reply.started":"2022-07-30T13:59:55.234767Z","shell.execute_reply":"2022-07-30T13:59:55.303250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item['item_name'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:59:55.305230Z","iopub.execute_input":"2022-07-30T13:59:55.306042Z","iopub.status.idle":"2022-07-30T13:59:55.324167Z","shell.execute_reply.started":"2022-07-30T13:59:55.306011Z","shell.execute_reply":"2022-07-30T13:59:55.323182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item['item_name'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:59:55.325396Z","iopub.execute_input":"2022-07-30T13:59:55.326189Z","iopub.status.idle":"2022-07-30T13:59:55.349869Z","shell.execute_reply.started":"2022-07-30T13:59:55.326148Z","shell.execute_reply":"2022-07-30T13:59:55.348588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item['item_name'].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:59:55.353817Z","iopub.execute_input":"2022-07-30T13:59:55.354628Z","iopub.status.idle":"2022-07-30T13:59:55.372724Z","shell.execute_reply.started":"2022-07-30T13:59:55.354582Z","shell.execute_reply":"2022-07-30T13:59:55.371908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item['item_name'] = label_encoder.fit_transform(item['item_name'])\nitem.tail()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:59:55.374228Z","iopub.execute_input":"2022-07-30T13:59:55.375303Z","iopub.status.idle":"2022-07-30T13:59:55.436763Z","shell.execute_reply.started":"2022-07-30T13:59:55.375260Z","shell.execute_reply":"2022-07-30T13:59:55.436193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item['item_category_id'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:59:55.437805Z","iopub.execute_input":"2022-07-30T13:59:55.438202Z","iopub.status.idle":"2022-07-30T13:59:55.444697Z","shell.execute_reply.started":"2022-07-30T13:59:55.438179Z","shell.execute_reply":"2022-07-30T13:59:55.443911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item['item_category_id'].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:59:55.446681Z","iopub.execute_input":"2022-07-30T13:59:55.447192Z","iopub.status.idle":"2022-07-30T13:59:55.455647Z","shell.execute_reply.started":"2022-07-30T13:59:55.447145Z","shell.execute_reply":"2022-07-30T13:59:55.454828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_ = item.iloc[:,[1,2]].values\nX_","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:59:55.457152Z","iopub.execute_input":"2022-07-30T13:59:55.457427Z","iopub.status.idle":"2022-07-30T13:59:55.466889Z","shell.execute_reply.started":"2022-07-30T13:59:55.457400Z","shell.execute_reply":"2022-07-30T13:59:55.466026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****I will cluster every item based on its category. I use K-Means clustering to cluster the data.\nas we know There are 1-84 categories of all 22169 items****","metadata":{}},{"cell_type":"markdown","source":"****first thing first, I use Elbow Method to determine the number of cluster****","metadata":{}},{"cell_type":"code","source":"from sklearn.cluster import KMeans\nwcss = []\nfor i in range(1, 11):\n    kmeans = KMeans(n_clusters = i, init = 'k-means++', random_state = 42)\n    kmeans.fit(X)\n    wcss.append(kmeans.inertia_)\nplt.plot(range(1, 11), wcss)\nplt.title('The Elbow Method')\nplt.xlabel('Number of clusters')\nplt.ylabel('WCSS')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:59:55.468045Z","iopub.execute_input":"2022-07-30T13:59:55.468339Z","iopub.status.idle":"2022-07-30T13:59:56.317664Z","shell.execute_reply.started":"2022-07-30T13:59:55.468312Z","shell.execute_reply":"2022-07-30T13:59:56.316757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"from the graph of elbow method, we can see the number of cluster is 3","metadata":{}},{"cell_type":"code","source":"kmeans = KMeans(n_clusters = 3, init = 'k-means++', random_state = 42)\ny_kmeans = kmeans.fit_predict(X_)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:59:56.319095Z","iopub.execute_input":"2022-07-30T13:59:56.319549Z","iopub.status.idle":"2022-07-30T13:59:57.879707Z","shell.execute_reply.started":"2022-07-30T13:59:56.319519Z","shell.execute_reply":"2022-07-30T13:59:57.878681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.scatter(X_[y_kmeans == 0, 0], X_[y_kmeans == 0, 1], s = 100, c = 'red', label = 'Cluster 1')\nplt.scatter(X_[y_kmeans == 1, 0], X_[y_kmeans == 1, 1], s = 100, c = 'blue', label = 'Cluster 2')\nplt.scatter(X_[y_kmeans == 2, 0], X_[y_kmeans == 2, 1], s = 100, c = 'green', label = 'Cluster 3')\nplt.scatter(kmeans.cluster_centers_[:, 0], kmeans.cluster_centers_[:, 1], s = 300, c = 'yellow', label = 'Centroids')\nplt.title('Clusters of items')\nplt.xlabel('item_id')\nplt.ylabel('item_category_id')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:59:57.881285Z","iopub.execute_input":"2022-07-30T13:59:57.881976Z","iopub.status.idle":"2022-07-30T13:59:59.021572Z","shell.execute_reply.started":"2022-07-30T13:59:57.881938Z","shell.execute_reply":"2022-07-30T13:59:59.020561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **EDA for Sales_train.csv**","metadata":{}},{"cell_type":"code","source":"Sales_train = pd.read_csv(\"../input/competitive-data-science-predict-future-sales/sales_train.csv\")\nSales_train.tail()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:59:59.023059Z","iopub.execute_input":"2022-07-30T13:59:59.024089Z","iopub.status.idle":"2022-07-30T14:00:01.963697Z","shell.execute_reply.started":"2022-07-30T13:59:59.024048Z","shell.execute_reply":"2022-07-30T14:00:01.962783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Sales_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:00:01.967128Z","iopub.execute_input":"2022-07-30T14:00:01.967425Z","iopub.status.idle":"2022-07-30T14:00:01.979824Z","shell.execute_reply.started":"2022-07-30T14:00:01.967397Z","shell.execute_reply":"2022-07-30T14:00:01.979167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Sales_train.isna().count","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:00:01.980765Z","iopub.execute_input":"2022-07-30T14:00:01.981335Z","iopub.status.idle":"2022-07-30T14:00:02.293377Z","shell.execute_reply.started":"2022-07-30T14:00:01.981309Z","shell.execute_reply":"2022-07-30T14:00:02.291962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Sales_train['shop_id'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:00:02.295575Z","iopub.execute_input":"2022-07-30T14:00:02.295969Z","iopub.status.idle":"2022-07-30T14:00:02.317817Z","shell.execute_reply.started":"2022-07-30T14:00:02.295935Z","shell.execute_reply":"2022-07-30T14:00:02.316945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Sales_train['shop_id'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:00:02.319082Z","iopub.execute_input":"2022-07-30T14:00:02.319662Z","iopub.status.idle":"2022-07-30T14:00:02.379397Z","shell.execute_reply.started":"2022-07-30T14:00:02.319620Z","shell.execute_reply":"2022-07-30T14:00:02.377676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Sales_train[Sales_train['shop_id']== 25]['date']","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:00:02.383374Z","iopub.execute_input":"2022-07-30T14:00:02.383694Z","iopub.status.idle":"2022-07-30T14:00:02.411549Z","shell.execute_reply.started":"2022-07-30T14:00:02.383665Z","shell.execute_reply":"2022-07-30T14:00:02.410383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shops=pd.read_csv(\"../input/competitive-data-science-predict-future-sales/shops.csv\")\nshops.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:00:02.412905Z","iopub.execute_input":"2022-07-30T14:00:02.413226Z","iopub.status.idle":"2022-07-30T14:00:02.430319Z","shell.execute_reply.started":"2022-07-30T14:00:02.413197Z","shell.execute_reply":"2022-07-30T14:00:02.429038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test=pd.read_csv(\"../input/competitive-data-science-predict-future-sales/test.csv\")\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:00:02.432013Z","iopub.execute_input":"2022-07-30T14:00:02.433016Z","iopub.status.idle":"2022-07-30T14:00:02.560369Z","shell.execute_reply.started":"2022-07-30T14:00:02.432977Z","shell.execute_reply":"2022-07-30T14:00:02.559555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['item_id'].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:00:02.561507Z","iopub.execute_input":"2022-07-30T14:00:02.561852Z","iopub.status.idle":"2022-07-30T14:00:02.572308Z","shell.execute_reply.started":"2022-07-30T14:00:02.561823Z","shell.execute_reply":"2022-07-30T14:00:02.571066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Predict the item price of test.csv based on the sales_train.csv  **","metadata":{}},{"cell_type":"code","source":"#Let's generate the model from the data train first\n#I use multiple linear regression\n\nX_Train=Sales_train.iloc[:,[2,3]]\nY_train=Sales_train.iloc[:,[4]]\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X_Train, Y_train, test_size = 0.2, random_state = 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:00:02.575204Z","iopub.execute_input":"2022-07-30T14:00:02.575904Z","iopub.status.idle":"2022-07-30T14:00:04.093261Z","shell.execute_reply.started":"2022-07-30T14:00:02.575874Z","shell.execute_reply":"2022-07-30T14:00:04.092312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nregressor = LinearRegression()\nregressor.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:00:04.100069Z","iopub.execute_input":"2022-07-30T14:00:04.101831Z","iopub.status.idle":"2022-07-30T14:00:04.247809Z","shell.execute_reply.started":"2022-07-30T14:00:04.101799Z","shell.execute_reply":"2022-07-30T14:00:04.246689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = regressor.predict(X_test)\ny_pred\n(y_pred,y_test.values)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:00:04.248883Z","iopub.execute_input":"2022-07-30T14:00:04.249131Z","iopub.status.idle":"2022-07-30T14:00:04.282320Z","shell.execute_reply.started":"2022-07-30T14:00:04.249103Z","shell.execute_reply":"2022-07-30T14:00:04.281456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Okay, Let's Execute the test.csv to predict the price of item using the model that we have generate from sales_train data****","metadata":{}},{"cell_type":"code","source":"X_test_test=test.iloc[:,[1,2]]","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:00:04.287232Z","iopub.execute_input":"2022-07-30T14:00:04.290291Z","iopub.status.idle":"2022-07-30T14:00:04.301903Z","shell.execute_reply.started":"2022-07-30T14:00:04.290245Z","shell.execute_reply":"2022-07-30T14:00:04.300958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_final = regressor.predict(X_test_test)\ny_pred_final\n\nprint(\"the price prediction of the test data\")\npd.DataFrame(y_pred_final)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:00:04.308139Z","iopub.execute_input":"2022-07-30T14:00:04.311345Z","iopub.status.idle":"2022-07-30T14:00:04.347515Z","shell.execute_reply.started":"2022-07-30T14:00:04.311296Z","shell.execute_reply":"2022-07-30T14:00:04.346648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Predict Item Count Per Day**","metadata":{}},{"cell_type":"code","source":"#Let's generate the model from the data train first\n#I use multiple linear regression\n\nX_tr=Sales_train.iloc[:,[2,3]]\nY_Tr=Sales_train.iloc[:,[5]]\nfrom sklearn.model_selection import train_test_split\nX_Train, X_test, y_train, y_test = train_test_split(X_tr, Y_Tr, test_size = 0.2, random_state = 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:00:04.350616Z","iopub.execute_input":"2022-07-30T14:00:04.350953Z","iopub.status.idle":"2022-07-30T14:00:05.761823Z","shell.execute_reply.started":"2022-07-30T14:00:04.350920Z","shell.execute_reply":"2022-07-30T14:00:05.760545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = LinearRegression()\nmodel.fit(X_Train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:00:05.763182Z","iopub.execute_input":"2022-07-30T14:00:05.763466Z","iopub.status.idle":"2022-07-30T14:00:05.950477Z","shell.execute_reply.started":"2022-07-30T14:00:05.763437Z","shell.execute_reply":"2022-07-30T14:00:05.949410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_prediction_training_data= model.predict(X_test)\ny_prediction_training_data","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:00:05.954120Z","iopub.execute_input":"2022-07-30T14:00:05.955226Z","iopub.status.idle":"2022-07-30T14:00:05.985617Z","shell.execute_reply.started":"2022-07-30T14:00:05.955191Z","shell.execute_reply":"2022-07-30T14:00:05.984732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_prediction_training_data,y_test","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:00:05.986867Z","iopub.execute_input":"2022-07-30T14:00:05.988304Z","iopub.status.idle":"2022-07-30T14:00:06.005250Z","shell.execute_reply.started":"2022-07-30T14:00:05.988270Z","shell.execute_reply":"2022-07-30T14:00:06.004418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Let's predict the test data****","metadata":{}},{"cell_type":"code","source":"Y_prediction_from_test_data = model.predict(X_test_test)\nY_prediction_from_test_data","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:00:06.006419Z","iopub.execute_input":"2022-07-30T14:00:06.007836Z","iopub.status.idle":"2022-07-30T14:00:06.031193Z","shell.execute_reply.started":"2022-07-30T14:00:06.007804Z","shell.execute_reply":"2022-07-30T14:00:06.030366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Predict The Future Sales Using Decision Tree Regression**","metadata":{}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeRegressor\nregressor = DecisionTreeRegressor(random_state = 0)\nregressor.fit(X_tr, Y_Tr)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:01:24.456815Z","iopub.execute_input":"2022-07-30T14:01:24.457210Z","iopub.status.idle":"2022-07-30T14:01:31.723629Z","shell.execute_reply.started":"2022-07-30T14:01:24.457180Z","shell.execute_reply":"2022-07-30T14:01:31.722414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_predict= regressor.predict(X_test)\ny_predict","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:04:07.122717Z","iopub.execute_input":"2022-07-30T14:04:07.123116Z","iopub.status.idle":"2022-07-30T14:04:07.355537Z","shell.execute_reply.started":"2022-07-30T14:04:07.123085Z","shell.execute_reply":"2022-07-30T14:04:07.354320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_predict,y_test","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:04:49.773727Z","iopub.execute_input":"2022-07-30T14:04:49.774186Z","iopub.status.idle":"2022-07-30T14:04:49.785984Z","shell.execute_reply.started":"2022-07-30T14:04:49.774155Z","shell.execute_reply":"2022-07-30T14:04:49.784250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_predict_DR= regressor.predict(X_test_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:10:43.518401Z","iopub.execute_input":"2022-07-30T14:10:43.519704Z","iopub.status.idle":"2022-07-30T14:10:43.578086Z","shell.execute_reply.started":"2022-07-30T14:10:43.519643Z","shell.execute_reply":"2022-07-30T14:10:43.576931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'ID':test['ID'],'item_cnt_month':y_predict_DR.ravel()})\nsubmission['item_cnt_month'] = submission['item_cnt_month']\nsubmission.to_csv('submission.csv',index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:11:10.617351Z","iopub.execute_input":"2022-07-30T14:11:10.617734Z","iopub.status.idle":"2022-07-30T14:11:10.938341Z","shell.execute_reply.started":"2022-07-30T14:11:10.617704Z","shell.execute_reply":"2022-07-30T14:11:10.937108Z"},"trusted":true},"execution_count":null,"outputs":[]}]}