{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Basic liberaries\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n\n#Sklearn Packages\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.tree import DecisionTreeRegressor\n\n\n\n#get the file paths\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n        \n#Define a function to check the RMSE       \ndef check_RMSE (y_train ,train_prediction , y_test ,  test_predicition):\n    print ('Root Mean squared error for the train data  =  ' , \n           mean_squared_error(y_train ,train_prediction , squared=False ))\n    print ('Root Mean squared error for the test data  =  ' , \n           mean_squared_error(y_test ,test_predicition , squared=False ))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-03T16:42:53.633774Z","iopub.execute_input":"2022-08-03T16:42:53.634230Z","iopub.status.idle":"2022-08-03T16:42:53.646052Z","shell.execute_reply.started":"2022-08-03T16:42:53.634188Z","shell.execute_reply":"2022-08-03T16:42:53.645039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1>Read the CSV's </h1>","metadata":{}},{"cell_type":"code","source":"df_test  = pd.read_csv('/kaggle/input/competitive-data-science-predict-future-sales/test.csv')\ndf_test.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:53.701441Z","iopub.execute_input":"2022-08-03T16:42:53.702092Z","iopub.status.idle":"2022-08-03T16:42:53.775241Z","shell.execute_reply.started":"2022-08-03T16:42:53.702052Z","shell.execute_reply":"2022-08-03T16:42:53.773924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_items = pd.read_csv('/kaggle/input/competitive-data-science-predict-future-sales/items.csv')\ndf_items.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:53.777249Z","iopub.execute_input":"2022-08-03T16:42:53.777706Z","iopub.status.idle":"2022-08-03T16:42:53.834089Z","shell.execute_reply.started":"2022-08-03T16:42:53.777654Z","shell.execute_reply":"2022-08-03T16:42:53.833171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train  = pd.read_csv('/kaggle/input/competitive-data-science-predict-future-sales/sales_train.csv')\ndf_train.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:53.836030Z","iopub.execute_input":"2022-08-03T16:42:53.836591Z","iopub.status.idle":"2022-08-03T16:42:55.453213Z","shell.execute_reply.started":"2022-08-03T16:42:53.836548Z","shell.execute_reply":"2022-08-03T16:42:55.452382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2> First challenge , the train and test data sets were not having the dame columns","metadata":{}},{"cell_type":"code","source":"# add month number 34 to the test dataset\ndf_test['date_block_num'] = 34\ndf_test = df_test[['date_block_num' , 'shop_id' , 'item_id' ]]\ndf_test.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:55.454850Z","iopub.execute_input":"2022-08-03T16:42:55.455358Z","iopub.status.idle":"2022-08-03T16:42:55.473368Z","shell.execute_reply.started":"2022-08-03T16:42:55.455314Z","shell.execute_reply":"2022-08-03T16:42:55.472133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# map the latest price for the items in the train data set to the test data set\nitem_price = dict(df_train.groupby('item_id')['item_price'].last().reset_index().values)\ndf_test['item_price'] = df_test.item_id.map(item_price)\ndf_test.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:55.474807Z","iopub.execute_input":"2022-08-03T16:42:55.475129Z","iopub.status.idle":"2022-08-03T16:42:55.647933Z","shell.execute_reply.started":"2022-08-03T16:42:55.475098Z","shell.execute_reply":"2022-08-03T16:42:55.646974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2> i will Remove the shop_id and item_id in the train dataset and not in the test dataset","metadata":{}},{"cell_type":"code","source":"df_train = df_train[df_train.item_id.isin (df_test.item_id)]\ndf_train = df_train[df_train.shop_id.isin (df_test.shop_id)]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:55.649130Z","iopub.execute_input":"2022-08-03T16:42:55.649443Z","iopub.status.idle":"2022-08-03T16:42:55.826358Z","shell.execute_reply.started":"2022-08-03T16:42:55.649411Z","shell.execute_reply":"2022-08-03T16:42:55.824965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2> Re-shape the train dataset and count the sum of sales per each month as required by the competition","metadata":{}},{"cell_type":"code","source":"df_train = df_train.groupby(['date_block_num' , 'shop_id' , 'item_id']).agg({'item_price': 'last', 'item_cnt_day': 'sum'}).reset_index()\ndf_train.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:55.827938Z","iopub.execute_input":"2022-08-03T16:42:55.828355Z","iopub.status.idle":"2022-08-03T16:42:56.170093Z","shell.execute_reply.started":"2022-08-03T16:42:55.828316Z","shell.execute_reply":"2022-08-03T16:42:56.169116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3> now (item_cnt_day)  represent the sum of sales per month for each item in each shop ","metadata":{}},{"cell_type":"markdown","source":"<h2> Add feature to be unique for shop and item for the test and train dataset","metadata":{}},{"cell_type":"code","source":"df_train['shop*item'] = df_train.shop_id *df_train.item_id\ndf_train.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:56.171565Z","iopub.execute_input":"2022-08-03T16:42:56.172073Z","iopub.status.idle":"2022-08-03T16:42:56.190506Z","shell.execute_reply.started":"2022-08-03T16:42:56.172020Z","shell.execute_reply":"2022-08-03T16:42:56.189153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['shop*item'] = df_test.shop_id *df_test.item_id\ndf_test.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:56.194023Z","iopub.execute_input":"2022-08-03T16:42:56.194408Z","iopub.status.idle":"2022-08-03T16:42:56.211263Z","shell.execute_reply.started":"2022-08-03T16:42:56.194368Z","shell.execute_reply":"2022-08-03T16:42:56.209908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2> from the item dataset let's map the categories to the item_id","metadata":{}},{"cell_type":"code","source":"df_items.drop('item_name' , axis  = 1 , inplace = True)\nitem_cat = dict(df_items.values)\n\ndf_train['item_cat'] = df_train.item_id.map(item_cat)\n\ndf_train.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:56.213397Z","iopub.execute_input":"2022-08-03T16:42:56.213738Z","iopub.status.idle":"2022-08-03T16:42:56.287433Z","shell.execute_reply.started":"2022-08-03T16:42:56.213700Z","shell.execute_reply":"2022-08-03T16:42:56.286333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#map the categories\ndf_test['item_cat'] = df_test.item_id.map(item_cat)\ndf_test.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:56.288748Z","iopub.execute_input":"2022-08-03T16:42:56.289114Z","iopub.status.idle":"2022-08-03T16:42:56.338460Z","shell.execute_reply.started":"2022-08-03T16:42:56.289063Z","shell.execute_reply":"2022-08-03T16:42:56.337365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2> I will concate the two train and test datasets to remove the outliers","metadata":{}},{"cell_type":"code","source":"df = pd.concat([df_train , df_test])\n#Normalize\ndf.item_price = np.log1p(df.item_price)\n#fil l the missing\ndf.item_price = df.item_price.fillna(df.item_price.mean())\n#rremove the outlier\ndf.item_cnt_day = df.item_cnt_day.apply(lambda x : 10 if x>10 else x)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:56.340763Z","iopub.execute_input":"2022-08-03T16:42:56.341134Z","iopub.status.idle":"2022-08-03T16:42:56.695280Z","shell.execute_reply.started":"2022-08-03T16:42:56.341096Z","shell.execute_reply":"2022-08-03T16:42:56.694161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:56.696562Z","iopub.execute_input":"2022-08-03T16:42:56.696881Z","iopub.status.idle":"2022-08-03T16:42:56.711341Z","shell.execute_reply.started":"2022-08-03T16:42:56.696849Z","shell.execute_reply":"2022-08-03T16:42:56.710094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2> V10 : encode columns","metadata":{}},{"cell_type":"code","source":"\n\ndef encode_the_numbers (column):\n    \"\"\"\n    function to encode the pandas column depend on thier average target from low to high\n    \"\"\"\n    helper_df = df.groupby(column)['item_cnt_day'].mean().sort_values(ascending = False).reset_index().reset_index()\n    maper = helper_df.groupby(column)[\"index\"].mean().to_dict()\n    df[f'{column}_mean'] = df[column].map(maper)\n    \n    helper_df = df.groupby(column)['item_cnt_day'].sum().sort_values(ascending = False).reset_index().reset_index()\n    maper = helper_df.groupby(column)[\"index\"].sum().to_dict()\n    df[f'{column}_sum'] = df[column].map(maper)\n    \n    helper_df = df.groupby(column)['item_cnt_day'].count().sort_values(ascending = False).reset_index().reset_index()\n    maper = helper_df.groupby(column)[\"index\"].count().to_dict()\n    df[f'{column}_count'] = df[column].map(maper)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:56.713110Z","iopub.execute_input":"2022-08-03T16:42:56.713537Z","iopub.status.idle":"2022-08-03T16:42:56.724430Z","shell.execute_reply.started":"2022-08-03T16:42:56.713492Z","shell.execute_reply":"2022-08-03T16:42:56.723512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns_to_encode = ['shop_id', 'item_id','shop*item', 'item_cat']\nfor column in columns_to_encode:\n    encode_the_numbers (column)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:56.725954Z","iopub.execute_input":"2022-08-03T16:42:56.726378Z","iopub.status.idle":"2022-08-03T16:42:58.029650Z","shell.execute_reply.started":"2022-08-03T16:42:56.726336Z","shell.execute_reply":"2022-08-03T16:42:58.028676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr_df = df.select_dtypes('number').drop('item_cnt_day', axis=1).corrwith(df.item_cnt_day).sort_values().reset_index().rename(columns = {'index':'feature' ,0:'correlation'})\n\nfig , ax = plt.subplots(figsize  = (5,20))\nax.barh(y =corr_df.feature , width = corr_df.correlation )\nax.set_title('correlation between featuer and target'.title() ,\n            fontsize = 16 , fontfamily = 'serif' , fontweight = 'bold')\nplt.show();","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:58.030928Z","iopub.execute_input":"2022-08-03T16:42:58.031405Z","iopub.status.idle":"2022-08-03T16:42:58.844545Z","shell.execute_reply.started":"2022-08-03T16:42:58.031369Z","shell.execute_reply":"2022-08-03T16:42:58.842006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2> split the train and test","metadata":{}},{"cell_type":"code","source":"df_train = df[df.item_cnt_day.notnull()]\ndf_train.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:58.846322Z","iopub.execute_input":"2022-08-03T16:42:58.846755Z","iopub.status.idle":"2022-08-03T16:42:58.936262Z","shell.execute_reply.started":"2022-08-03T16:42:58.846703Z","shell.execute_reply":"2022-08-03T16:42:58.934909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df[df.item_cnt_day.isnull()]\ndf_test.drop ('item_cnt_day' , axis = 1 , inplace  = True)\ndf_test.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:58.937778Z","iopub.execute_input":"2022-08-03T16:42:58.938147Z","iopub.status.idle":"2022-08-03T16:42:58.984973Z","shell.execute_reply.started":"2022-08-03T16:42:58.938099Z","shell.execute_reply":"2022-08-03T16:42:58.983818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2> prepare the X and y","metadata":{}},{"cell_type":"code","source":"X = df_train.drop('item_cnt_day' , axis = 1).values\ny = df_train.item_cnt_day.values","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:58.986876Z","iopub.execute_input":"2022-08-03T16:42:58.987217Z","iopub.status.idle":"2022-08-03T16:42:59.081298Z","shell.execute_reply.started":"2022-08-03T16:42:58.987181Z","shell.execute_reply":"2022-08-03T16:42:59.080123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Scale the X","metadata":{}},{"cell_type":"code","source":"SC = MinMaxScaler()\nSC.fit(X)\nX = SC.transform(X)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:59.082913Z","iopub.execute_input":"2022-08-03T16:42:59.083267Z","iopub.status.idle":"2022-08-03T16:42:59.182380Z","shell.execute_reply.started":"2022-08-03T16:42:59.083231Z","shell.execute_reply":"2022-08-03T16:42:59.181181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train , x_test , y_train , y_test = train_test_split(X , y , test_size = 0.30 ,  random_state=10)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:59.183722Z","iopub.execute_input":"2022-08-03T16:42:59.184043Z","iopub.status.idle":"2022-08-03T16:42:59.363538Z","shell.execute_reply.started":"2022-08-03T16:42:59.184011Z","shell.execute_reply":"2022-08-03T16:42:59.362414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Random forest","metadata":{}},{"cell_type":"code","source":"# from sklearn.ensemble import RandomForestRegressor\n\nreg = RandomForestRegressor(n_estimators = 25)\nreg.fit(x_train, y_train)\ntrain_prediction = reg.predict(x_train)\ntest_prediction = reg.predict(x_test)\n\ncheck_RMSE(y_train, train_prediction, y_test, test_prediction)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:42:59.365142Z","iopub.execute_input":"2022-08-03T16:42:59.365469Z","iopub.status.idle":"2022-08-03T16:45:43.156047Z","shell.execute_reply.started":"2022-08-03T16:42:59.365436Z","shell.execute_reply":"2022-08-03T16:45:43.154917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Decision Tree Regressor","metadata":{}},{"cell_type":"code","source":"'''dt_reg = DecisionTreeRegressor(random_state = 42)\ndt_reg.fit(x_train, y_train)\n\ntrain_prediction = dt_reg.predict(x_train)\ntest_prediction = dt_reg.predict(x_test)\n\ncheck_RMSE(y_train, train_prediction, y_test, test_prediction)'''","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:45:43.158048Z","iopub.execute_input":"2022-08-03T16:45:43.158616Z","iopub.status.idle":"2022-08-03T16:45:43.168988Z","shell.execute_reply.started":"2022-08-03T16:45:43.158563Z","shell.execute_reply":"2022-08-03T16:45:43.164438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Elastic Net","metadata":{}},{"cell_type":"code","source":"'''from sklearn.linear_model import ElasticNet\n\nspandex = ElasticNet(alpha = 0.004)\nspandex.fit(x_train, y_train)\n\ntrain_prediction = spandex.predict(x_train)\ntest_prediction = spandex.predict(x_test)\n\ncheck_RMSE(y_train, train_prediction, y_test, test_prediction)'''","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:45:43.171849Z","iopub.execute_input":"2022-08-03T16:45:43.173470Z","iopub.status.idle":"2022-08-03T16:45:43.181202Z","shell.execute_reply.started":"2022-08-03T16:45:43.173393Z","shell.execute_reply":"2022-08-03T16:45:43.179910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cat Boost Regressor","metadata":{}},{"cell_type":"code","source":"'''from catboost import CatBoostRegressor\n\nkitty = CatBoostRegressor()\nkitty.fit(x_train, y_train)\n\ntrain_prediction = kitty.predict(x_train)\ntest_prediction = kitty.predict(x_test)\n\ncheck_RMSE(y_train, train_prediction, y_test, test_prediction)\n\n# RMSE ~ 1.2'''","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:45:43.182946Z","iopub.execute_input":"2022-08-03T16:45:43.183675Z","iopub.status.idle":"2022-08-03T16:45:43.193195Z","shell.execute_reply.started":"2022-08-03T16:45:43.183588Z","shell.execute_reply":"2022-08-03T16:45:43.192170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## KNN","metadata":{}},{"cell_type":"code","source":"'''from sklearn.neighbors import KNeighborsRegressor\n\nknn = KNeighborsRegressor()\nknn.fit(x_train, y_train)\ntrain_prediction = knn.predict(x_train)\ntest_prediction = knn.predict(x_test)\n\ncheck_RMSE(y_train, train_prediction, y_test, test_prediction)'''","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:45:43.194721Z","iopub.execute_input":"2022-08-03T16:45:43.195300Z","iopub.status.idle":"2022-08-03T16:45:43.206179Z","shell.execute_reply.started":"2022-08-03T16:45:43.195249Z","shell.execute_reply":"2022-08-03T16:45:43.204931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Linear reg","metadata":{}},{"cell_type":"code","source":"'''from sklearn.linear_model import LinearRegression\n\nlr = LinearRegression()\nlr.fit(x_train, y_train)\ntrain_prediction = lr.predict(x_train)\ntest_prediction = lr.predict(x_test)\n\ncheck_RMSE(y_train, train_prediction, y_test, test_prediction)'''","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:45:43.211187Z","iopub.execute_input":"2022-08-03T16:45:43.212654Z","iopub.status.idle":"2022-08-03T16:45:43.220185Z","shell.execute_reply.started":"2022-08-03T16:45:43.212603Z","shell.execute_reply":"2022-08-03T16:45:43.219034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## SVR","metadata":{}},{"cell_type":"code","source":"'''from sklearn.svm import SVR\n\nsvr = SVR()\nsvr.fit(x_train, y_train)\ntrain_prediction = svr.predict(x_train)\ntest_prediction = svr.predict(x_test)\n\ncheck_RMSE(y_train, train_prediction, y_test, test_prediction)'''","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:45:43.222387Z","iopub.execute_input":"2022-08-03T16:45:43.223021Z","iopub.status.idle":"2022-08-03T16:45:43.231736Z","shell.execute_reply.started":"2022-08-03T16:45:43.222962Z","shell.execute_reply":"2022-08-03T16:45:43.230523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Ridge","metadata":{}},{"cell_type":"code","source":"'''\nfrom sklearn.linear_model import Ridge\nRidge=Ridge()\nRidge.fit(x_train,y_train)\ntrain_prediction  = Ridge.predict(x_train)\ntest_predicition  = Ridge.predict(x_test)\n\ncheck_RMSE (y_train ,train_prediction , y_test ,  test_predicition)\n'''\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:45:43.233413Z","iopub.execute_input":"2022-08-03T16:45:43.234092Z","iopub.status.idle":"2022-08-03T16:45:43.245699Z","shell.execute_reply.started":"2022-08-03T16:45:43.234043Z","shell.execute_reply":"2022-08-03T16:45:43.244361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## BayesianRidge","metadata":{}},{"cell_type":"code","source":"'''\nfrom sklearn.linear_model import BayesianRidge\nBayesian = BayesianRidge()\nBayesian.fit(x_train,y_train)\ntrain_prediction  = Bayesian.predict(x_train)\ntest_predicition  = Bayesian.predict(x_test)\n\ncheck_RMSE (y_train ,train_prediction , y_test ,  test_predicition)\n'''","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:45:43.247879Z","iopub.execute_input":"2022-08-03T16:45:43.248561Z","iopub.status.idle":"2022-08-03T16:45:43.256330Z","shell.execute_reply.started":"2022-08-03T16:45:43.248507Z","shell.execute_reply":"2022-08-03T16:45:43.255254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2> prepare the test data for submission","metadata":{}},{"cell_type":"code","source":"X_submission =df_test.values\nX_submission = SC.transform(X_submission)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:45:43.257726Z","iopub.execute_input":"2022-08-03T16:45:43.258099Z","iopub.status.idle":"2022-08-03T16:45:43.337371Z","shell.execute_reply.started":"2022-08-03T16:45:43.258053Z","shell.execute_reply":"2022-08-03T16:45:43.336032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Select random forest","metadata":{}},{"cell_type":"code","source":"predection  = reg.predict(X_submission)\nsample_submission  = pd.read_csv('../input/competitive-data-science-predict-future-sales/sample_submission.csv')\nsample_submission.item_cnt_month = predection\nsample_submission.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:45:43.338890Z","iopub.execute_input":"2022-08-03T16:45:43.339249Z","iopub.status.idle":"2022-08-03T16:45:44.450202Z","shell.execute_reply.started":"2022-08-03T16:45:43.339209Z","shell.execute_reply":"2022-08-03T16:45:44.448938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.to_csv('submission.csv' , index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:45:44.451601Z","iopub.execute_input":"2022-08-03T16:45:44.452013Z","iopub.status.idle":"2022-08-03T16:45:44.810098Z","shell.execute_reply.started":"2022-08-03T16:45:44.451970Z","shell.execute_reply":"2022-08-03T16:45:44.808990Z"},"trusted":true},"execution_count":null,"outputs":[]}]}