{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-22T19:12:38.229531Z","iopub.execute_input":"2022-07-22T19:12:38.229931Z","iopub.status.idle":"2022-07-22T19:12:38.239075Z","shell.execute_reply.started":"2022-07-22T19:12:38.229898Z","shell.execute_reply":"2022-07-22T19:12:38.237635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let us import required libraries for the project","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\n%matplotlib inline\nfrom tensorflow import keras\nfrom sklearn.preprocessing import StandardScaler\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import LSTM, Dense","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:26:01.373942Z","iopub.execute_input":"2022-07-22T19:26:01.374310Z","iopub.status.idle":"2022-07-22T19:26:08.809269Z","shell.execute_reply.started":"2022-07-22T19:26:01.374281Z","shell.execute_reply":"2022-07-22T19:26:08.807882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item_categories_df = pd.read_csv(\"/kaggle/input/competitive-data-science-predict-future-sales/item_categories.csv\")\nshops_df = pd.read_csv(\"/kaggle/input/competitive-data-science-predict-future-sales/shops.csv\")\nsales_train_df = pd.read_csv(\"/kaggle/input/competitive-data-science-predict-future-sales/sales_train.csv\")\nitems_df = pd.read_csv(\"/kaggle/input/competitive-data-science-predict-future-sales/items.csv\")\ntest_df = pd.read_csv('/kaggle/input/competitive-data-science-predict-future-sales/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:12:38.269429Z","iopub.execute_input":"2022-07-22T19:12:38.269983Z","iopub.status.idle":"2022-07-22T19:12:40.165001Z","shell.execute_reply.started":"2022-07-22T19:12:38.269948Z","shell.execute_reply":"2022-07-22T19:12:40.164064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let us check the train data","metadata":{}},{"cell_type":"code","source":"sales_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:12:40.166694Z","iopub.execute_input":"2022-07-22T19:12:40.167539Z","iopub.status.idle":"2022-07-22T19:12:40.179897Z","shell.execute_reply.started":"2022-07-22T19:12:40.167504Z","shell.execute_reply":"2022-07-22T19:12:40.179020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check the dimensions,datatyes, and other statistics of the data","metadata":{}},{"cell_type":"code","source":"sales_train_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:12:40.181620Z","iopub.execute_input":"2022-07-22T19:12:40.182268Z","iopub.status.idle":"2022-07-22T19:12:40.195525Z","shell.execute_reply.started":"2022-07-22T19:12:40.182231Z","shell.execute_reply":"2022-07-22T19:12:40.194326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sales_train_df.info()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:12:40.197894Z","iopub.execute_input":"2022-07-22T19:12:40.198846Z","iopub.status.idle":"2022-07-22T19:12:40.212238Z","shell.execute_reply.started":"2022-07-22T19:12:40.198797Z","shell.execute_reply":"2022-07-22T19:12:40.210992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sales_train_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:12:40.213499Z","iopub.execute_input":"2022-07-22T19:12:40.214829Z","iopub.status.idle":"2022-07-22T19:12:40.691338Z","shell.execute_reply.started":"2022-07-22T19:12:40.214763Z","shell.execute_reply":"2022-07-22T19:12:40.690173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sales_train_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:12:40.692640Z","iopub.execute_input":"2022-07-22T19:12:40.692985Z","iopub.status.idle":"2022-07-22T19:12:41.032349Z","shell.execute_reply.started":"2022-07-22T19:12:40.692953Z","shell.execute_reply":"2022-07-22T19:12:41.031166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Now let's change the date format to pandas date format.","metadata":{}},{"cell_type":"code","source":"sales_train_df['date'] = pd.to_datetime(sales_train_df['date'])\nsales_train_df['date']","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:12:41.033878Z","iopub.execute_input":"2022-07-22T19:12:41.034233Z","iopub.status.idle":"2022-07-22T19:12:41.555841Z","shell.execute_reply.started":"2022-07-22T19:12:41.034202Z","shell.execute_reply":"2022-07-22T19:12:41.554676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sales_train_df['month_year'] = sales_train_df['date'].dt.to_period('M')\nsales_train_df['month_year']","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:12:41.557597Z","iopub.execute_input":"2022-07-22T19:12:41.557969Z","iopub.status.idle":"2022-07-22T19:12:41.887640Z","shell.execute_reply.started":"2022-07-22T19:12:41.557938Z","shell.execute_reply":"2022-07-22T19:12:41.886470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Now let's consider only valid item_price and item_cnt_day","metadata":{}},{"cell_type":"code","source":"sales_train_df = sales_train_df[sales_train_df['item_price']>0]\nsales_train_df = sales_train_df[sales_train_df['item_cnt_day']>0]","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:12:41.889285Z","iopub.execute_input":"2022-07-22T19:12:41.889629Z","iopub.status.idle":"2022-07-22T19:12:42.208996Z","shell.execute_reply.started":"2022-07-22T19:12:41.889597Z","shell.execute_reply":"2022-07-22T19:12:42.207831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sales_train_df","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:12:42.211946Z","iopub.execute_input":"2022-07-22T19:12:42.212419Z","iopub.status.idle":"2022-07-22T19:12:42.234412Z","shell.execute_reply.started":"2022-07-22T19:12:42.212370Z","shell.execute_reply":"2022-07-22T19:12:42.233188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"monthly_data = sales_train_df.pivot_table(\n    index = ['shop_id','item_id'],\n    values = ['item_cnt_day'],\n    columns = ['date_block_num'],\n    fill_value = 0,\n    aggfunc='sum')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:16:12.178466Z","iopub.execute_input":"2022-07-22T19:16:12.178931Z","iopub.status.idle":"2022-07-22T19:16:15.541256Z","shell.execute_reply.started":"2022-07-22T19:16:12.178880Z","shell.execute_reply":"2022-07-22T19:16:15.540389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"monthly_data.reset_index(inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:23:07.770238Z","iopub.execute_input":"2022-07-22T19:23:07.770711Z","iopub.status.idle":"2022-07-22T19:23:07.786401Z","shell.execute_reply.started":"2022-07-22T19:23:07.770676Z","shell.execute_reply":"2022-07-22T19:23:07.785103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = monthly_data.drop(columns= ['shop_id','item_id'], level=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:23:21.566566Z","iopub.execute_input":"2022-07-22T19:23:21.566997Z","iopub.status.idle":"2022-07-22T19:23:21.706884Z","shell.execute_reply.started":"2022-07-22T19:23:21.566962Z","shell.execute_reply":"2022-07-22T19:23:21.705567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.fillna(0,inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:23:53.136158Z","iopub.execute_input":"2022-07-22T19:23:53.136556Z","iopub.status.idle":"2022-07-22T19:23:53.146683Z","shell.execute_reply.started":"2022-07-22T19:23:53.136522Z","shell.execute_reply":"2022-07-22T19:23:53.145524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = np.expand_dims(train_data.values[:,:-1],axis = 2)\ny_train = train_data.values[:,-1:]","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:24:09.551904Z","iopub.execute_input":"2022-07-22T19:24:09.552258Z","iopub.status.idle":"2022-07-22T19:24:09.558140Z","shell.execute_reply.started":"2022-07-22T19:24:09.552228Z","shell.execute_reply":"2022-07-22T19:24:09.556894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_rows = monthly_data.merge(\n    test_df,\n    on = ['item_id','shop_id'],\n    how = 'right')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:24:32.987357Z","iopub.execute_input":"2022-07-22T19:24:32.987770Z","iopub.status.idle":"2022-07-22T19:24:33.148744Z","shell.execute_reply.started":"2022-07-22T19:24:32.987735Z","shell.execute_reply":"2022-07-22T19:24:33.147418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = test_rows.drop(test_rows.columns[:5], axis=1).drop('ID', axis=1)\nx_test.fillna(0,inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:25:03.945522Z","iopub.execute_input":"2022-07-22T19:25:03.945966Z","iopub.status.idle":"2022-07-22T19:25:04.039730Z","shell.execute_reply.started":"2022-07-22T19:25:03.945931Z","shell.execute_reply":"2022-07-22T19:25:04.038553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = np.expand_dims(x_test,axis = 2)\nprint(x_train.shape,y_train.shape,x_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:25:22.930418Z","iopub.execute_input":"2022-07-22T19:25:22.930864Z","iopub.status.idle":"2022-07-22T19:25:22.937921Z","shell.execute_reply.started":"2022-07-22T19:25:22.930819Z","shell.execute_reply":"2022-07-22T19:25:22.936306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = tf.keras.models.Sequential()    \nmodel.add(LSTM(64, input_shape=(33, 1), return_sequences=False))\nmodel.add(Dense(1))\n    \nmodel.compile(\n    loss = 'mse',\n    optimizer = 'adam', \n    metrics = ['mean_squared_error']        \n)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:26:14.271215Z","iopub.execute_input":"2022-07-22T19:26:14.272187Z","iopub.status.idle":"2022-07-22T19:26:14.705456Z","shell.execute_reply.started":"2022-07-22T19:26:14.272151Z","shell.execute_reply":"2022-07-22T19:26:14.704185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    x_train, \n    y_train, \n    epochs=10, \n    batch_size=4096,\n    verbose=1, \n    shuffle=True,\n    validation_split=0.4)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:26:38.182917Z","iopub.execute_input":"2022-07-22T19:26:38.183362Z","iopub.status.idle":"2022-07-22T19:31:44.962710Z","shell.execute_reply.started":"2022-07-22T19:26:38.183324Z","shell.execute_reply":"2022-07-22T19:31:44.961621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history[\"loss\"], color=\"r\")\nplt.plot(history.history[\"val_loss\"], color=\"g\")\nplt.legend([\"Training\", \"Validation\"])\nplt.xlabel(\"epochs\")\nplt.ylabel(\"loss\")\nplt.title('Evaluation')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:31:44.965228Z","iopub.execute_input":"2022-07-22T19:31:44.965939Z","iopub.status.idle":"2022-07-22T19:31:45.178059Z","shell.execute_reply.started":"2022-07-22T19:31:44.965902Z","shell.execute_reply":"2022-07-22T19:31:45.176933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predict = model.predict(x_test)\nsubmission = pd.DataFrame({'ID':test_df['ID'],'item_cnt_month':test_predict.ravel()})\nsubmission['item_cnt_month'] = submission['item_cnt_month']\nsubmission.to_csv('submission.csv',index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:32:46.275179Z","iopub.execute_input":"2022-07-22T19:32:46.275570Z","iopub.status.idle":"2022-07-22T19:33:41.182915Z","shell.execute_reply.started":"2022-07-22T19:32:46.275539Z","shell.execute_reply":"2022-07-22T19:33:41.181887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.linear_model import LogisticRegression\n# from sklearn.metrics import classification_report, confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2022-07-22T19:20:49.135966Z","iopub.execute_input":"2022-07-22T19:20:49.136341Z","iopub.status.idle":"2022-07-22T19:20:49.366100Z","shell.execute_reply.started":"2022-07-22T19:20:49.136312Z","shell.execute_reply":"2022-07-22T19:20:49.364970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = LogisticRegression()\n# model.fit()","metadata":{},"execution_count":null,"outputs":[]}]}