{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-29T00:24:15.236177Z","iopub.execute_input":"2022-07-29T00:24:15.237905Z","iopub.status.idle":"2022-07-29T00:24:15.248468Z","shell.execute_reply.started":"2022-07-29T00:24:15.237854Z","shell.execute_reply":"2022-07-29T00:24:15.247515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datetime import datetime\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import accuracy_score\nimport plotly.express as px","metadata":{"execution":{"iopub.status.busy":"2022-07-29T00:24:17.207373Z","iopub.execute_input":"2022-07-29T00:24:17.208034Z","iopub.status.idle":"2022-07-29T00:24:17.213142Z","shell.execute_reply.started":"2022-07-29T00:24:17.207998Z","shell.execute_reply":"2022-07-29T00:24:17.212236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oil = pd.read_csv('/kaggle/input/store-sales-time-series-forecasting/oil.csv')\nsubmission = pd.read_csv('/kaggle/input/store-sales-time-series-forecasting/sample_submission.csv')\nholidays = pd.read_csv('/kaggle/input/store-sales-time-series-forecasting/holidays_events.csv')\nstores = pd.read_csv('/kaggle/input/store-sales-time-series-forecasting/stores.csv')\ntrain = pd.read_csv('/kaggle/input/store-sales-time-series-forecasting/train.csv')\ntest = pd.read_csv('/kaggle/input/store-sales-time-series-forecasting/test.csv')\ntransactions = pd.read_csv('/kaggle/input/store-sales-time-series-forecasting/transactions.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T00:24:20.205034Z","iopub.execute_input":"2022-07-29T00:24:20.205718Z","iopub.status.idle":"2022-07-29T00:24:22.270601Z","shell.execute_reply.started":"2022-07-29T00:24:20.205676Z","shell.execute_reply":"2022-07-29T00:24:22.269507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dateが文字列データ(オブジェクト)なので，timestamp型に変更\ntrain['date'] = pd.to_datetime(train['date'])\ntest['date'] = pd.to_datetime(test['date'])\noil['date'] = pd.to_datetime(oil['date'])\nholidays['date'] = pd.to_datetime(holidays['date'])\ntransactions['date'] = pd.to_datetime(transactions['date'])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T00:24:24.124700Z","iopub.execute_input":"2022-07-29T00:24:24.125378Z","iopub.status.idle":"2022-07-29T00:24:24.620639Z","shell.execute_reply.started":"2022-07-29T00:24:24.125313Z","shell.execute_reply":"2022-07-29T00:24:24.619667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"oilの欠損値を補完","metadata":{}},{"cell_type":"code","source":"oil.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T00:24:27.051466Z","iopub.execute_input":"2022-07-29T00:24:27.052116Z","iopub.status.idle":"2022-07-29T00:24:27.064219Z","shell.execute_reply.started":"2022-07-29T00:24:27.052083Z","shell.execute_reply":"2022-07-29T00:24:27.062853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# oilの欠損を線形補完\noil_new = pd.DataFrame(pd.date_range(start=oil['date'].min(), end=oil['date'].max(), freq='D'), columns=['date'])\noil_new = pd.merge(oil_new, oil, how='left', on='date')\noil_new['dcoilwtico'] = oil_new.drop('date', axis=1).interpolate(limit_direction='both')['dcoilwtico']\n\n# trainとtestに分ける\noil_train = oil_new[(train['date'].min() <= oil_new['date']) & (oil_new['date'] <= train['date'].max())]\noil_test = oil_new[(test['date'].min() <= oil_new['date']) & (oil_new['date'] <= test['date'].max())]\n\n# oilを結合\ntrain = pd.merge(train, oil_train, how='left', on='date')\ntest = pd.merge(test, oil_test, how='left', on='date')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T00:24:29.434882Z","iopub.execute_input":"2022-07-29T00:24:29.435262Z","iopub.status.idle":"2022-07-29T00:24:29.941725Z","shell.execute_reply.started":"2022-07-29T00:24:29.435228Z","shell.execute_reply":"2022-07-29T00:24:29.940525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T00:24:32.600441Z","iopub.execute_input":"2022-07-29T00:24:32.600985Z","iopub.status.idle":"2022-07-29T00:24:32.616773Z","shell.execute_reply.started":"2022-07-29T00:24:32.600941Z","shell.execute_reply":"2022-07-29T00:24:32.615906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"familyとstore_nbrをダミー化","metadata":{}},{"cell_type":"code","source":"train = pd.get_dummies(train,columns = ['family','store_nbr'],drop_first = True)\ntest = pd.get_dummies(test,columns = ['family','store_nbr'],drop_first = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T00:24:34.501994Z","iopub.execute_input":"2022-07-29T00:24:34.502463Z","iopub.status.idle":"2022-07-29T00:24:36.589481Z","shell.execute_reply.started":"2022-07-29T00:24:34.502426Z","shell.execute_reply":"2022-07-29T00:24:36.588150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"売り上げが0の時は休業日として削除","metadata":{}},{"cell_type":"code","source":"df = train\ndf = df[df['sales'] != 0.000]\ndf","metadata":{"execution":{"iopub.status.busy":"2022-07-29T00:24:37.802045Z","iopub.execute_input":"2022-07-29T00:24:37.802571Z","iopub.status.idle":"2022-07-29T00:24:39.522066Z","shell.execute_reply.started":"2022-07-29T00:24:37.802523Z","shell.execute_reply":"2022-07-29T00:24:39.521272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = df\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-07-29T00:24:42.520509Z","iopub.execute_input":"2022-07-29T00:24:42.520916Z","iopub.status.idle":"2022-07-29T00:24:42.693917Z","shell.execute_reply.started":"2022-07-29T00:24:42.520884Z","shell.execute_reply":"2022-07-29T00:24:42.692778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# salesを分離\ntrain_y = train['sales']\ntrain.drop('sales', inplace=True, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T00:24:45.465537Z","iopub.execute_input":"2022-07-29T00:24:45.465952Z","iopub.status.idle":"2022-07-29T00:24:45.849179Z","shell.execute_reply.started":"2022-07-29T00:24:45.465917Z","shell.execute_reply":"2022-07-29T00:24:45.847689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"線形回帰を行う\n目的関数(date，onpromotion，store_nbr，family，dcoilwtico)","metadata":{}},{"cell_type":"code","source":"train['day']  = pd.to_datetime(train['date']).dt.day\ntrain['month']  = pd.to_datetime(train['date']).dt.month\ntrain['year']  = pd.to_datetime(train['date']).dt.year\ntest['day']  = pd.to_datetime(test['date']).dt.day\ntest['month']  = pd.to_datetime(test['date']).dt.month\ntest['year']  = pd.to_datetime(test['date']).dt.year\ntrain.drop('date',axis=1,inplace=True)\ntest.drop('date',axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T00:26:07.131853Z","iopub.execute_input":"2022-07-29T00:26:07.132500Z","iopub.status.idle":"2022-07-29T00:26:08.378645Z","shell.execute_reply.started":"2022-07-29T00:26:07.132443Z","shell.execute_reply":"2022-07-29T00:26:08.377553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = LinearRegression()\nmodel.fit(train, train_y)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T00:26:12.679744Z","iopub.execute_input":"2022-07-29T00:26:12.680134Z","iopub.status.idle":"2022-07-29T00:26:26.844740Z","shell.execute_reply.started":"2022-07-29T00:26:12.680103Z","shell.execute_reply":"2022-07-29T00:26:26.843152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(test)\npred","metadata":{"execution":{"iopub.status.busy":"2022-07-29T00:26:30.017518Z","iopub.execute_input":"2022-07-29T00:26:30.018767Z","iopub.status.idle":"2022-07-29T00:26:30.044724Z","shell.execute_reply.started":"2022-07-29T00:26:30.018720Z","shell.execute_reply":"2022-07-29T00:26:30.043327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['sales'] = pred\nsubmission.to_csv('submission.csv' , index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T00:26:33.105870Z","iopub.execute_input":"2022-07-29T00:26:33.106614Z","iopub.status.idle":"2022-07-29T00:26:33.211006Z","shell.execute_reply.started":"2022-07-29T00:26:33.106578Z","shell.execute_reply":"2022-07-29T00:26:33.210116Z"},"trusted":true},"execution_count":null,"outputs":[]}]}