{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-04T07:22:39.240183Z","iopub.execute_input":"2022-08-04T07:22:39.240637Z","iopub.status.idle":"2022-08-04T07:22:39.248386Z","shell.execute_reply.started":"2022-08-04T07:22:39.240601Z","shell.execute_reply":"2022-08-04T07:22:39.247293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\nimport warnings\nwarnings.simplefilter('ignore')\n\nimport random","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:39.295620Z","iopub.execute_input":"2022-08-04T07:22:39.296022Z","iopub.status.idle":"2022-08-04T07:22:39.304317Z","shell.execute_reply.started":"2022-08-04T07:22:39.295992Z","shell.execute_reply":"2022-08-04T07:22:39.303454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fix_seed(seed):\n    # random\n    random.seed(seed)\n    # Numpy\n    np.random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n\nSEED = 42\nfix_seed(SEED)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:39.348278Z","iopub.execute_input":"2022-08-04T07:22:39.349402Z","iopub.status.idle":"2022-08-04T07:22:39.354867Z","shell.execute_reply.started":"2022-08-04T07:22:39.349359Z","shell.execute_reply":"2022-08-04T07:22:39.353713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"../input/Covid19-Death-Predictions/train.csv\")\ndf","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:39.453843Z","iopub.execute_input":"2022-08-04T07:22:39.454733Z","iopub.status.idle":"2022-08-04T07:22:39.716106Z","shell.execute_reply.started":"2022-08-04T07:22:39.454696Z","shell.execute_reply":"2022-08-04T07:22:39.714901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(\"../input/Covid19-Death-Predictions/test.csv\")\ntest","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:39.718321Z","iopub.execute_input":"2022-08-04T07:22:39.719076Z","iopub.status.idle":"2022-08-04T07:22:39.844785Z","shell.execute_reply.started":"2022-08-04T07:22:39.719034Z","shell.execute_reply":"2022-08-04T07:22:39.843551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:39.846401Z","iopub.execute_input":"2022-08-04T07:22:39.847457Z","iopub.status.idle":"2022-08-04T07:22:39.874555Z","shell.execute_reply.started":"2022-08-04T07:22:39.847405Z","shell.execute_reply":"2022-08-04T07:22:39.873323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:39.877194Z","iopub.execute_input":"2022-08-04T07:22:39.877678Z","iopub.status.idle":"2022-08-04T07:22:39.884945Z","shell.execute_reply.started":"2022-08-04T07:22:39.877643Z","shell.execute_reply":"2022-08-04T07:22:39.883966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:39.886701Z","iopub.execute_input":"2022-08-04T07:22:39.887089Z","iopub.status.idle":"2022-08-04T07:22:39.954379Z","shell.execute_reply.started":"2022-08-04T07:22:39.887054Z","shell.execute_reply":"2022-08-04T07:22:39.953569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:39.955612Z","iopub.execute_input":"2022-08-04T07:22:39.956133Z","iopub.status.idle":"2022-08-04T07:22:39.960771Z","shell.execute_reply.started":"2022-08-04T07:22:39.956102Z","shell.execute_reply":"2022-08-04T07:22:39.959569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le=LabelEncoder()\n\nle.fit(df[\"Location\"])\ndf[\"Location\"] = le.transform(df[\"Location\"])\ntest[\"Location\"] = le.transform(test[\"Location\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:39.962482Z","iopub.execute_input":"2022-08-04T07:22:39.962942Z","iopub.status.idle":"2022-08-04T07:22:40.014966Z","shell.execute_reply.started":"2022-08-04T07:22:39.962909Z","shell.execute_reply":"2022-08-04T07:22:40.013871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = df.copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:40.016218Z","iopub.execute_input":"2022-08-04T07:22:40.016538Z","iopub.status.idle":"2022-08-04T07:22:40.027868Z","shell.execute_reply.started":"2022-08-04T07:22:40.016494Z","shell.execute_reply":"2022-08-04T07:22:40.026737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import preprocessing\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.model_selection import StratifiedKFold","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:40.031027Z","iopub.execute_input":"2022-08-04T07:22:40.032859Z","iopub.status.idle":"2022-08-04T07:22:40.038769Z","shell.execute_reply.started":"2022-08-04T07:22:40.032806Z","shell.execute_reply":"2022-08-04T07:22:40.037532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folds = train.copy()\nFold = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\nfor n, (train_index, val_index) in enumerate(Fold.split(folds, folds[\"Next Week's Deaths\"])):\n    folds.loc[val_index, 'fold'] = int(n)\nfolds['fold'] = folds['fold'].astype(int)\nprint(folds.groupby(['fold', \"Next Week's Deaths\"]).size())","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:40.040569Z","iopub.execute_input":"2022-08-04T07:22:40.041027Z","iopub.status.idle":"2022-08-04T07:22:40.890884Z","shell.execute_reply.started":"2022-08-04T07:22:40.040985Z","shell.execute_reply":"2022-08-04T07:22:40.889558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folds","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:40.893273Z","iopub.execute_input":"2022-08-04T07:22:40.893803Z","iopub.status.idle":"2022-08-04T07:22:40.950572Z","shell.execute_reply.started":"2022-08-04T07:22:40.893754Z","shell.execute_reply":"2022-08-04T07:22:40.949493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p_train = folds[folds[\"fold\"] != 0]\np_val = folds[folds[\"fold\"] == 0]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:40.953406Z","iopub.execute_input":"2022-08-04T07:22:40.953864Z","iopub.status.idle":"2022-08-04T07:22:40.978975Z","shell.execute_reply.started":"2022-08-04T07:22:40.953817Z","shell.execute_reply":"2022-08-04T07:22:40.977906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p_train","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:40.980396Z","iopub.execute_input":"2022-08-04T07:22:40.981311Z","iopub.status.idle":"2022-08-04T07:22:41.040877Z","shell.execute_reply.started":"2022-08-04T07:22:40.981274Z","shell.execute_reply":"2022-08-04T07:22:41.039578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# An error will occur later, so reassign the index.\n# 後ほどエラーが出るので、indexを振りなおす。\n\np_train = p_train.reset_index(drop=True)\np_val = p_val.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:41.042707Z","iopub.execute_input":"2022-08-04T07:22:41.043054Z","iopub.status.idle":"2022-08-04T07:22:41.052640Z","shell.execute_reply.started":"2022-08-04T07:22:41.043023Z","shell.execute_reply":"2022-08-04T07:22:41.051713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p_train","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:41.054012Z","iopub.execute_input":"2022-08-04T07:22:41.054968Z","iopub.status.idle":"2022-08-04T07:22:41.116086Z","shell.execute_reply.started":"2022-08-04T07:22:41.054933Z","shell.execute_reply":"2022-08-04T07:22:41.114859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:41.117971Z","iopub.execute_input":"2022-08-04T07:22:41.118706Z","iopub.status.idle":"2022-08-04T07:22:41.124063Z","shell.execute_reply.started":"2022-08-04T07:22:41.118650Z","shell.execute_reply":"2022-08-04T07:22:41.122828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fix_seed(SEED) # for repetability","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:41.125918Z","iopub.execute_input":"2022-08-04T07:22:41.126356Z","iopub.status.idle":"2022-08-04T07:22:41.136410Z","shell.execute_reply.started":"2022-08-04T07:22:41.126314Z","shell.execute_reply":"2022-08-04T07:22:41.135584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defining the feature columns and the target\n\nFEATURES = [\"Location\",\"Weekly Cases\",\"Year\",\"Weekly Cases per Million\",\"Weekly Deaths\",\"Weekly Deaths per Million\",\"Total Vaccinations\",\"People Vaccinated\",\"People Fully Vaccinated\",\"Total Boosters\",\"Daily Vaccinations\",\"Total Vaccinations per Hundred\",\"People Vaccinated per Hundred\",\"People Fully Vaccinated per Hundred\",\"Total Boosters per Hundred\",\"Daily Vaccinations per Hundred\",\"Daily People Vaccinated\",\"Daily People Vaccinated per Hundred\"]\nTARGET = \"Next Week's Deaths\"","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:41.138122Z","iopub.execute_input":"2022-08-04T07:22:41.138845Z","iopub.status.idle":"2022-08-04T07:22:41.148437Z","shell.execute_reply.started":"2022-08-04T07:22:41.138791Z","shell.execute_reply":"2022-08-04T07:22:41.147362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p_train[FEATURES]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:41.149787Z","iopub.execute_input":"2022-08-04T07:22:41.150124Z","iopub.status.idle":"2022-08-04T07:22:41.192396Z","shell.execute_reply.started":"2022-08-04T07:22:41.150093Z","shell.execute_reply":"2022-08-04T07:22:41.191235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_train = lgb.Dataset(p_train[FEATURES], p_train[TARGET])\nlgb_eval = lgb.Dataset(p_val[FEATURES], p_val[TARGET])","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:41.197868Z","iopub.execute_input":"2022-08-04T07:22:41.198231Z","iopub.status.idle":"2022-08-04T07:22:41.210171Z","shell.execute_reply.started":"2022-08-04T07:22:41.198200Z","shell.execute_reply":"2022-08-04T07:22:41.208917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# example of parameters\nlgbm_params = {\n    'objective': 'binary', # Binary classification : 2値分類ではこれを使う\n    'seed': 42, # random seed : これを固定すると、再現性が出る\n    'metric': 'auc', \n#    'learning_rate': 0.01,\n#    'max_bin': 800, # depth\n#    'num_leaves': 80, # leaves,\n    \"verbose\":-1,\n    \"deterministic\":True\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:41.211497Z","iopub.execute_input":"2022-08-04T07:22:41.212222Z","iopub.status.idle":"2022-08-04T07:22:41.216482Z","shell.execute_reply.started":"2022-08-04T07:22:41.212189Z","shell.execute_reply":"2022-08-04T07:22:41.215710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[FEATURES].head(3)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:41.217682Z","iopub.execute_input":"2022-08-04T07:22:41.218150Z","iopub.status.idle":"2022-08-04T07:22:41.252798Z","shell.execute_reply.started":"2022-08-04T07:22:41.218120Z","shell.execute_reply":"2022-08-04T07:22:41.252024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_list = ['Location']\n\n\nmodel = lgb.train(lgbm_params, lgb_train, valid_sets=lgb_eval,\n                  verbose_eval=50,  # Learning result output every 50 iterations : 50イテレーション毎に学習結果出力\n                  num_boost_round=1000,  # Specify the maximum number of iterations : 最大イテレーション回数指定\n                  early_stopping_rounds=100, # Early stopping number : early stoppingを採用するiteration回数\n                  categorical_feature = cat_list # manual categorical feature setting\n                 \n                 )","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:41.253980Z","iopub.execute_input":"2022-08-04T07:22:41.254456Z","iopub.status.idle":"2022-08-04T07:22:50.618589Z","shell.execute_reply.started":"2022-08-04T07:22:41.254426Z","shell.execute_reply":"2022-08-04T07:22:50.617527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\n\nmodel_name = \"LGBMmodel.bin\"\n\n# saving model\npickle.dump(model, open(model_name, 'wb'))\n\n# loading model\nmodel = pickle.load(open(model_name, 'rb'))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:50.621604Z","iopub.execute_input":"2022-08-04T07:22:50.621961Z","iopub.status.idle":"2022-08-04T07:22:50.769985Z","shell.execute_reply.started":"2022-08-04T07:22:50.621931Z","shell.execute_reply":"2022-08-04T07:22:50.769003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:50.771566Z","iopub.execute_input":"2022-08-04T07:22:50.772279Z","iopub.status.idle":"2022-08-04T07:22:50.778013Z","shell.execute_reply.started":"2022-08-04T07:22:50.772241Z","shell.execute_reply":"2022-08-04T07:22:50.776770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.save_model(f'model.txt')\nlgb.plot_importance(model, importance_type='gain')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:50.779871Z","iopub.execute_input":"2022-08-04T07:22:50.780438Z","iopub.status.idle":"2022-08-04T07:22:51.178358Z","shell.execute_reply.started":"2022-08-04T07:22:50.780390Z","shell.execute_reply":"2022-08-04T07:22:51.177113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ライブラリのインポート\nimport pandas as pd # 基本ライブラリ\nimport numpy as np # 基本ライブラリ\nimport matplotlib.pyplot as plt # グラフ描画用\nimport seaborn as sns; sns.set() # グラフ描画用\nimport warnings # 実行に関係ない警告を無視\nwarnings.filterwarnings('ignore')\nimport lightgbm as lgb #LightGBM\nfrom sklearn import datasets\nfrom sklearn.model_selection import train_test_split # データセット分割用\nfrom sklearn.metrics import mean_squared_error # モデル評価用(平均二乗誤差)\nfrom sklearn.metrics import r2_score # モデル評価用(決定係数)\n\n# データフレームを綺麗に出力する関数\nimport IPython\ndef display(*dfs, head=True):\n    for df in dfs:\n        IPython.display.display(df.head() if head else df)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:51.179921Z","iopub.execute_input":"2022-08-04T07:22:51.180347Z","iopub.status.idle":"2022-08-04T07:22:51.189761Z","shell.execute_reply.started":"2022-08-04T07:22:51.180293Z","shell.execute_reply":"2022-08-04T07:22:51.188904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# データの確認\nprint(df.shape) # データサイズの確認(データ数,特徴量数(変数の個数))\ndisplay(df) # df.head()に同じ(文中に入れるときはdisplay()を使う)\n\n# 説明変数,目的変数\nX = df.drop(\"Next Week's Deaths\",axis=1).values # 説明変数(Next Week's Deaths以外の特徴量)\ny = df[\"Next Week's Deaths\"].values # 目的変数(Next Week's Deaths)\n\n# トレーニングデータ,テストデータの分割\nX_train, X_test, y_train, y_test = train_test_split(X, y,test_size=0.20, random_state=2)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:51.191075Z","iopub.execute_input":"2022-08-04T07:22:51.191696Z","iopub.status.idle":"2022-08-04T07:22:51.264029Z","shell.execute_reply.started":"2022-08-04T07:22:51.191661Z","shell.execute_reply":"2022-08-04T07:22:51.262838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# モデルの学習\nmodel = lgb.LGBMRegressor() # モデルのインスタンスの作成\nmodel.fit(X_train, y_train) # モデルの学習\n\n# テストデータの予測\ny_pred = model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:51.265656Z","iopub.execute_input":"2022-08-04T07:22:51.266005Z","iopub.status.idle":"2022-08-04T07:22:52.109456Z","shell.execute_reply.started":"2022-08-04T07:22:51.265973Z","shell.execute_reply":"2022-08-04T07:22:52.108489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 真値と予測値の表示\ndf_pred = pd.DataFrame({\"Next Week's Deaths\":y_test,\"Next Week's Deaths_pred\":y_pred})\ndisplay(df_pred)\n\n# 散布図を描画(真値 vs 予測値)\nplt.plot(y_test, y_test, color = 'red', label = 'x=y') # 直線y = x (真値と予測値が同じ場合は直線状に点がプロットされる)\nplt.scatter(y_test, y_pred) # 散布図のプロット\nplt.xlabel('y') # x軸ラベル\nplt.ylabel('y_test') # y軸ラベル\nplt.title('y vs y_pred') # グラフタイトル","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:52.111155Z","iopub.execute_input":"2022-08-04T07:22:52.111984Z","iopub.status.idle":"2022-08-04T07:22:52.411211Z","shell.execute_reply.started":"2022-08-04T07:22:52.111938Z","shell.execute_reply":"2022-08-04T07:22:52.410409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# モデル評価\n# rmse : 平均二乗誤差の平方根\nmse = mean_squared_error(y_test, y_pred) # MSE(平均二乗誤差)の算出\nrmse = np.sqrt(mse) # RSME = √MSEの算出\nprint('RMSE :',rmse)\n\n# r2 : 決定係数\nr2 = r2_score(y_test,y_pred)\nprint('R2 :',r2)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:52.412421Z","iopub.execute_input":"2022-08-04T07:22:52.413008Z","iopub.status.idle":"2022-08-04T07:22:52.422106Z","shell.execute_reply.started":"2022-08-04T07:22:52.412972Z","shell.execute_reply":"2022-08-04T07:22:52.421045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 学習に使用するデータを設定\nlgb_train = lgb.Dataset(X_train, y_train)\nlgb_eval = lgb.Dataset(X_test, y_test, reference=lgb_train) \n\n# LightGBM parameters\nparams = {\n        'task': 'train',\n        'boosting_type': 'gbdt',\n        'objective': 'regression', # 目的 : 回帰  \n        'metric': {'rmse'}, # 評価指標 : rsme(平均二乗誤差の平方根) \n}\n\n# モデルの学習\nmodel = lgb.train(params,\n                  train_set=lgb_train, # トレーニングデータの指定\n                  valid_sets=lgb_eval, # 検証データの指定\n                  )\n\n# テストデータの予測\ny_pred = model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:52.423658Z","iopub.execute_input":"2022-08-04T07:22:52.423996Z","iopub.status.idle":"2022-08-04T07:22:53.385371Z","shell.execute_reply.started":"2022-08-04T07:22:52.423964Z","shell.execute_reply":"2022-08-04T07:22:53.384486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 真値と予測値の表示\ndf_pred = pd.DataFrame({\"Next Week's Deaths\":y_test,\"Next Week's Deaths_pred\":y_pred})\ndisplay(df_pred)\n\n# 散布図を描画(真値 vs 予測値)\nplt.plot(y_test, y_test, color = 'red', label = 'x=y') # 直線y = x (真値と予測値が同じ場合は直線状に点がプロットされる)\nplt.scatter(y_test, y_pred) # 散布図のプロット\nplt.xlabel('y') # x軸ラベル\nplt.ylabel('y_test') # y軸ラベル\nplt.title('y vs y_pred') # グラフタイトル","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:53.386932Z","iopub.execute_input":"2022-08-04T07:22:53.387632Z","iopub.status.idle":"2022-08-04T07:22:53.683978Z","shell.execute_reply.started":"2022-08-04T07:22:53.387591Z","shell.execute_reply":"2022-08-04T07:22:53.682806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# モデル評価\n# rmse : 平均二乗誤差の平方根\nmse = mean_squared_error(y_test, y_pred) # MSE(平均二乗誤差)の算出\nrmse = np.sqrt(mse) # RSME = √MSEの算出\nprint('RMSE :',rmse)\n\n#r2 : 決定係数\nr2 = r2_score(y_test,y_pred)\nprint('R2 :',r2)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:53.685821Z","iopub.execute_input":"2022-08-04T07:22:53.686566Z","iopub.status.idle":"2022-08-04T07:22:53.696534Z","shell.execute_reply.started":"2022-08-04T07:22:53.686502Z","shell.execute_reply":"2022-08-04T07:22:53.695345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 学習に使用するデータを設定\nlgb_train = lgb.Dataset(X_train, y_train)\nlgb_eval = lgb.Dataset(X_test, y_test, reference=lgb_train)\n\n# LightGBM parameters\nparams = {\n        'task': 'train',\n        'boosting_type': 'gbdt',\n        'objective': 'regression', # 目的 : 回帰  \n        'metric': {'rmse'}, # 評価指標 : rsme(平均二乗誤差の平方根) \n        'learning_rate': 0.1,\n        'num_leaves': 23,\n        'min_data_in_leaf': 1,\n        'num_iteration': 1000, #1000回学習\n        'verbose': 0\n}\n\n# モデルの学習\nmodel = lgb.train(params, # パラメータ\n            train_set=lgb_train, # トレーニングデータの指定\n            valid_sets=lgb_eval, # 検証データの指定\n            early_stopping_rounds=100 # 100回ごとに検証精度の改善を検討　→ 精度が改善しないなら学習を終了(過学習に陥るのを防ぐ)\n               )\n\n# テストデータの予測\ny_pred = model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:53.697667Z","iopub.execute_input":"2022-08-04T07:22:53.697988Z","iopub.status.idle":"2022-08-04T07:22:59.813661Z","shell.execute_reply.started":"2022-08-04T07:22:53.697959Z","shell.execute_reply":"2022-08-04T07:22:59.812723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 真値と予測値の表示\ndf_pred = pd.DataFrame({\"Next Week's Deaths\":y_test,\"Next Week's Deaths_pred\":y_pred})\ndisplay(df_pred)\n\n# 散布図を描画(真値 vs 予測値)\nplt.plot(y_test, y_test, color = 'red', label = 'x=y') # 直線y = x (真値と予測値が同じ場合は直線状に点がプロットされる)\nplt.scatter(y_test, y_pred) # 散布図のプロット\nplt.xlabel('y') # x軸ラベル\nplt.ylabel('y_test') # y軸ラベル\nplt.title('y vs y_pred') # グラフタイトル","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:22:59.815192Z","iopub.execute_input":"2022-08-04T07:22:59.815932Z","iopub.status.idle":"2022-08-04T07:23:00.110087Z","shell.execute_reply.started":"2022-08-04T07:22:59.815894Z","shell.execute_reply":"2022-08-04T07:23:00.109260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# モデル評価\n# rmse : 平均二乗誤差の平方根\nmse = mean_squared_error(y_test, y_pred) # MSE(平均二乗誤差)の算出\nrmse = np.sqrt(mse) # RSME = √MSEの算出\nprint('RMSE :',rmse)\n\n#r2 : 決定係数\nr2 = r2_score(y_test,y_pred)\nprint('R2 :',r2)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:23:00.114214Z","iopub.execute_input":"2022-08-04T07:23:00.114837Z","iopub.status.idle":"2022-08-04T07:23:00.122687Z","shell.execute_reply.started":"2022-08-04T07:23:00.114801Z","shell.execute_reply":"2022-08-04T07:23:00.121568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:23:00.123944Z","iopub.execute_input":"2022-08-04T07:23:00.124260Z","iopub.status.idle":"2022-08-04T07:23:00.133164Z","shell.execute_reply.started":"2022-08-04T07:23:00.124230Z","shell.execute_reply":"2022-08-04T07:23:00.132351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from numpy import random\nimport pandas as pd\n\nrandom.seed(5)\nrandom.randint(100, size=(1, 1))\ndata_array = random.randint(100, size=(4, 3))\n\nprint(\"NumPy Data Array is:\")\nprint(y_pred)\n\nprint(\"\")\n\ndata_df = pd.DataFrame(y_pred)\nprint(\"The DataFrame generated from the NumPy array is:\")\nprint(data_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T07:36:57.447219Z","iopub.execute_input":"2022-08-04T07:36:57.447633Z","iopub.status.idle":"2022-08-04T07:36:57.458710Z","shell.execute_reply.started":"2022-08-04T07:36:57.447597Z","shell.execute_reply":"2022-08-04T07:36:57.457568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}