{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":27783,"databundleVersionId":2555459,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Python3.10 だと mlb モジュールが使えないため、3.7にする\n# 参考: # 参考: https://www.kaggle.com/discussions/general/173536\n# Internet on でないとインストールができないが、 Internet off　でないとノートを提出できない\n# 詰んでいる\n!conda create -n newCondaEnvironment -c cctbx202208 -y\n!source /opt/conda/bin/activate newCondaEnvironment && conda install -c cctbx202208 python=3.7 -y","metadata":{"execution":{"iopub.status.busy":"2024-04-18T22:54:17.154601Z","iopub.execute_input":"2024-04-18T22:54:17.154993Z","iopub.status.idle":"2024-04-18T22:55:52.495504Z","shell.execute_reply.started":"2024-04-18T22:54:17.154955Z","shell.execute_reply":"2024-04-18T22:55:52.494003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-18T00:47:54.644425Z","iopub.execute_input":"2024-04-18T00:47:54.644908Z","iopub.status.idle":"2024-04-18T00:47:55.803212Z","shell.execute_reply.started":"2024-04-18T00:47:54.644873Z","shell.execute_reply":"2024-04-18T00:47:55.802004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nimport pickle\nimport os\nimport datetime as dt\n\nimport matplotlib.pyplot as plt\n\nimport lightgbm as lgb\n\nfrom sklearn.metrics import mean_absolute_error\n\nimport warnings\nwarnings.simplefilter(\"ignore\")\n\n# https://stackoverflow.com/a/8885688\npd.options.display.float_format = \"{:10.4f}\".format","metadata":{"execution":{"iopub.status.busy":"2024-04-18T00:47:55.805349Z","iopub.execute_input":"2024-04-18T00:47:55.805907Z","iopub.status.idle":"2024-04-18T00:47:58.465031Z","shell.execute_reply.started":"2024-04-18T00:47:55.805874Z","shell.execute_reply":"2024-04-18T00:47:58.464130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/mlb-player-digital-engagement-forecasting/train_updated.csv\")\nprint(train.shape)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T00:47:58.471845Z","iopub.execute_input":"2024-04-18T00:47:58.472253Z","iopub.status.idle":"2024-04-18T00:49:50.143352Z","shell.execute_reply.started":"2024-04-18T00:47:58.472205Z","shell.execute_reply":"2024-04-18T00:49:50.142070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 試行錯誤の速度向上のため20200401以降にしぼる\ntrain = train.loc[train[\"date\"] >= 20200401, :].reset_index(drop=True)\nprint(train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-18T00:49:50.144835Z","iopub.execute_input":"2024-04-18T00:49:50.145252Z","iopub.status.idle":"2024-04-18T00:49:50.159208Z","shell.execute_reply.started":"2024-04-18T00:49:50.145205Z","shell.execute_reply":"2024-04-18T00:49:50.158011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def unpack_json(json_str):\n    return np.nan if pd.isna(json_str) else pd.read_json(json_str)\n\ndef extract_data(\n    input_df,\n    col=\"events\",\n    show=False\n):\n    output_df = pd.DataFrame()\n    \n    # なぜrangeでなくarangeを使うのか謎だったが右が参考になる https://stackoverflow.com/a/75533644\n    for i in np.arange(len(input_df)):\n        if show:\n            print(\"\\r{}/{}\".format(i+1, len(input_df), end=\"\"))\n        try:\n            output_df = pd.concat(\n                [\n                    output_df,\n                    unpack_json(input_df[col].iloc[i]),\n                ],\n                axis=0,\n                ignore_index=True,\n            )\n        except:\n            pass\n    \n    if show:\n        print(\"\")\n        print(output_df.shape)\n        display(output_df.head())\n\n    return output_df","metadata":{"execution":{"iopub.status.busy":"2024-04-18T00:49:50.160994Z","iopub.execute_input":"2024-04-18T00:49:50.161418Z","iopub.status.idle":"2024-04-18T00:49:50.173780Z","shell.execute_reply.started":"2024-04-18T00:49:50.161380Z","shell.execute_reply":"2024-04-18T00:49:50.172551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_engagement = extract_data(train, col=\"nextDayPlayerEngagement\", show=True)","metadata":{"_kg_hide-output":false,"execution":{"iopub.status.busy":"2024-04-18T00:49:50.175402Z","iopub.execute_input":"2024-04-18T00:49:50.176407Z","iopub.status.idle":"2024-04-18T00:50:09.888580Z","shell.execute_reply.started":"2024-04-18T00:49:50.176374Z","shell.execute_reply":"2024-04-18T00:50:09.887603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_engagement[\"date_playerId\"] = df_engagement[\"engagementMetricsDate\"].str.replace(\"-\", \"\") + \"_\" + df_engagement[\"playerId\"].astype(str)\ndf_engagement.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T00:50:09.890212Z","iopub.execute_input":"2024-04-18T00:50:09.891018Z","iopub.status.idle":"2024-04-18T00:50:11.477750Z","shell.execute_reply.started":"2024-04-18T00:50:09.890978Z","shell.execute_reply":"2024-04-18T00:50:11.476505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 推論実施日＝推論対象日の前日\ndf_engagement[\"date\"] = pd.to_datetime(df_engagement[\"engagementMetricsDate\"], format=\"%Y-%m-%d\") + dt.timedelta(days=-1)\n# 推論実施日から曜日と年月の特徴量作成\ndf_engagement[\"dayofweek\"] = df_engagement[\"date\"].dt.dayofweek\ndf_engagement[\"yearmonth\"] = df_engagement[\"date\"].astype(str).apply(lambda x: x[:7])\ndf_engagement.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T00:50:11.479375Z","iopub.execute_input":"2024-04-18T00:50:11.479847Z","iopub.status.idle":"2024-04-18T00:50:13.415850Z","shell.execute_reply.started":"2024-04-18T00:50:11.479806Z","shell.execute_reply":"2024-04-18T00:50:13.414673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# players.csv\ndf_players = pd.read_csv(\"/kaggle/input/mlb-player-digital-engagement-forecasting/players.csv\")\nprint(df_players.shape)\nprint(df_players[\"playerId\"].agg(\"nunique\"))\ndf_players.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T00:50:13.419772Z","iopub.execute_input":"2024-04-18T00:50:13.420144Z","iopub.status.idle":"2024-04-18T00:50:13.459025Z","shell.execute_reply.started":"2024-04-18T00:50:13.420111Z","shell.execute_reply":"2024-04-18T00:50:13.457945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_players[\"playerForTestSetAndFuturePreds\"] = np.where(df_players[\"playerForTestSetAndFuturePreds\"] == True, 1, 0)\nprint(df_players[\"playerForTestSetAndFuturePreds\"].sum())\nprint(df_players[\"playerForTestSetAndFuturePreds\"].mean())","metadata":{"execution":{"iopub.status.busy":"2024-04-18T00:50:13.460453Z","iopub.execute_input":"2024-04-18T00:50:13.461300Z","iopub.status.idle":"2024-04-18T00:50:13.469625Z","shell.execute_reply.started":"2024-04-18T00:50:13.461260Z","shell.execute_reply":"2024-04-18T00:50:13.468520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_engagement.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T00:50:13.471054Z","iopub.execute_input":"2024-04-18T00:50:13.471449Z","iopub.status.idle":"2024-04-18T00:50:13.492070Z","shell.execute_reply.started":"2024-04-18T00:50:13.471419Z","shell.execute_reply":"2024-04-18T00:50:13.490794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_players.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T00:50:13.493411Z","iopub.execute_input":"2024-04-18T00:50:13.493816Z","iopub.status.idle":"2024-04-18T00:50:13.511012Z","shell.execute_reply.started":"2024-04-18T00:50:13.493785Z","shell.execute_reply":"2024-04-18T00:50:13.509585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_engagement に df_players を LEFT JOIN\ndf_train = pd.merge(df_engagement, df_players, on=[\"playerId\"], how=\"left\")\nprint(df_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-18T00:50:13.512761Z","iopub.execute_input":"2024-04-18T00:50:13.513919Z","iopub.status.idle":"2024-04-18T00:50:14.158791Z","shell.execute_reply.started":"2024-04-18T00:50:13.513875Z","shell.execute_reply":"2024-04-18T00:50:14.157589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = df_train[[\n    \"playerId\",\n    \"dayofweek\",\n    \"birthCity\",\n    \"birthStateProvince\",\n    \"birthCountry\",\n    \"heightInches\",\n    \"weight\",\n    \"primaryPositionCode\",\n    \"primaryPositionName\",\n    \"playerForTestSetAndFuturePreds\",\n]]\n\ny_train = df_train[[\n    \"target1\",\n    \"target2\",\n    \"target3\",\n    \"target4\",\n]]\n\nid_train = df_train[[\n    \"engagementMetricsDate\",\n    \"playerId\",\n    \"date_playerId\",\n    \"date\",\n    \"yearmonth\",\n    \"playerForTestSetAndFuturePreds\",\n]]\n\nprint(x_train.shape, y_train.shape, id_train.shape)\nx_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T00:50:14.160347Z","iopub.execute_input":"2024-04-18T00:50:14.160833Z","iopub.status.idle":"2024-04-18T00:50:14.320290Z","shell.execute_reply.started":"2024-04-18T00:50:14.160793Z","shell.execute_reply":"2024-04-18T00:50:14.319138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lightgbmで扱うためcategory型に\nfor col in [\n    \"playerId\",\n    \"dayofweek\",\n    \"birthCity\",\n    \"birthStateProvince\",\n    \"birthCountry\",\n    \"primaryPositionCode\",\n    \"primaryPositionName\",\n]:\n    x_train[col] = x_train[col].astype(\"category\")","metadata":{"execution":{"iopub.status.busy":"2024-04-18T00:50:14.321800Z","iopub.execute_input":"2024-04-18T00:50:14.322453Z","iopub.status.idle":"2024-04-18T00:50:14.960936Z","shell.execute_reply.started":"2024-04-18T00:50:14.322413Z","shell.execute_reply":"2024-04-18T00:50:14.959769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.sort_values(\"date\").tail()","metadata":{"execution":{"iopub.status.busy":"2024-04-18T00:50:14.962556Z","iopub.execute_input":"2024-04-18T00:50:14.963045Z","iopub.status.idle":"2024-04-18T00:50:15.333804Z","shell.execute_reply.started":"2024-04-18T00:50:14.963007Z","shell.execute_reply":"2024-04-18T00:50:15.332987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 8.3.4 バリデーション設計\n\n## 説明変数として使って良いデータはなにか\n- デジタルエンゲージメント（target1, target2, ...）は2021/7/31ぶんまで利用できる。それ以降の日付のぶんは利用できない\n    - つまり説明変数として使う場合、8/1は前日より過去の値、8/2なら2日前よりも過去の値、・・・、8/31なら31日前より過去の値を利用できる\n    - 31パターンのモデルを作成する手もあるが管理が大変なので、ベースラインでは1モデルとし、「先月のデジタルエンゲージメントなら利用可能」とする\n\n## 古いデータは学習に使うべきか\n- 著者の環境では期間を増やすほど精度が改善したとのことだが、ベースラインでは「直近1年」(=検証データ含め13ヶ月)とする\n\n## 学習用データセットから検証データをどう作るか\n- ベースラインでは検証データ期間を「2021/5」「2021/6」「2021/7」のパターンにわける。パターンごとに、直近12ヶ月を学習データとする","metadata":{}},{"cell_type":"code","source":"# 学習データと検証データの期間のリスト\nlist_cv_month = [\n    [\n        [\n            \"2020-05\",\n            \"2020-06\",\n            \"2020-07\",\n            \"2020-08\",\n            \"2020-09\",\n            \"2020-10\",\n            \"2020-11\",\n            \"2020-12\",\n            \"2021-01\",\n            \"2021-02\",\n            \"2021-03\",\n            \"2021-04\",\n        ],\n        [\n            \"2021-05\",\n        ],\n    ],\n    [\n        [\n            \"2020-06\",\n            \"2020-07\",\n            \"2020-08\",\n            \"2020-09\",\n            \"2020-10\",\n            \"2020-11\",\n            \"2020-12\",\n            \"2021-01\",\n            \"2021-02\",\n            \"2021-03\",\n            \"2021-04\",\n            \"2021-05\",\n        ],\n        [\n            \"2021-06\",\n        ],\n    ],\n    [\n        [\n            \"2020-07\",\n            \"2020-08\",\n            \"2020-09\",\n            \"2020-10\",\n            \"2020-11\",\n            \"2020-12\",\n            \"2021-01\",\n            \"2021-02\",\n            \"2021-03\",\n            \"2021-04\",\n            \"2021-05\",\n            \"2021-06\",\n        ],\n        [\n            \"2021-07\",        \n        ],\n    ]\n]","metadata":{"execution":{"iopub.status.busy":"2024-04-18T00:50:15.335033Z","iopub.execute_input":"2024-04-18T00:50:15.335842Z","iopub.status.idle":"2024-04-18T00:50:15.343911Z","shell.execute_reply.started":"2024-04-18T00:50:15.335811Z","shell.execute_reply":"2024-04-18T00:50:15.342756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_lgb(\n    input_x,\n    input_y,\n    input_id,\n    params: dict,\n    list_nfold=[0, 1, 2],\n    mode_train=\"train\",\n):\n    # 推論値を格納する変数\n    df_valid_pred = pd.DataFrame()\n    # 評価値を入れる変数\n    metrics = []\n    # 重要度を格納する変数\n    df_imp = pd.DataFrame()\n    \n    # corss validation 用のインデクス作成\n    cv = []\n    for month_tr, month_va in list_cv_month:\n        _cv = [\n            input_id.index[input_id[\"yearmonth\"].isin(month_tr)],\n            input_id.index[input_id[\"yearmonth\"].isin(month_va) & (input_id[\"playerForTestSetAndFuturePreds\"] == 1)],\n        ]\n        cv.append(_cv)\n\n    # fold * target 毎にモデル学習\n    for nfold in list_nfold:\n        for i, target in enumerate([\"target1\", \"target2\", \"target3\", \"target4\"]):\n            print(\"-\"*20, target, \", fold:\", nfold, \"-\"*20)\n            \n            idx_tr, idx_va = cv[nfold][0], cv[nfold][1]\n            x_tr, y_tr, id_tr = input_x.loc[idx_tr, :], input_y.loc[idx_tr, target], input_id.loc[idx_tr, :]\n            x_va, y_va, id_va = input_x.loc[idx_va, :], input_y.loc[idx_va, target], input_id.loc[idx_va, :]\n            print(x_tr.shape, y_tr.shape, id_tr.shape)\n            print(x_va.shape, y_va.shape, id_va.shape)\n            \n            # 保存するモデルのファイル名\n            filepath = \"model_lgb_{}_fold{}.h5\".format(target, nfold)\n            \n            if mode_train == \"train\":\n                print(\"training start.\")\n                model = lgb.LGBMRegressor(**params)\n                model.fit(\n                    x_tr,\n                    y_tr,\n                    eval_set=[(x_tr, y_tr), (x_va, y_va)],\n                    callbacks=[lgb.early_stopping(stopping_rounds=50),],\n                )\n                with open(filepath, \"wb\") as f:\n                    pickle.dump(model, f, protocol=4)\n            else:\n                print(\"model load.\")\n                with open(filepath, \"rb\") as f:\n                    model = pickle.load(f)\n                print(\"Done.\")\n            \n            # validの推論値取得\n            y_va_pred = model.predict(x_va)\n            tmp_pred = pd.concat(\n                [\n                    id_va,\n                    pd.DataFrame({\n                        \"target\": target,\n                        \"nfold\": nfold,\n                        \"true\": y_va,\n                        \"pred\": y_va_pred,\n                    })\n                ],\n                axis=1,\n            )\n            df_valid_pred = pd.concat([df_valid_pred, tmp_pred], axis=0, ignore_index=True)\n            \n            # 評価値の算出\n            metric_va = mean_absolute_error(y_va, y_va_pred)\n            metrics.append([target, nfold, metric_va])\n            \n            # 重要度\n            tmp_imp = pd.DataFrame({\n                \"col\": x_tr.columns,\n                \"imp\": model.feature_importances_,\n                \"target\": target,\n                \"nfold\": nfold,\n            })\n            df_imp = pd.concat(\n                [\n                    df_imp,\n                    tmp_imp,\n                ],\n                axis=0,\n                ignore_index=True\n            )\n            \n    print(\"-\"*10, \"result\", \"-\"*10)\n    # 評価値\n    df_metrics = pd.DataFrame(metrics, columns=[\"target\", \"nfold\", \"mae\"])\n    print(\"MCMAE: {:.4f}\".format(df_metrics[\"mae\"].mean()))\n    \n    # validの推論値\n    df_valid_pred_all = pd.pivot_table(\n        df_valid_pred,\n        # プレイヤー * 日付 ごとに\n        index=[\n            \"engagementMetricsDate\",\n            \"playerId\",\n            \"date_playerId\",\n            \"date\",\n            \"yearmonth\",\n            \"playerForTestSetAndFuturePreds\",\n        ],\n        # target * fold ごとの\n        columns=[\n            \"target\",\n            \"nfold\",\n        ],\n        # 真の値と予測値を算出\n        values=[\n            \"true\",\n            \"pred\"\n        ],\n        # なぜ sum を使うのか?\n        aggfunc=np.sum,\n    )\n    df_valid_pred_all.columns = [\n        \"{}_fold{}_{}\".format(j, k, i) for i, j, k in df_valid_pred_all.columns\n    ]\n    df_valid_pred_all = df_valid_pred_all.reset_index(drop=False)\n    \n    return df_valid_pred_all, df_metrics, df_imp","metadata":{"execution":{"iopub.status.busy":"2024-04-18T02:00:09.049053Z","iopub.execute_input":"2024-04-18T02:00:09.050018Z","iopub.status.idle":"2024-04-18T02:00:09.075274Z","shell.execute_reply.started":"2024-04-18T02:00:09.049975Z","shell.execute_reply":"2024-04-18T02:00:09.074317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# モデル学習\nparams = {\n    \"boosting_type\": \"gbdt\",\n    \"objective\": \"regression_l1\",\n    \"metric\": \"mean_absolute_error\",\n    \"learning_rate\": 0.05,\n    \"num_leaves\": 32,\n    \"subsample\": 0.7,\n    \"subsample_freq\": 1,\n    \"feature_fraction\": 0.8,\n    \"min_data_in_leaf\": 50,\n    \"min_sum_hessian_in_leaf\": 50,\n    \"n_estimators\": 1000,\n    \"random_state\": 123,\n    \"importance_type\": \"gain\",\n}\n\ndf_valid_pred, df_metrics, df_imp = train_lgb(\n    x_train,\n    y_train,\n    id_train,\n    params,\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-18T02:00:12.569181Z","iopub.execute_input":"2024-04-18T02:00:12.569929Z","iopub.status.idle":"2024-04-18T02:09:27.217970Z","shell.execute_reply.started":"2024-04-18T02:00:12.569893Z","shell.execute_reply":"2024-04-18T02:09:27.216773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"MCMAE: {:.4f}\".format(df_metrics[\"mae\"].mean()))\ndisplay(\n    pd.pivot_table(\n        df_metrics,\n        index=\"nfold\",\n        columns=\"target\",\n        values=\"mae\",\n        aggfunc=np.mean,\n        margins=True,\n    )\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-18T02:41:22.365348Z","iopub.execute_input":"2024-04-18T02:41:22.365752Z","iopub.status.idle":"2024-04-18T02:41:22.407094Z","shell.execute_reply.started":"2024-04-18T02:41:22.365723Z","shell.execute_reply":"2024-04-18T02:41:22.406004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_imp.groupby([\"col\"])[\"imp\"].agg([\"mean\", \"std\"]).sort_values(\"mean\", ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-18T02:44:40.266579Z","iopub.execute_input":"2024-04-18T02:44:40.266922Z","iopub.status.idle":"2024-04-18T02:44:40.285436Z","shell.execute_reply.started":"2024-04-18T02:44:40.266895Z","shell.execute_reply":"2024-04-18T02:44:40.284015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 8.3.6 モデル推論","metadata":{}},{"cell_type":"code","source":"# ModuleNotFoundError: No module named 'mlb.competition' が出る -> Python3.7でないとだめ\n# 参考： https://www.kaggle.com/competitions/mlb-player-digital-engagement-forecasting/discussion/248839\n\n# 推論時に受け取るデータのフォーマット確認\n# import mlb\n# env = mlb.make_env() # initialize the environment\n# iter_test = env.iter_test() # iterator which loops over each date in test set\n\n# for (test_df, prediction_df) in iter_test:\n#     display(test_df.head())\n#     display(prediction_df.head())\n#     break","metadata":{"execution":{"iopub.status.busy":"2024-04-18T03:54:05.105050Z","iopub.execute_input":"2024-04-18T03:54:05.105485Z","iopub.status.idle":"2024-04-18T03:54:05.157455Z","shell.execute_reply.started":"2024-04-18T03:54:05.105456Z","shell.execute_reply":"2024-04-18T03:54:05.156207Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 3.10\n!python --version","metadata":{"execution":{"iopub.status.busy":"2024-04-18T04:04:30.538268Z","iopub.execute_input":"2024-04-18T04:04:30.538730Z","iopub.status.idle":"2024-04-18T04:04:31.778744Z","shell.execute_reply.started":"2024-04-18T04:04:30.538678Z","shell.execute_reply":"2024-04-18T04:04:31.777203Z"}},"execution_count":null,"outputs":[]}]}