{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-17T13:09:04.13361Z","iopub.execute_input":"2022-06-17T13:09:04.134551Z","iopub.status.idle":"2022-06-17T13:09:04.168Z","shell.execute_reply.started":"2022-06-17T13:09:04.134433Z","shell.execute_reply":"2022-06-17T13:09:04.167148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# データ読み込み","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/spaceship-titanic/train.csv')\ntest = pd.read_csv('../input/spaceship-titanic/test.csv')\nsubmit = pd.read_csv('../input/spaceship-titanic/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:04.169392Z","iopub.execute_input":"2022-06-17T13:09:04.170279Z","iopub.status.idle":"2022-06-17T13:09:04.274102Z","shell.execute_reply.started":"2022-06-17T13:09:04.17024Z","shell.execute_reply":"2022-06-17T13:09:04.273126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:04.27554Z","iopub.execute_input":"2022-06-17T13:09:04.276082Z","iopub.status.idle":"2022-06-17T13:09:04.319715Z","shell.execute_reply.started":"2022-06-17T13:09:04.276039Z","shell.execute_reply":"2022-06-17T13:09:04.318428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:04.322359Z","iopub.execute_input":"2022-06-17T13:09:04.322885Z","iopub.status.idle":"2022-06-17T13:09:04.342419Z","shell.execute_reply.started":"2022-06-17T13:09:04.322834Z","shell.execute_reply":"2022-06-17T13:09:04.341105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:04.343932Z","iopub.execute_input":"2022-06-17T13:09:04.344912Z","iopub.status.idle":"2022-06-17T13:09:04.365094Z","shell.execute_reply.started":"2022-06-17T13:09:04.344868Z","shell.execute_reply":"2022-06-17T13:09:04.364242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install sweetviz\nimport sweetviz as sv","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:04.366395Z","iopub.execute_input":"2022-06-17T13:09:04.367491Z","iopub.status.idle":"2022-06-17T13:09:20.545765Z","shell.execute_reply.started":"2022-06-17T13:09:04.367437Z","shell.execute_reply":"2022-06-17T13:09:20.544528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # train のEDA\n# my_report_train = sv.analyze(train)\n# my_report_train.show_html(\"sweetviz_report_Spaceship_train_V1.html\")\n\n# # train と test の関係\n# my_report_trainVStest = sv.compare([train, \"Train\"], [test, \"Test\"], \"Transported\")\n# my_report_trainVStest.show_html(\"sweetviz_report_Spaceship_trainVStest_V1.html\")","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:20.547944Z","iopub.execute_input":"2022-06-17T13:09:20.548479Z","iopub.status.idle":"2022-06-17T13:09:20.553911Z","shell.execute_reply.started":"2022-06-17T13:09:20.548432Z","shell.execute_reply":"2022-06-17T13:09:20.552788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 前処理 \n1. NaNがあるかないか\n2. Cabin分裂(deck(encoding),side(encoding),num(そのまま))\n3. サービス系(RoomService, FoodCourt, ShoppingMall, Spa, VRDeck)の欠損値をLightGBMで予測して補完\n4. HomePlanetとDestination合併\n3. 家族(nameから)\n4. 同室人数\n5. カテゴリ変数の欠損値補完\n6. カテゴリ変数の変換 (HomePlanet・Destination・CryoSleep・VIP・Transportedを数値変換)\n7. 同室確認\n8. サービス料合計\n9. 不要な列を削除\n10. clipping\n11. binning","metadata":{}},{"cell_type":"code","source":"df = pd.concat([train.drop([\"Transported\"], axis=1), test], ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:20.555351Z","iopub.execute_input":"2022-06-17T13:09:20.555723Z","iopub.status.idle":"2022-06-17T13:09:20.580797Z","shell.execute_reply.started":"2022-06-17T13:09:20.555685Z","shell.execute_reply":"2022-06-17T13:09:20.579851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#　NaNがあるかないか\ndf_colmns_list = df.drop([\"PassengerId\"], axis=1).columns\n\nfor column in df_colmns_list:\n    df[\"Nan_\"+ column] = np.where(df[column].isna(), 1, 0)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:20.582162Z","iopub.execute_input":"2022-06-17T13:09:20.583099Z","iopub.status.idle":"2022-06-17T13:09:20.615055Z","shell.execute_reply.started":"2022-06-17T13:09:20.583061Z","shell.execute_reply":"2022-06-17T13:09:20.613816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cabin分裂(deck(encoding),side(encoding),num(そのまま))\n\nCabinAry_df = df[\"Cabin\"].str.split(\"/\", expand=True)\n\ndf[\"Cabin_Deck\"] = CabinAry_df[0]\ndf[\"Cabin_Num\"] = CabinAry_df[1]\ndf[\"Cabin_Side\"] = CabinAry_df[2]\n\n# Cabin_Num がoblect型になっていてlightgbmに突っ込めないからfloat型にする\ndf[\"Cabin_Num\"] = df[\"Cabin_Num\"].astype(float)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:20.619324Z","iopub.execute_input":"2022-06-17T13:09:20.619709Z","iopub.status.idle":"2022-06-17T13:09:20.658143Z","shell.execute_reply.started":"2022-06-17T13:09:20.619674Z","shell.execute_reply":"2022-06-17T13:09:20.65746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgbm\nfrom lightgbm import early_stopping, log_evaluation\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nimport warnings\nwarnings.simplefilter('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:20.659185Z","iopub.execute_input":"2022-06-17T13:09:20.660566Z","iopub.status.idle":"2022-06-17T13:09:21.871155Z","shell.execute_reply.started":"2022-06-17T13:09:20.660512Z","shell.execute_reply":"2022-06-17T13:09:21.869887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 欠損値学習のためのLabelEncording\n\nNaN_cat_columns_df = [\"HomePlanet\",\"CryoSleep\",\"Cabin_Deck\"]\nNaN_drop_list = [\"PassengerId\",\"Cabin\",'Nan_HomePlanet', 'Nan_CryoSleep', 'Nan_Cabin', 'Nan_Destination',\n                 'Nan_Age', 'Nan_VIP', 'Nan_RoomService', 'Nan_FoodCourt','Nan_ShoppingMall',\n                 'Nan_Spa', 'Nan_VRDeck', 'Nan_Name',\"Cabin_Side\",\"Destination\",\"VIP\",\"Name\"]\ndf_NaN = df.drop(NaN_drop_list, axis=1) # 欠損値学習に使わないカラムを落とす\n\nfor c in NaN_cat_columns_df:\n    le = LabelEncoder()\n    le.fit(df_NaN[c])\n    df_NaN[c] = le.transform(df_NaN[c])","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:21.872619Z","iopub.execute_input":"2022-06-17T13:09:21.873327Z","iopub.status.idle":"2022-06-17T13:09:21.902995Z","shell.execute_reply.started":"2022-06-17T13:09:21.873287Z","shell.execute_reply":"2022-06-17T13:09:21.901892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# サービス系(RoomService, FoodCourt, ShoppingMall, Spa, VRDeck)の欠損値をLightGBMで予測して補完\n\nService_list = [\"RoomService\", \"FoodCourt\", \"ShoppingMall\", \"Spa\", \"VRDeck\"]\n\nfor column_service in Service_list:\n    NonExist_NaN_df = df_NaN[df_NaN[column_service].notna()]  # columnにNaNがないdf\n    Only_NaN_df = df_NaN[df_NaN[column_service].isna()]  # columnがNaNのみのdf\n\n    NaN_train_X_row = NonExist_NaN_df.drop([column_service], axis=1)\n    NaN_train_y_row = NonExist_NaN_df[column_service]\n    NaN_test_X = Only_NaN_df.drop([column_service], axis=1)\n    NaN_test_y = Only_NaN_df[column_service]\n\n    NaN_train_X,NaN_valid_X, NaN_train_y, NaN_valid_y = train_test_split(NaN_train_X_row, NaN_train_y_row, test_size=0.25, random_state=42)\n\n    lgb_NaN_train = lgbm.Dataset(NaN_train_X,NaN_train_y)\n    lgb_NaN_valid = lgbm.Dataset(NaN_valid_X,NaN_valid_y)\n\n    params = {\n                    \"objective\": \"regression\", \n                    'metric': 'rmse',\n                    \"learning_rate\": .1,\n                    \"reg_lambda\": .1,\n                    \"reg_alpha\": 0,\n                    \"max_depth\": 5, \n                    \"n_estimators\": 10000, \n                    \"colsample_bytree\": .5, \n                    \"min_child_samples\": 10,\n                    \"subsample_freq\": 3,\n                    \"subsample\": .9,\n                    \"random_state\": 1,\n                    'verbose': -1\n                }\n\n    gbm = lgbm.train(params,\n                     train_set=lgb_NaN_train,\n                     valid_sets=[lgb_NaN_valid],\n                     callbacks=[early_stopping(stopping_rounds=100,\n                                    verbose=False),\n                               log_evaluation(0)]\n                     )\n\n    NaN_valid_y_pred = gbm.predict(NaN_valid_X)\n    NaN_score = mean_squared_error(y_true=NaN_valid_y, y_pred=NaN_valid_y_pred, squared=False)\n    print(f'{column_service}:RMSE={NaN_score}\\n')\n\n    # feature importanceを表示\n    importance = pd.DataFrame(gbm.feature_importance(importance_type='gain'), index=NaN_train_X.columns, columns=['importance'])\n    importance = importance.sort_values('importance', ascending=False)\n    display(importance)\n    print(\"-\" * 50)\n\n    NaN_test_y_pred = gbm.predict(NaN_test_X)\n    NaN_test_y = pd.Series(data=NaN_test_y_pred, index=NaN_test_y.index)\n    \n    df[column_service].fillna(NaN_test_y, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:21.904517Z","iopub.execute_input":"2022-06-17T13:09:21.905054Z","iopub.status.idle":"2022-06-17T13:09:25.818908Z","shell.execute_reply.started":"2022-06-17T13:09:21.905015Z","shell.execute_reply":"2022-06-17T13:09:25.817773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:25.820139Z","iopub.execute_input":"2022-06-17T13:09:25.820738Z","iopub.status.idle":"2022-06-17T13:09:25.853954Z","shell.execute_reply.started":"2022-06-17T13:09:25.820695Z","shell.execute_reply":"2022-06-17T13:09:25.853114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# HomePlanetとDestination合併\n\ndf[\"Home×Dest\"] = df[\"HomePlanet\"] + df[\"Destination\"]","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:25.857215Z","iopub.execute_input":"2022-06-17T13:09:25.857589Z","iopub.status.idle":"2022-06-17T13:09:25.870128Z","shell.execute_reply.started":"2022-06-17T13:09:25.857556Z","shell.execute_reply":"2022-06-17T13:09:25.869205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 家族(nameから)\n\ndf[\"Family\"] = df[\"Name\"].str.split(\" \", expand=True)[1]","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:25.87179Z","iopub.execute_input":"2022-06-17T13:09:25.872431Z","iopub.status.idle":"2022-06-17T13:09:25.927568Z","shell.execute_reply.started":"2022-06-17T13:09:25.872384Z","shell.execute_reply":"2022-06-17T13:09:25.92676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 同室人数\n\ncabin_group = df.groupby(\"Cabin\")\ndf_Sameroom = pd.DataFrame({\"SameRoomNum\":cabin_group.size()})\ndf = pd.merge(df,df_Sameroom,how=\"left\",on=\"Cabin\")","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:25.9288Z","iopub.execute_input":"2022-06-17T13:09:25.929234Z","iopub.status.idle":"2022-06-17T13:09:25.975102Z","shell.execute_reply.started":"2022-06-17T13:09:25.929204Z","shell.execute_reply":"2022-06-17T13:09:25.974193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 同室確認\n\ndf[\"SameRoomBinary\"] = np.where((df[\"SameRoomNum\"]==1) | (df[\"SameRoomNum\"].isna()), 0, 1)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:25.976647Z","iopub.execute_input":"2022-06-17T13:09:25.977365Z","iopub.status.idle":"2022-06-17T13:09:25.985001Z","shell.execute_reply.started":"2022-06-17T13:09:25.977321Z","shell.execute_reply":"2022-06-17T13:09:25.984196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  サービス料合計\n\ndf[\"Service_Sum\"] = df[Service_list].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:25.986288Z","iopub.execute_input":"2022-06-17T13:09:25.986605Z","iopub.status.idle":"2022-06-17T13:09:26.013059Z","shell.execute_reply.started":"2022-06-17T13:09:25.986576Z","shell.execute_reply":"2022-06-17T13:09:26.011758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# カテゴリ変数の変換 (HomePlanet,Destination,CryoSleep,VIP,Cabin_Deck,Cabin_Side,Home×Dest,Family,Transported を数値変換)\n\nfrom sklearn.preprocessing import LabelEncoder\n\ncat_columns_df = [\"HomePlanet\",\"Destination\",\"CryoSleep\",\"VIP\",\"Cabin_Deck\",\"Cabin_Side\",\"Home×Dest\",\"Family\"]\n\nfor c in cat_columns_df:\n    le = LabelEncoder()\n    le.fit(df[c])\n    df[c] = le.transform(df[c])","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:26.014744Z","iopub.execute_input":"2022-06-17T13:09:26.01521Z","iopub.status.idle":"2022-06-17T13:09:26.07574Z","shell.execute_reply.started":"2022-06-17T13:09:26.015168Z","shell.execute_reply":"2022-06-17T13:09:26.075029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 前処理後の train のEDA\nmy_report_train = sv.analyze(train)\nmy_report_train.show_html(\"sweetviz_report_Spaceship_train_V2.html\")\n\n# 前処理後の train と test の関係\nmy_report_trainVStest = sv.compare([train, \"Train\"], [test, \"Test\"], \"Transported\")\nmy_report_trainVStest.show_html(\"sweetviz_report_Spaceship_trainVStest_V2.html\")","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:26.077124Z","iopub.execute_input":"2022-06-17T13:09:26.077721Z","iopub.status.idle":"2022-06-17T13:09:44.009379Z","shell.execute_reply.started":"2022-06-17T13:09:26.077679Z","shell.execute_reply":"2022-06-17T13:09:44.007243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:44.011278Z","iopub.execute_input":"2022-06-17T13:09:44.011881Z","iopub.status.idle":"2022-06-17T13:09:44.020062Z","shell.execute_reply.started":"2022-06-17T13:09:44.011834Z","shell.execute_reply":"2022-06-17T13:09:44.019024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 不要な列を削除\ndrop_list = ['PassengerId', 'Cabin', 'Name', \"VIP\", \"Destination\", \"HomePlanet\", \"SameRoomNum\"]\n\ndf.drop(drop_list, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:44.021367Z","iopub.execute_input":"2022-06-17T13:09:44.021742Z","iopub.status.idle":"2022-06-17T13:09:44.042833Z","shell.execute_reply.started":"2022-06-17T13:09:44.021709Z","shell.execute_reply":"2022-06-17T13:09:44.041472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:44.04504Z","iopub.execute_input":"2022-06-17T13:09:44.045444Z","iopub.status.idle":"2022-06-17T13:09:44.08644Z","shell.execute_reply.started":"2022-06-17T13:09:44.045405Z","shell.execute_reply":"2022-06-17T13:09:44.085196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# clipping","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:44.088453Z","iopub.execute_input":"2022-06-17T13:09:44.090144Z","iopub.status.idle":"2022-06-17T13:09:44.095307Z","shell.execute_reply.started":"2022-06-17T13:09:44.090084Z","shell.execute_reply":"2022-06-17T13:09:44.094339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# binning","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:44.09703Z","iopub.execute_input":"2022-06-17T13:09:44.098397Z","iopub.status.idle":"2022-06-17T13:09:44.108919Z","shell.execute_reply.started":"2022-06-17T13:09:44.098345Z","shell.execute_reply":"2022-06-17T13:09:44.107628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 学習","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import accuracy_score, f1_score, log_loss","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:44.111033Z","iopub.execute_input":"2022-06-17T13:09:44.111541Z","iopub.status.idle":"2022-06-17T13:09:44.19536Z","shell.execute_reply.started":"2022-06-17T13:09:44.111491Z","shell.execute_reply":"2022-06-17T13:09:44.194608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X = df.iloc[:train.shape[0],]\ntrain_y = train[\"Transported\"].astype(int)\ntest = df.iloc[train.shape[0]:,:]","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:44.198726Z","iopub.execute_input":"2022-06-17T13:09:44.199845Z","iopub.status.idle":"2022-06-17T13:09:44.205965Z","shell.execute_reply.started":"2022-06-17T13:09:44.199797Z","shell.execute_reply":"2022-06-17T13:09:44.204876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class LightGBMCV:\n    def __init__(self, fold, params=None):\n        if params is None:\n            self.params = {\n                \"objective\": \"binary\", \n                \"learning_rate\": .1,\n                \"reg_lambda\": .1,\n                \"reg_alpha\": 0,\n                \"max_depth\": 5, \n                \"n_estimators\": 10000, \n                \"colsample_bytree\": .5, \n                \"min_child_samples\": 10,\n                \"subsample_freq\": 3,\n                \"subsample\": .9,\n                \"importance_type\": \"gain\", \n                \"random_state\": 1\n            }\n        else:\n            self.params = params\n        self.fold = fold\n    \n    @property\n    def models(self):\n        return self._models\n    \n    @property\n    def pred_array(self):\n        return self._pred_array\n    \n    def fit(self, X, y, early_stopping, score_func, **kwargs):\n        self._feature_name = X.columns\n        X, y = X.values, y.values\n        self._models = []\n        self._pred_array = np.zeros(len(y), dtype=np.float32)\n            \n        cv = self.fold.split(X, y)\n        for i, (idx_train, idx_valid) in enumerate(cv):\n            X_train, y_train = X[idx_train], y[idx_train]\n            X_valid, y_valid = X[idx_valid], y[idx_valid]\n            \n            model = lgbm.LGBMModel(**self.params)\n            model.fit(\n                X_train,\n                y_train,\n                eval_set=[(X_valid, y_valid)],\n                callbacks=[early_stopping, log_evaluation(period=0, show_stdv=False)]\n            )\n            self._models.append(model)\n            y_pred = model.predict(X_valid, **kwargs)\n            self._pred_array[idx_valid] = y_pred\n            if score_func in [accuracy_score, f1_score]:\n                score = score_func(y_valid, np.where(y_pred >= 0.5, 1, 0), **kwargs)\n            else:\n                score = score_func(y_valid, y_pred, **kwargs)\n            print(f\" - fold{i + 1} - {score: .4f}\")\n        \n        if score_func in [accuracy_score, f1_score]:\n            total_score = score_func(y, np.rint(self._pred_array), **kwargs)\n        else:\n            total_score = score_func(y, self._pred_array, **kwargs)\n        print(f\": {total_score: .4f}\")\n        \n    def predict(self, test):\n        test = test.values\n        pred = np.array([model.predict(test) for model in self._models])\n        pred = np.mean(pred, axis=0)\n        return pred\n    \n    @property\n    def df_feature_importance(self):\n        return self._df_feature_importance\n        \n    def visualize_importance(self, top_num=10):\n        \n        fig, ax = plt.subplots(1, 1, figsize=(max(8, 1.2*top_num), 20))\n        \n        self._df_feature_importance = pd.DataFrame()\n        for idx, clf  in enumerate(self._models):\n            _df = pd.DataFrame()\n            _df[\"feature_importance\"] = clf.feature_importances_\n            _df[\"feature_name\"] = self._feature_name\n            _df[\"fold\"] = idx + 1\n            self._df_feature_importance = pd.concat([self._df_feature_importance, _df])\n\n        order = self._df_feature_importance.groupby(\"feature_name\")[\"feature_importance\"].sum()\\\n                                .sort_values(ascending=False).index[:top_num]\n\n        sns.boxenplot(\n            x=\"feature_importance\", \n            y=\"feature_name\", \n            data=self._df_feature_importance, \n            order=order, ax=ax\n        )\n        ax.grid()\n        \n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:44.207932Z","iopub.execute_input":"2022-06-17T13:09:44.208255Z","iopub.status.idle":"2022-06-17T13:09:44.235914Z","shell.execute_reply.started":"2022-06-17T13:09:44.208227Z","shell.execute_reply":"2022-06-17T13:09:44.234466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# fit","metadata":{}},{"cell_type":"code","source":"fold = StratifiedKFold(n_splits=5, shuffle=True, random_state=0)\nmodel = LightGBMCV(fold=fold)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:44.237732Z","iopub.execute_input":"2022-06-17T13:09:44.238085Z","iopub.status.idle":"2022-06-17T13:09:44.249687Z","shell.execute_reply.started":"2022-06-17T13:09:44.238056Z","shell.execute_reply":"2022-06-17T13:09:44.248744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel.fit(\n    train_X, \n    train_y, \n    early_stopping=early_stopping(100, verbose=False),\n    score_func=accuracy_score\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:44.251345Z","iopub.execute_input":"2022-06-17T13:09:44.251972Z","iopub.status.idle":"2022-06-17T13:09:46.789972Z","shell.execute_reply.started":"2022-06-17T13:09:44.251927Z","shell.execute_reply":"2022-06-17T13:09:46.789219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_prob = model.predict(test)\n\nfig, ax = plt.subplots(1, 1, figsize=(10, 6))\nsns.histplot(x=model.pred_array, bins=50, alpha=0.5, ax=ax, stat=\"density\", label=\"Out Of Fold (Train)\", color=\"tab:blue\")\nsns.histplot(x=predict_prob, bins=50, alpha=0.5, ax=ax, stat=\"density\", label=\"Test\", color=\"tab:red\")\n\nax.set_title(\"Probability Density\")\nax.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:46.791513Z","iopub.execute_input":"2022-06-17T13:09:46.792245Z","iopub.status.idle":"2022-06-17T13:09:47.502734Z","shell.execute_reply.started":"2022-06-17T13:09:46.792202Z","shell.execute_reply":"2022-06-17T13:09:47.501726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.visualize_importance(top_num=25)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:47.503982Z","iopub.execute_input":"2022-06-17T13:09:47.504633Z","iopub.status.idle":"2022-06-17T13:09:48.298774Z","shell.execute_reply.started":"2022-06-17T13:09:47.504597Z","shell.execute_reply":"2022-06-17T13:09:48.297644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_prob","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:48.300383Z","iopub.execute_input":"2022-06-17T13:09:48.301447Z","iopub.status.idle":"2022-06-17T13:09:48.31027Z","shell.execute_reply.started":"2022-06-17T13:09:48.301397Z","shell.execute_reply":"2022-06-17T13:09:48.308935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# spaceship_tiatanic.competition_submit(\n#     submit.assign(Transported=np.where(predict_prob >= 0.5, True, False)),\n#     message=\"cv; StratifiedKfold(5) features; add CabinCount DestinationFromDepature AgeByHomePlanetMean AgeByHomePlanetStd LuxuryBilledAmount LuxuryBilledCount, without VIP from Baseline\",\n#     file_name=\"7th_sub\",\n#     path=\"submission\"\n# )","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:48.312595Z","iopub.execute_input":"2022-06-17T13:09:48.31335Z","iopub.status.idle":"2022-06-17T13:09:48.319569Z","shell.execute_reply.started":"2022-06-17T13:09:48.313298Z","shell.execute_reply":"2022-06-17T13:09:48.318735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = submit.assign(Transported=np.where(predict_prob >= 0.5, True, False))\n\nsub.to_csv(\"spaceship_StratifiedKFold_5-fold_CV_Yuiki's_Model_Service_fullNaN.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-17T13:09:48.320953Z","iopub.execute_input":"2022-06-17T13:09:48.321481Z","iopub.status.idle":"2022-06-17T13:09:48.346633Z","shell.execute_reply.started":"2022-06-17T13:09:48.321438Z","shell.execute_reply":"2022-06-17T13:09:48.345492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 課題\n1. trainとtestをdfでまとめて特徴量を作る\n2. for分を減らして実行速度上げる\n3. ","metadata":{}}]}