{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install optuna","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-01T03:40:41.237849Z","iopub.execute_input":"2022-08-01T03:40:41.238509Z","iopub.status.idle":"2022-08-01T03:40:54.890109Z","shell.execute_reply.started":"2022-08-01T03:40:41.238415Z","shell.execute_reply":"2022-08-01T03:40:54.889094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport optuna\n\nimport seaborn as sns\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:54.892148Z","iopub.execute_input":"2022-08-01T03:40:54.892493Z","iopub.status.idle":"2022-08-01T03:40:56.396695Z","shell.execute_reply.started":"2022-08-01T03:40:54.892459Z","shell.execute_reply":"2022-08-01T03:40:56.395838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# データを取得\ntrain_df = pd.read_csv('../input/spaceship-titanic/train.csv')\ntest_df = pd.read_csv('../input/spaceship-titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:56.397935Z","iopub.execute_input":"2022-08-01T03:40:56.398219Z","iopub.status.idle":"2022-08-01T03:40:56.479895Z","shell.execute_reply.started":"2022-08-01T03:40:56.398191Z","shell.execute_reply":"2022-08-01T03:40:56.478659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:56.483338Z","iopub.execute_input":"2022-08-01T03:40:56.483778Z","iopub.status.idle":"2022-08-01T03:40:56.518479Z","shell.execute_reply.started":"2022-08-01T03:40:56.483738Z","shell.execute_reply":"2022-08-01T03:40:56.517026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:56.522297Z","iopub.execute_input":"2022-08-01T03:40:56.522667Z","iopub.status.idle":"2022-08-01T03:40:56.543667Z","shell.execute_reply.started":"2022-08-01T03:40:56.522633Z","shell.execute_reply":"2022-08-01T03:40:56.542538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# データの欠損を確認\ntrain_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:56.545171Z","iopub.execute_input":"2022-08-01T03:40:56.545620Z","iopub.status.idle":"2022-08-01T03:40:56.569328Z","shell.execute_reply.started":"2022-08-01T03:40:56.545589Z","shell.execute_reply":"2022-08-01T03:40:56.567682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### trainデータとtestデータを連結","metadata":{}},{"cell_type":"code","source":"# テストデータには\"transported\"項目がないので追加してカラムを揃える\ntest_df[\"Transported\"] = np.nan ","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:56.573368Z","iopub.execute_input":"2022-08-01T03:40:56.574012Z","iopub.status.idle":"2022-08-01T03:40:56.583008Z","shell.execute_reply.started":"2022-08-01T03:40:56.573971Z","shell.execute_reply":"2022-08-01T03:40:56.581272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"train/test\"] = \"train\"\ntest_df[\"train/test\"] = \"test\"","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:56.585706Z","iopub.execute_input":"2022-08-01T03:40:56.586116Z","iopub.status.idle":"2022-08-01T03:40:56.594170Z","shell.execute_reply.started":"2022-08-01T03:40:56.586084Z","shell.execute_reply":"2022-08-01T03:40:56.593126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_df = pd.concat([train_df, test_df])\nall_df = all_df.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:56.595874Z","iopub.execute_input":"2022-08-01T03:40:56.596591Z","iopub.status.idle":"2022-08-01T03:40:56.625315Z","shell.execute_reply.started":"2022-08-01T03:40:56.596445Z","shell.execute_reply":"2022-08-01T03:40:56.624198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:56.629474Z","iopub.execute_input":"2022-08-01T03:40:56.630397Z","iopub.status.idle":"2022-08-01T03:40:56.653898Z","shell.execute_reply.started":"2022-08-01T03:40:56.630352Z","shell.execute_reply":"2022-08-01T03:40:56.652847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# \"PassengerId\"はアンダーバーでgroup_idとid_numに分ける\nall_df[\"group_id\"] = all_df[\"PassengerId\"].str.split(\"_\",expand=True)[0]\nall_df[\"id_num\"] = all_df[\"PassengerId\"].str.split(\"_\",expand=True)[1]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:56.655329Z","iopub.execute_input":"2022-08-01T03:40:56.656001Z","iopub.status.idle":"2022-08-01T03:40:56.911246Z","shell.execute_reply.started":"2022-08-01T03:40:56.655962Z","shell.execute_reply":"2022-08-01T03:40:56.910247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:56.912610Z","iopub.execute_input":"2022-08-01T03:40:56.912895Z","iopub.status.idle":"2022-08-01T03:40:56.936831Z","shell.execute_reply.started":"2022-08-01T03:40:56.912867Z","shell.execute_reply":"2022-08-01T03:40:56.935599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:56.938511Z","iopub.execute_input":"2022-08-01T03:40:56.938829Z","iopub.status.idle":"2022-08-01T03:40:56.975012Z","shell.execute_reply.started":"2022-08-01T03:40:56.938774Z","shell.execute_reply":"2022-08-01T03:40:56.974004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# typeがobjectなのでintに変換\nall_df[\"group_id\"] = all_df[\"group_id\"].astype(int)\nall_df[\"id_num\"] = all_df[\"id_num\"].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:56.976868Z","iopub.execute_input":"2022-08-01T03:40:56.977383Z","iopub.status.idle":"2022-08-01T03:40:56.987297Z","shell.execute_reply.started":"2022-08-01T03:40:56.977348Z","shell.execute_reply":"2022-08-01T03:40:56.986345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:56.990494Z","iopub.execute_input":"2022-08-01T03:40:56.991116Z","iopub.status.idle":"2022-08-01T03:40:57.018544Z","shell.execute_reply.started":"2022-08-01T03:40:56.991085Z","shell.execute_reply":"2022-08-01T03:40:57.017826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:57.019715Z","iopub.execute_input":"2022-08-01T03:40:57.020809Z","iopub.status.idle":"2022-08-01T03:40:57.035915Z","shell.execute_reply.started":"2022-08-01T03:40:57.020762Z","shell.execute_reply":"2022-08-01T03:40:57.035003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:57.037591Z","iopub.execute_input":"2022-08-01T03:40:57.037980Z","iopub.status.idle":"2022-08-01T03:40:57.060662Z","shell.execute_reply.started":"2022-08-01T03:40:57.037939Z","shell.execute_reply":"2022-08-01T03:40:57.059341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# データ整理","metadata":{}},{"cell_type":"markdown","source":"## Homeplanet","metadata":{}},{"cell_type":"code","source":"sns.countplot(all_df[\"HomePlanet\"],hue = all_df[\"Transported\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:57.062856Z","iopub.execute_input":"2022-08-01T03:40:57.065224Z","iopub.status.idle":"2022-08-01T03:40:57.270077Z","shell.execute_reply.started":"2022-08-01T03:40:57.065193Z","shell.execute_reply":"2022-08-01T03:40:57.268973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 欠損を最頻値の\"earth\"で埋める\nall_df[\"HomePlanet\"].fillna(\"Earth\",inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:57.271359Z","iopub.execute_input":"2022-08-01T03:40:57.271624Z","iopub.status.idle":"2022-08-01T03:40:57.279526Z","shell.execute_reply.started":"2022-08-01T03:40:57.271596Z","shell.execute_reply":"2022-08-01T03:40:57.278715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cryosleep","metadata":{}},{"cell_type":"code","source":"sns.countplot(all_df[\"CryoSleep\"],hue = all_df[\"Transported\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:57.280669Z","iopub.execute_input":"2022-08-01T03:40:57.281706Z","iopub.status.idle":"2022-08-01T03:40:57.461167Z","shell.execute_reply.started":"2022-08-01T03:40:57.281664Z","shell.execute_reply":"2022-08-01T03:40:57.460179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 欠損を最頻値の\"false\"で埋める\nall_df[\"CryoSleep\"].fillna(\"False\",inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:57.462339Z","iopub.execute_input":"2022-08-01T03:40:57.462624Z","iopub.status.idle":"2022-08-01T03:40:57.469578Z","shell.execute_reply.started":"2022-08-01T03:40:57.462595Z","shell.execute_reply":"2022-08-01T03:40:57.468834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# objectからboolに変換\nall_df[\"CryoSleep\"]=all_df[\"CryoSleep\"].astype(bool)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:57.470715Z","iopub.execute_input":"2022-08-01T03:40:57.471509Z","iopub.status.idle":"2022-08-01T03:40:57.484405Z","shell.execute_reply.started":"2022-08-01T03:40:57.471478Z","shell.execute_reply":"2022-08-01T03:40:57.483081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cabin","metadata":{}},{"cell_type":"code","source":"# ○/○/○と3要素からなっているのでスラッシュで区切ってCabin1, 2, 3とする\nall_df[\"Cabin1\"] = all_df[\"Cabin\"].str.split(\"/\",expand=True)[0]\nall_df[\"Cabin2\"] = all_df[\"Cabin\"].str.split(\"/\",expand=True)[1]\nall_df[\"Cabin3\"] = all_df[\"Cabin\"].str.split(\"/\",expand=True)[2]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:57.485667Z","iopub.execute_input":"2022-08-01T03:40:57.486035Z","iopub.status.idle":"2022-08-01T03:40:57.586563Z","shell.execute_reply.started":"2022-08-01T03:40:57.486003Z","shell.execute_reply":"2022-08-01T03:40:57.585583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Cabin1","metadata":{}},{"cell_type":"code","source":"sns.countplot(all_df[\"Cabin1\"],hue = all_df[\"Transported\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:57.588592Z","iopub.execute_input":"2022-08-01T03:40:57.589055Z","iopub.status.idle":"2022-08-01T03:40:57.816426Z","shell.execute_reply.started":"2022-08-01T03:40:57.589023Z","shell.execute_reply":"2022-08-01T03:40:57.815132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 欠損値を最頻値の\"F\"で埋めておく\nall_df[\"Cabin1\"].fillna(\"F\",inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:57.818645Z","iopub.execute_input":"2022-08-01T03:40:57.819005Z","iopub.status.idle":"2022-08-01T03:40:57.826076Z","shell.execute_reply.started":"2022-08-01T03:40:57.818975Z","shell.execute_reply":"2022-08-01T03:40:57.825068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Cabin2","metadata":{}},{"cell_type":"code","source":"all_df[\"Cabin2\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:57.827316Z","iopub.execute_input":"2022-08-01T03:40:57.827585Z","iopub.status.idle":"2022-08-01T03:40:57.853721Z","shell.execute_reply.started":"2022-08-01T03:40:57.827557Z","shell.execute_reply":"2022-08-01T03:40:57.852732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 欠損をCabin2に存在しない値（2000としておく）で埋める\nall_df[\"Cabin2\"].fillna(2000,inplace=True)\n# intに変換\nall_df[\"Cabin2\"]=all_df[\"Cabin2\"].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:57.855023Z","iopub.execute_input":"2022-08-01T03:40:57.855716Z","iopub.status.idle":"2022-08-01T03:40:57.867547Z","shell.execute_reply.started":"2022-08-01T03:40:57.855690Z","shell.execute_reply":"2022-08-01T03:40:57.866474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(all_df[\"Cabin2\"],bins=350,color=\"blue\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:57.878027Z","iopub.execute_input":"2022-08-01T03:40:57.878949Z","iopub.status.idle":"2022-08-01T03:40:58.631846Z","shell.execute_reply.started":"2022-08-01T03:40:57.878918Z","shell.execute_reply":"2022-08-01T03:40:58.631106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 4グループに分割\nall_df.loc[(0<=all_df[\"Cabin2\"])&(all_df[\"Cabin2\"]<350),\"Cabin2\"] = 1\nall_df.loc[(350<=all_df[\"Cabin2\"])&(all_df[\"Cabin2\"]<600),\"Cabin2\"] = 2\nall_df.loc[(600<=all_df[\"Cabin2\"])&(all_df[\"Cabin2\"]<1500),\"Cabin2\"] = 3\nall_df.loc[(1500<=all_df[\"Cabin2\"])&(all_df[\"Cabin2\"]<2000),\"Cabin2\"] = 4","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:58.633109Z","iopub.execute_input":"2022-08-01T03:40:58.633377Z","iopub.status.idle":"2022-08-01T03:40:58.645169Z","shell.execute_reply.started":"2022-08-01T03:40:58.633349Z","shell.execute_reply":"2022-08-01T03:40:58.643768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_df[\"Cabin2\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:58.647046Z","iopub.execute_input":"2022-08-01T03:40:58.647374Z","iopub.status.idle":"2022-08-01T03:40:58.658669Z","shell.execute_reply.started":"2022-08-01T03:40:58.647344Z","shell.execute_reply":"2022-08-01T03:40:58.657866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1グループが1番多いので先ほど2000で補完した欠損部分を1グループに変更\nall_df.loc[all_df[\"Cabin2\"]==2000,\"Cabin2\"]=1\nsns.countplot(all_df[\"Cabin2\"],hue=all_df[\"Transported\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:58.660325Z","iopub.execute_input":"2022-08-01T03:40:58.660860Z","iopub.status.idle":"2022-08-01T03:40:58.835143Z","shell.execute_reply.started":"2022-08-01T03:40:58.660815Z","shell.execute_reply":"2022-08-01T03:40:58.833353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Cabin3","metadata":{}},{"cell_type":"code","source":"sns.countplot(all_df[\"Cabin3\"],hue=all_df[\"Transported\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:58.836768Z","iopub.execute_input":"2022-08-01T03:40:58.837254Z","iopub.status.idle":"2022-08-01T03:40:58.987534Z","shell.execute_reply.started":"2022-08-01T03:40:58.837224Z","shell.execute_reply":"2022-08-01T03:40:58.986702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_df[\"Cabin3\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:58.988657Z","iopub.execute_input":"2022-08-01T03:40:58.989371Z","iopub.status.idle":"2022-08-01T03:40:59.000214Z","shell.execute_reply.started":"2022-08-01T03:40:58.989336Z","shell.execute_reply":"2022-08-01T03:40:58.999433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 欠損を最頻値の\"S\"で埋める\nall_df[\"Cabin3\"].fillna(\"S\",inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:59.001486Z","iopub.execute_input":"2022-08-01T03:40:59.002287Z","iopub.status.idle":"2022-08-01T03:40:59.009496Z","shell.execute_reply.started":"2022-08-01T03:40:59.002252Z","shell.execute_reply":"2022-08-01T03:40:59.008701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Destination","metadata":{}},{"cell_type":"code","source":"sns.countplot(all_df[\"Destination\"],hue=all_df[\"Transported\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:59.011409Z","iopub.execute_input":"2022-08-01T03:40:59.011688Z","iopub.status.idle":"2022-08-01T03:40:59.160420Z","shell.execute_reply.started":"2022-08-01T03:40:59.011661Z","shell.execute_reply":"2022-08-01T03:40:59.159680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 欠損を最頻値\"TRAPPIST-1e\"で埋める\nall_df[\"Destination\"].fillna(\"TRAPPIST-1e\",inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:59.161444Z","iopub.execute_input":"2022-08-01T03:40:59.162010Z","iopub.status.idle":"2022-08-01T03:40:59.169050Z","shell.execute_reply.started":"2022-08-01T03:40:59.161978Z","shell.execute_reply":"2022-08-01T03:40:59.167490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Age","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (23,5))\nsns.countplot(all_df[\"Age\"],hue=all_df[\"Transported\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:40:59.170694Z","iopub.execute_input":"2022-08-01T03:40:59.171482Z","iopub.status.idle":"2022-08-01T03:41:00.252925Z","shell.execute_reply.started":"2022-08-01T03:40:59.171415Z","shell.execute_reply":"2022-08-01T03:41:00.251375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.FacetGrid(all_df, col='Transported').map(plt.hist, 'Age', bins=20)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:00.254507Z","iopub.execute_input":"2022-08-01T03:41:00.255578Z","iopub.status.idle":"2022-08-01T03:41:00.661835Z","shell.execute_reply.started":"2022-08-01T03:41:00.255535Z","shell.execute_reply":"2022-08-01T03:41:00.660833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#欠損値を中央値で埋める\nall_df['Age'].fillna(all_df['Age'].median(), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:00.663262Z","iopub.execute_input":"2022-08-01T03:41:00.663552Z","iopub.status.idle":"2022-08-01T03:41:00.672670Z","shell.execute_reply.started":"2022-08-01T03:41:00.663522Z","shell.execute_reply":"2022-08-01T03:41:00.671346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## VIP","metadata":{}},{"cell_type":"code","source":"sns.countplot(all_df[\"VIP\"],hue=all_df[\"Transported\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:00.674495Z","iopub.execute_input":"2022-08-01T03:41:00.675515Z","iopub.status.idle":"2022-08-01T03:41:00.868159Z","shell.execute_reply.started":"2022-08-01T03:41:00.675474Z","shell.execute_reply":"2022-08-01T03:41:00.867146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 欠損を最頻値\"False\"で埋める\nall_df[\"VIP\"].fillna(\"False\",inplace=True)\n# objectからboolに変換\nall_df[\"VIP\"]=all_df[\"VIP\"].astype(bool)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:00.869296Z","iopub.execute_input":"2022-08-01T03:41:00.869623Z","iopub.status.idle":"2022-08-01T03:41:00.882534Z","shell.execute_reply.started":"2022-08-01T03:41:00.869587Z","shell.execute_reply":"2022-08-01T03:41:00.881291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## RoomService～VRDeck","metadata":{}},{"cell_type":"code","source":"# すべて中央値で埋める\nall_df[\"RoomService\"].fillna(all_df[\"RoomService\"].median(),inplace=True)\nall_df[\"FoodCourt\"].fillna(all_df[\"FoodCourt\"].median(),inplace=True)\nall_df[\"ShoppingMall\"].fillna(all_df[\"ShoppingMall\"].median(),inplace=True)\nall_df[\"Spa\"].fillna(all_df[\"Spa\"].median(),inplace=True)\nall_df[\"VRDeck\"].fillna(all_df[\"VRDeck\"].median(),inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:00.884308Z","iopub.execute_input":"2022-08-01T03:41:00.884752Z","iopub.status.idle":"2022-08-01T03:41:00.901299Z","shell.execute_reply.started":"2022-08-01T03:41:00.884712Z","shell.execute_reply":"2022-08-01T03:41:00.900325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 合計の特徴量を作っておく\nall_df[\"pay\"]=all_df[\"RoomService\"]+all_df[\"FoodCourt\"]+all_df[\"ShoppingMall\"]+all_df[\"Spa\"]+all_df[\"VRDeck\"]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:00.902769Z","iopub.execute_input":"2022-08-01T03:41:00.904414Z","iopub.status.idle":"2022-08-01T03:41:00.916115Z","shell.execute_reply.started":"2022-08-01T03:41:00.904378Z","shell.execute_reply":"2022-08-01T03:41:00.914989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:00.917572Z","iopub.execute_input":"2022-08-01T03:41:00.917976Z","iopub.status.idle":"2022-08-01T03:41:00.939637Z","shell.execute_reply.started":"2022-08-01T03:41:00.917934Z","shell.execute_reply":"2022-08-01T03:41:00.938236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:00.941169Z","iopub.execute_input":"2022-08-01T03:41:00.941733Z","iopub.status.idle":"2022-08-01T03:41:00.954761Z","shell.execute_reply.started":"2022-08-01T03:41:00.941707Z","shell.execute_reply":"2022-08-01T03:41:00.953988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 学習データを形成","metadata":{}},{"cell_type":"code","source":"# nameは使わないので削除\nall_df=all_df.drop([\"Name\"],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:00.956077Z","iopub.execute_input":"2022-08-01T03:41:00.956544Z","iopub.status.idle":"2022-08-01T03:41:00.966088Z","shell.execute_reply.started":"2022-08-01T03:41:00.956504Z","shell.execute_reply":"2022-08-01T03:41:00.965143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ラベルエンコーディング\nfrom sklearn.preprocessing import LabelEncoder\nlbl_enc = LabelEncoder()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:00.967455Z","iopub.execute_input":"2022-08-01T03:41:00.968091Z","iopub.status.idle":"2022-08-01T03:41:00.976085Z","shell.execute_reply.started":"2022-08-01T03:41:00.968053Z","shell.execute_reply":"2022-08-01T03:41:00.974865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_df[\"HomePlanet\"]=lbl_enc.fit_transform(all_df[\"HomePlanet\"])\nall_df[\"Destination\"]=lbl_enc.fit_transform(all_df[\"Destination\"])\nall_df[\"Cabin1\"]=lbl_enc.fit_transform(all_df[\"Cabin1\"])\nall_df[\"Cabin3\"]=lbl_enc.fit_transform(all_df[\"Cabin3\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:00.977814Z","iopub.execute_input":"2022-08-01T03:41:00.978118Z","iopub.status.idle":"2022-08-01T03:41:01.017209Z","shell.execute_reply.started":"2022-08-01T03:41:00.978088Z","shell.execute_reply":"2022-08-01T03:41:01.016111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:01.019925Z","iopub.execute_input":"2022-08-01T03:41:01.020292Z","iopub.status.idle":"2022-08-01T03:41:01.049313Z","shell.execute_reply.started":"2022-08-01T03:41:01.020254Z","shell.execute_reply":"2022-08-01T03:41:01.048040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 相関関係を可視化","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(14,15))\nsns.heatmap(all_df.corr(), annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:01.053071Z","iopub.execute_input":"2022-08-01T03:41:01.054091Z","iopub.status.idle":"2022-08-01T03:41:02.358671Z","shell.execute_reply.started":"2022-08-01T03:41:01.054048Z","shell.execute_reply":"2022-08-01T03:41:02.357607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cabin, PassengerId, group_id, id_numも使わないので削除\nall_df = all_df.drop([\"PassengerId\",\"Cabin\",\"group_id\"],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:02.360163Z","iopub.execute_input":"2022-08-01T03:41:02.360524Z","iopub.status.idle":"2022-08-01T03:41:02.369449Z","shell.execute_reply.started":"2022-08-01T03:41:02.360485Z","shell.execute_reply":"2022-08-01T03:41:02.368227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 学習用データ\nx = all_df.loc[(all_df[\"train/test\"]==\"train\")].drop([\"Transported\",\"train/test\"],axis=1)\nt = all_df.loc[(all_df[\"train/test\"]==\"train\")].iloc[:,10:11]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:02.371057Z","iopub.execute_input":"2022-08-01T03:41:02.371339Z","iopub.status.idle":"2022-08-01T03:41:02.388665Z","shell.execute_reply.started":"2022-08-01T03:41:02.371310Z","shell.execute_reply":"2022-08-01T03:41:02.387854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:02.389860Z","iopub.execute_input":"2022-08-01T03:41:02.390154Z","iopub.status.idle":"2022-08-01T03:41:02.410539Z","shell.execute_reply.started":"2022-08-01T03:41:02.390119Z","shell.execute_reply":"2022-08-01T03:41:02.409843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:02.411567Z","iopub.execute_input":"2022-08-01T03:41:02.411839Z","iopub.status.idle":"2022-08-01T03:41:02.420642Z","shell.execute_reply.started":"2022-08-01T03:41:02.411803Z","shell.execute_reply":"2022-08-01T03:41:02.419336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = all_df.loc[(all_df[\"train/test\"]==\"test\")].drop([\"Transported\",\"train/test\"],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:02.421932Z","iopub.execute_input":"2022-08-01T03:41:02.422624Z","iopub.status.idle":"2022-08-01T03:41:02.433570Z","shell.execute_reply.started":"2022-08-01T03:41:02.422598Z","shell.execute_reply":"2022-08-01T03:41:02.432823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### ハイパーパラメータの決定","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nimport lightgbm as lgb\nfrom sklearn.model_selection import StratifiedKFold","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:02.434602Z","iopub.execute_input":"2022-08-01T03:41:02.435776Z","iopub.status.idle":"2022-08-01T03:41:03.503946Z","shell.execute_reply.started":"2022-08-01T03:41:02.435728Z","shell.execute_reply":"2022-08-01T03:41:03.502595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# k分割交差検証\nkf = StratifiedKFold(n_splits=10, shuffle=True, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:03.505461Z","iopub.execute_input":"2022-08-01T03:41:03.505801Z","iopub.status.idle":"2022-08-01T03:41:03.510640Z","shell.execute_reply.started":"2022-08-01T03:41:03.505751Z","shell.execute_reply":"2022-08-01T03:41:03.509721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# optunaが選んだ1組のハイパーパラメータで10回交差検証を行い，検証データの正解率の平均を出した後, 最大化するハイパーパラメータを見つける\ndef objective(trial):\n    params={\"metric\":\"auc\",\n              \"objective\":\"binary\", \n              \"max_depth\":trial.suggest_int(\"max_depth\",5,1000),\n              \"num_leaves\":trial.suggest_int('num_leaves',10,3000),\n              \"min_child_samples\":trial.suggest_int(\"min_child_samples\",50,500),\n              \"learning_rate\":trial.suggest_uniform(\"learning_rate\",0.01,1.5),\n              \"feature_fraction\":trial.suggest_uniform(\"feature_fraction\",0,1),\n              \"bagging_fraction\":trial.suggest_uniform(\"bagging_fraction\",0,1),\n              \"verbose\": -1}\n    \n\n    val_scores=[]\n\n    for i, (train__, val__) in enumerate(kf.split(x,t)):\n        x_train, x_val=x.iloc[train__], x.iloc[val__]\n        t_train, t_val=t.iloc[train__], t.iloc[val__]\n        lgb_train=lgb.Dataset(x_train, t_train)\n        lgb_eval=lgb.Dataset(x_val, t_val)\n\n        lgbm=lgb.train(params,\n                        lgb_train,\n                        valid_sets=lgb_eval,\n                        num_boost_round=1000,\n                        early_stopping_rounds=10,\n                        verbose_eval=False)\n          \n        t_train_pred=np.round(lgbm.predict(x_train))\n        t_val_pred=np.round(lgbm.predict(x_val))\n        scoretrain=metrics.accuracy_score(t_train[\"Transported\"],t_train_pred)\n        scoreval=metrics.accuracy_score(t_val[\"Transported\"],t_val_pred)\n        val_scores.append(scoreval)\n\n    cv_score=np.mean(val_scores)\n\n    return cv_score","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:03.511600Z","iopub.execute_input":"2022-08-01T03:41:03.511905Z","iopub.status.idle":"2022-08-01T03:41:03.526029Z","shell.execute_reply.started":"2022-08-01T03:41:03.511873Z","shell.execute_reply":"2022-08-01T03:41:03.524804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study = optuna.create_study(direction=\"maximize\")\nstudy.optimize(objective, n_trials=70)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:41:03.527486Z","iopub.execute_input":"2022-08-01T03:41:03.527827Z","iopub.status.idle":"2022-08-01T03:43:19.668345Z","shell.execute_reply.started":"2022-08-01T03:41:03.527766Z","shell.execute_reply":"2022-08-01T03:43:19.667527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study.best_params   # ベストなパラメータ","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:43:19.671839Z","iopub.execute_input":"2022-08-01T03:43:19.673671Z","iopub.status.idle":"2022-08-01T03:43:19.681343Z","shell.execute_reply.started":"2022-08-01T03:43:19.673636Z","shell.execute_reply":"2022-08-01T03:43:19.680416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# モデル構築","metadata":{}},{"cell_type":"code","source":"x_train,x_val,t_train,t_val = train_test_split(x,t,test_size=0.25,stratify=t)\nlgb_train = lgb.Dataset(x_train, t_train)\nlgb_eval = lgb.Dataset(x_val, t_val)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:43:19.682300Z","iopub.execute_input":"2022-08-01T03:43:19.682536Z","iopub.status.idle":"2022-08-01T03:43:19.730507Z","shell.execute_reply.started":"2022-08-01T03:43:19.682512Z","shell.execute_reply":"2022-08-01T03:43:19.729285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bestparams={\"metric\":\"auc\",\n            \"objective\":\"binary\", \n            \"max_depth\":study.best_params[\"max_depth\"],\n            \"num_leaves\":study.best_params[\"num_leaves\"],\n            \"min_child_samples\":study.best_params[\"min_child_samples\"],\n            \"learning_rate\":study.best_params[\"learning_rate\"],\n            \"feature_fraction\":study.best_params[\"feature_fraction\"],\n            \"bagging_fraction\":study.best_params[\"bagging_fraction\"]}\n\n\nbest_lgbm=lgb.train(bestparams,\n                    lgb_train,\n                    valid_sets=lgb_eval,\n                    num_boost_round=1000,\n                    early_stopping_rounds=50,\n                    verbose_eval=50)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:43:19.731852Z","iopub.execute_input":"2022-08-01T03:43:19.732121Z","iopub.status.idle":"2022-08-01T03:43:20.002753Z","shell.execute_reply.started":"2022-08-01T03:43:19.732094Z","shell.execute_reply":"2022-08-01T03:43:20.002007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# テストデータに対して予測\npred = np.round(best_lgbm.predict(x_test))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:43:20.006237Z","iopub.execute_input":"2022-08-01T03:43:20.008107Z","iopub.status.idle":"2022-08-01T03:43:20.034477Z","shell.execute_reply.started":"2022-08-01T03:43:20.008072Z","shell.execute_reply":"2022-08-01T03:43:20.033642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Accuracy","metadata":{}},{"cell_type":"code","source":"print(metrics.accuracy_score(t_train[\"Transported\"],np.round(best_lgbm.predict(x_train))))\nprint(metrics.accuracy_score(t_val[\"Transported\"],np.round(best_lgbm.predict(x_val))))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:43:20.035914Z","iopub.execute_input":"2022-08-01T03:43:20.036436Z","iopub.status.idle":"2022-08-01T03:43:20.079885Z","shell.execute_reply.started":"2022-08-01T03:43:20.036402Z","shell.execute_reply":"2022-08-01T03:43:20.078465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Precision","metadata":{}},{"cell_type":"code","source":"print(metrics.precision_score(t_train[\"Transported\"],np.round(best_lgbm.predict(x_train))))\nprint(metrics.precision_score(t_val[\"Transported\"],np.round(best_lgbm.predict(x_val))))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:51:21.798132Z","iopub.execute_input":"2022-08-01T03:51:21.798497Z","iopub.status.idle":"2022-08-01T03:51:21.870419Z","shell.execute_reply.started":"2022-08-01T03:51:21.798467Z","shell.execute_reply":"2022-08-01T03:51:21.869658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Recall","metadata":{}},{"cell_type":"code","source":"print(metrics.recall_score(t_train[\"Transported\"],np.round(best_lgbm.predict(x_train))))\nprint(metrics.recall_score(t_val[\"Transported\"],np.round(best_lgbm.predict(x_val))))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:52:05.679328Z","iopub.execute_input":"2022-08-01T03:52:05.679735Z","iopub.status.idle":"2022-08-01T03:52:05.807588Z","shell.execute_reply.started":"2022-08-01T03:52:05.679704Z","shell.execute_reply":"2022-08-01T03:52:05.806573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 提出用データの作成\npredict_df = pd.DataFrame(pred.astype(int),columns=[\"Transported\"])\npredict_df[\"Transported\"] = predict_df[\"Transported\"].astype(bool)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:43:20.083142Z","iopub.execute_input":"2022-08-01T03:43:20.083682Z","iopub.status.idle":"2022-08-01T03:43:20.092679Z","shell.execute_reply.started":"2022-08-01T03:43:20.083649Z","shell.execute_reply":"2022-08-01T03:43:20.091610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.concat([test_df[\"PassengerId\"], predict_df],axis=1)\nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:43:20.094084Z","iopub.execute_input":"2022-08-01T03:43:20.094438Z","iopub.status.idle":"2022-08-01T03:43:20.112263Z","shell.execute_reply.started":"2022-08-01T03:43:20.094400Z","shell.execute_reply":"2022-08-01T03:43:20.111029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# csv出力\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:43:20.115323Z","iopub.execute_input":"2022-08-01T03:43:20.115572Z","iopub.status.idle":"2022-08-01T03:43:20.130076Z","shell.execute_reply.started":"2022-08-01T03:43:20.115547Z","shell.execute_reply":"2022-08-01T03:43:20.129123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 特徴量の重要度を可視化","metadata":{}},{"cell_type":"code","source":"# 特徴量重要度を棒グラフでプロットする関数 \ndef plot_feature_importance(df):\n    n_features = len(df)                              # 特徴量数(説明変数の個数) \n    df_plot = df.sort_values('importance')            # df_importanceをプロット用に特徴量重要度を昇順ソート \n    f_importance_plot = df_plot['importance'].values  # 特徴量重要度の取得 \n    plt.barh(range(n_features), f_importance_plot, align='center') \n    cols_plot = df_plot.index          # 特徴量の取得 \n    plt.yticks(np.arange(n_features), cols_plot)      # x軸,y軸の値の設定\n    plt.xlabel('Feature importance')                  # x軸のタイトル\n    plt.ylabel('Feature')     ","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:43:20.131162Z","iopub.execute_input":"2022-08-01T03:43:20.131418Z","iopub.status.idle":"2022-08-01T03:43:20.138766Z","shell.execute_reply.started":"2022-08-01T03:43:20.131391Z","shell.execute_reply":"2022-08-01T03:43:20.138068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# importanceを表示する\nimportance = pd.DataFrame(best_lgbm.feature_importance(importance_type=\"gain\"), index=x.columns, columns=['importance'])\nimportance = importance.sort_values('importance', ascending=False)\ndisplay(importance)\n# 特徴量重要度の可視化\nplot_feature_importance(importance)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T03:43:20.139838Z","iopub.execute_input":"2022-08-01T03:43:20.140066Z","iopub.status.idle":"2022-08-01T03:43:20.370444Z","shell.execute_reply.started":"2022-08-01T03:43:20.140044Z","shell.execute_reply":"2022-08-01T03:43:20.369161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}