{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-18T04:16:50.812008Z","iopub.execute_input":"2022-07-18T04:16:50.812523Z","iopub.status.idle":"2022-07-18T04:16:50.819381Z","shell.execute_reply.started":"2022-07-18T04:16:50.812481Z","shell.execute_reply":"2022-07-18T04:16:50.817793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 概要\n\n1. Data Exploration\n2. Data Visualization\n4. Feature Engineering\n5. Data Preprocessing\n6. Model Building\n7. Submission","metadata":{}},{"cell_type":"code","source":"# データセット読み込み\n\ntrain_data = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntest_data = pd.read_csv(\"/kaggle/input/titanic/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:16:50.825434Z","iopub.execute_input":"2022-07-18T04:16:50.825938Z","iopub.status.idle":"2022-07-18T04:16:50.844161Z","shell.execute_reply.started":"2022-07-18T04:16:50.825893Z","shell.execute_reply":"2022-07-18T04:16:50.842438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Exploration","metadata":{}},{"cell_type":"code","source":"# 列名を表示させる\ntrain_data.columns.values","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:16:50.847316Z","iopub.execute_input":"2022-07-18T04:16:50.848026Z","iopub.status.idle":"2022-07-18T04:16:50.855890Z","shell.execute_reply.started":"2022-07-18T04:16:50.847980Z","shell.execute_reply":"2022-07-18T04:16:50.854725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 要約情報を出力する\ntrain_data.info()\nprint('_'*40)\ntest_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:16:50.858915Z","iopub.execute_input":"2022-07-18T04:16:50.859518Z","iopub.status.idle":"2022-07-18T04:16:50.889557Z","shell.execute_reply.started":"2022-07-18T04:16:50.859480Z","shell.execute_reply":"2022-07-18T04:16:50.887942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#　要約統計量の出力\ntrain_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:16:50.892397Z","iopub.execute_input":"2022-07-18T04:16:50.893644Z","iopub.status.idle":"2022-07-18T04:16:50.926959Z","shell.execute_reply.started":"2022-07-18T04:16:50.893594Z","shell.execute_reply":"2022-07-18T04:16:50.925601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# カテゴリカルデータの要約情報の出力\ntrain_data.describe(include=['O'])","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:16:50.928272Z","iopub.execute_input":"2022-07-18T04:16:50.928765Z","iopub.status.idle":"2022-07-18T04:16:50.950669Z","shell.execute_reply.started":"2022-07-18T04:16:50.928735Z","shell.execute_reply":"2022-07-18T04:16:50.949590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ピボットテーブルで平均値を出力\npd.pivot_table(train_data, index = 'Survived', values = ['Age','SibSp','Parch','Fare'])","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:16:50.953852Z","iopub.execute_input":"2022-07-18T04:16:50.955035Z","iopub.status.idle":"2022-07-18T04:16:50.978505Z","shell.execute_reply.started":"2022-07-18T04:16:50.954991Z","shell.execute_reply":"2022-07-18T04:16:50.977189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" # Data Visualization","metadata":{}},{"cell_type":"code","source":"# データを定量、定性データに分ける\n\ndf_num = train_data[['Age','SibSp','Parch','Fare']]\ndf_cat = train_data[['Survived','Pclass','Sex','Ticket','Cabin','Embarked']]","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:16:50.980233Z","iopub.execute_input":"2022-07-18T04:16:50.980709Z","iopub.status.idle":"2022-07-18T04:16:50.993862Z","shell.execute_reply.started":"2022-07-18T04:16:50.980673Z","shell.execute_reply":"2022-07-18T04:16:50.992414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 定量データ（'Age','SibSp','Parch','Fare'）のヒストグラム\n\nfor i in df_num.columns:\n    plt.hist(df_num[i])\n    plt.title(i)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:16:50.996838Z","iopub.execute_input":"2022-07-18T04:16:50.997608Z","iopub.status.idle":"2022-07-18T04:16:51.762240Z","shell.execute_reply.started":"2022-07-18T04:16:50.997570Z","shell.execute_reply":"2022-07-18T04:16:51.760707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# カテゴリカルデータの内訳の棒グラフ\n\nfor i in df_cat.columns:\n    sns.barplot(df_cat[i].value_counts().index,df_cat[i].value_counts()).set_title(i)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:16:51.765075Z","iopub.execute_input":"2022-07-18T04:16:51.765764Z","iopub.status.idle":"2022-07-18T04:17:02.531644Z","shell.execute_reply.started":"2022-07-18T04:16:51.765711Z","shell.execute_reply":"2022-07-18T04:17:02.530271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Survived列に応じたAge列のデータのヒストグラム\n\ng = sns.FacetGrid(train_data, col='Survived')\ng.map(plt.hist, 'Age', bins=20)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:02.533349Z","iopub.execute_input":"2022-07-18T04:17:02.533739Z","iopub.status.idle":"2022-07-18T04:17:02.968023Z","shell.execute_reply.started":"2022-07-18T04:17:02.533705Z","shell.execute_reply":"2022-07-18T04:17:02.966785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#PclassとSurvivedの数値に応じたAge列のヒストグラム\n\ngrid = sns.FacetGrid(train_data, col='Survived', row='Pclass', size=2.2, aspect=1.6)\ngrid.map(plt.hist, 'Age', alpha=.5, bins=20)\ngrid.add_legend();","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:02.971698Z","iopub.execute_input":"2022-07-18T04:17:02.972737Z","iopub.status.idle":"2022-07-18T04:17:04.275839Z","shell.execute_reply.started":"2022-07-18T04:17:02.972693Z","shell.execute_reply":"2022-07-18T04:17:04.274396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Embarked&SurvivedとSex&Fareを比較した棒グラフ\n\ngrid = sns.FacetGrid(train_data, row='Embarked', col='Survived', size=2.2, aspect=1.6)\ngrid.map(sns.barplot, 'Sex', 'Fare', alpha=.5, ci=None)\ngrid.add_legend()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:04.277882Z","iopub.execute_input":"2022-07-18T04:17:04.278431Z","iopub.status.idle":"2022-07-18T04:17:05.199917Z","shell.execute_reply.started":"2022-07-18T04:17:04.278380Z","shell.execute_reply":"2022-07-18T04:17:05.198761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Pclass&SexによるAge列のヒストグラム\n\ngrid = sns.FacetGrid(train_data, row='Pclass', col='Sex', size=2.2, aspect=1.6)\ngrid.map(plt.hist, 'Age', alpha=.5, bins=20)\ngrid.add_legend()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:05.201413Z","iopub.execute_input":"2022-07-18T04:17:05.201873Z","iopub.status.idle":"2022-07-18T04:17:06.541954Z","shell.execute_reply.started":"2022-07-18T04:17:05.201837Z","shell.execute_reply":"2022-07-18T04:17:06.540444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ヒートマップ\n\nprint(df_num.corr())\nsns.heatmap(df_num.corr())","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:06.543732Z","iopub.execute_input":"2022-07-18T04:17:06.544144Z","iopub.status.idle":"2022-07-18T04:17:06.808765Z","shell.execute_reply.started":"2022-07-18T04:17:06.544107Z","shell.execute_reply":"2022-07-18T04:17:06.807499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#　訓練データとテストデータのヒートマップ\nfig, axs = plt.subplots(nrows=2, figsize=(20, 20))\n\nsns.heatmap(train_data.drop(['PassengerId'], axis=1).corr(), ax=axs[0], annot=True, square=True, cmap='coolwarm', annot_kws={'size': 14})\nsns.heatmap(test_data.drop(['PassengerId'], axis=1).corr(), ax=axs[1], annot=True, square=True, cmap='coolwarm', annot_kws={'size': 14})\n\nfor i in range(2):    \n    axs[i].tick_params(axis='x', labelsize=14)\n    axs[i].tick_params(axis='y', labelsize=14)\n    \naxs[0].set_title('Training Set Correlations', size=15)\naxs[1].set_title('Test Set Correlations', size=15)\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:06.810411Z","iopub.execute_input":"2022-07-18T04:17:06.811355Z","iopub.status.idle":"2022-07-18T04:17:08.249585Z","shell.execute_reply.started":"2022-07-18T04:17:06.811317Z","shell.execute_reply":"2022-07-18T04:17:08.248350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"# 客室数の新規カテゴリカルデータ列を作る\n\ntrain_data['cabin_multiple'] = train_data.Cabin.apply(lambda x: 0 if pd.isna(x) else len(x.split(' ')))\ntrain_data['cabin_multiple'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:08.251231Z","iopub.execute_input":"2022-07-18T04:17:08.251865Z","iopub.status.idle":"2022-07-18T04:17:08.264383Z","shell.execute_reply.started":"2022-07-18T04:17:08.251828Z","shell.execute_reply":"2022-07-18T04:17:08.263044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 訓練データとテストデータの両方同時に処理できるようにする\n\ntrain_data['train_test'] = 1\ntest_data['train_test'] = 0\ntest_data['Survived'] = np.NaN\nall_data = pd.concat([train_data,test_data])# 訓練データとテストデータの両方同時に処理できるようにする\n\nall_data.columns\nall_data.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:08.268946Z","iopub.execute_input":"2022-07-18T04:17:08.269378Z","iopub.status.idle":"2022-07-18T04:17:08.298756Z","shell.execute_reply.started":"2022-07-18T04:17:08.269342Z","shell.execute_reply":"2022-07-18T04:17:08.297550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 新規客室数のカテゴリカルデータ\nall_data['cabin_multiple'] = all_data.Cabin.apply(lambda x: 0 if pd.isna(x) else len(x.split(' '))) \n\n#客室データの先頭文字が入っている新規カテゴリカルデータ\nall_data['cabin_adv'] = all_data.Cabin.apply(lambda x: str(x)[0]) \n\n#Ticket列が数字かどうか判断した列（0：数字でない、1：数字）\nall_data['numeric_ticket'] = all_data.Ticket.apply(lambda x: 1 if x.isnumeric() else 0) \n\n#Ticket列のデータの先頭の文字を抜き出した列（抜き出せない場合、0）\nall_data['ticket_letters'] = all_data.Ticket.apply(lambda x: ''.join(x.split(' ')[:-1]).replace('.','').replace('/','').lower() if len(x.split(' ')[:-1]) >0 else 0) \n\n#Name列から敬称を抜き出した列\nall_data['name_title'] = all_data.Name.apply(lambda x: x.split(',')[1].split('.')[0].strip()) ","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:08.300625Z","iopub.execute_input":"2022-07-18T04:17:08.300982Z","iopub.status.idle":"2022-07-18T04:17:08.320452Z","shell.execute_reply.started":"2022-07-18T04:17:08.300950Z","shell.execute_reply":"2022-07-18T04:17:08.319030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{}},{"cell_type":"code","source":"#欠損値を平均値で穴埋めする\nall_data.Age = all_data.Age.fillna(train_data.Age.median())\nall_data.Fare = all_data.Fare.fillna(train_data.Fare.median())","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:08.322023Z","iopub.execute_input":"2022-07-18T04:17:08.322514Z","iopub.status.idle":"2022-07-18T04:17:08.341269Z","shell.execute_reply.started":"2022-07-18T04:17:08.322476Z","shell.execute_reply":"2022-07-18T04:17:08.339959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Embarked列に欠損値のある行を削除する\nall_data.dropna(subset=['Embarked'],inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:08.343000Z","iopub.execute_input":"2022-07-18T04:17:08.343592Z","iopub.status.idle":"2022-07-18T04:17:08.355865Z","shell.execute_reply.started":"2022-07-18T04:17:08.343541Z","shell.execute_reply":"2022-07-18T04:17:08.354768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#SibSp、Fare列の対数を取得する\n\nall_data['norm_sibsp'] = np.log(all_data.SibSp+1)\nall_data['norm_fare'] = np.log(all_data.Fare+1)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:08.357464Z","iopub.execute_input":"2022-07-18T04:17:08.357940Z","iopub.status.idle":"2022-07-18T04:17:08.370175Z","shell.execute_reply.started":"2022-07-18T04:17:08.357893Z","shell.execute_reply":"2022-07-18T04:17:08.368660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Pclassの型をstr型にする\n\nall_data.Pclass = all_data.Pclass.astype(str)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:08.372261Z","iopub.execute_input":"2022-07-18T04:17:08.372705Z","iopub.status.idle":"2022-07-18T04:17:08.387914Z","shell.execute_reply.started":"2022-07-18T04:17:08.372666Z","shell.execute_reply":"2022-07-18T04:17:08.386827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#カテゴリカル変数ををダミー変数に変換\n\nall_dummies = pd.get_dummies(all_data[['Pclass','Sex','Age','SibSp','Parch','norm_fare','Embarked','cabin_adv','cabin_multiple','numeric_ticket','name_title','train_test']])\nX_train = all_dummies[all_dummies.train_test == 1].drop(['train_test'], axis =1)\nX_test = all_dummies[all_dummies.train_test == 0].drop(['train_test'], axis =1)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:08.389505Z","iopub.execute_input":"2022-07-18T04:17:08.390614Z","iopub.status.idle":"2022-07-18T04:17:08.421884Z","shell.execute_reply.started":"2022-07-18T04:17:08.390562Z","shell.execute_reply":"2022-07-18T04:17:08.420539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_trainデータを生成\ny_train = all_data[all_data.train_test==1].Survived","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:08.423667Z","iopub.execute_input":"2022-07-18T04:17:08.424018Z","iopub.status.idle":"2022-07-18T04:17:08.430715Z","shell.execute_reply.started":"2022-07-18T04:17:08.423986Z","shell.execute_reply":"2022-07-18T04:17:08.429417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#データの正規化\n\nfrom sklearn.preprocessing import StandardScaler\nscale = StandardScaler()\nall_dummies_scaled = all_dummies.copy()\nall_dummies_scaled[['Age','SibSp','Parch','norm_fare']]= scale.fit_transform(all_dummies_scaled[['Age','SibSp','Parch','norm_fare']])\nall_dummies_scaled","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:08.432347Z","iopub.execute_input":"2022-07-18T04:17:08.433363Z","iopub.status.idle":"2022-07-18T04:17:08.473354Z","shell.execute_reply.started":"2022-07-18T04:17:08.433325Z","shell.execute_reply":"2022-07-18T04:17:08.472168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 正規化された訓練データとテストデータを生成\n\nX_train_scaled = all_dummies_scaled[all_dummies_scaled.train_test == 1].drop(['train_test'], axis =1)\nX_test_scaled = all_dummies_scaled[all_dummies_scaled.train_test == 0].drop(['train_test'], axis =1)\n\ny_train = all_data[all_data.train_test==1].Survived","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:08.474837Z","iopub.execute_input":"2022-07-18T04:17:08.475441Z","iopub.status.idle":"2022-07-18T04:17:08.488285Z","shell.execute_reply.started":"2022-07-18T04:17:08.475402Z","shell.execute_reply":"2022-07-18T04:17:08.486806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Building","metadata":{}},{"cell_type":"code","source":"# scikit-learnライブラリの読み込み\n\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn import tree\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.svm import SVC\nfrom xgboost import XGBClassifier\n\ngnb = GaussianNB()\nlr = LogisticRegression(max_iter = 2000)\ndt = tree.DecisionTreeClassifier(random_state = 1)\nrf = RandomForestClassifier(random_state = 1)\nknn = KNeighborsClassifier()\nsvc = SVC(probability = True)\nxgb = XGBClassifier(random_state =1)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:08.490101Z","iopub.execute_input":"2022-07-18T04:17:08.490781Z","iopub.status.idle":"2022-07-18T04:17:08.498566Z","shell.execute_reply.started":"2022-07-18T04:17:08.490733Z","shell.execute_reply":"2022-07-18T04:17:08.497492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# VotingClassifierを使ってアンサンブル学習\n\nfrom sklearn.ensemble import VotingClassifier\nvoting_clf = VotingClassifier(estimators = [('lr',lr),('knn',knn),('rf',rf),('gnb',gnb),('svc',svc),('xgb',xgb)], voting = 'soft') \ncv = cross_val_score(voting_clf,X_train_scaled,y_train,cv=5)\nvoting_clf.fit(X_train_scaled,y_train)\ny_hat_base_vc = voting_clf.predict(X_test_scaled).astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:18:45.725666Z","iopub.execute_input":"2022-07-18T04:18:45.726220Z","iopub.status.idle":"2022-07-18T04:18:50.836145Z","shell.execute_reply.started":"2022-07-18T04:18:45.726167Z","shell.execute_reply":"2022-07-18T04:18:50.834642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"basic_submission = {'PassengerId': test_data.PassengerId, 'Survived': y_hat_base_vc}\nbase_submission = pd.DataFrame(data=basic_submission)\nbase_submission.to_csv('base_submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T04:17:13.388549Z","iopub.execute_input":"2022-07-18T04:17:13.389328Z","iopub.status.idle":"2022-07-18T04:17:14.344753Z","shell.execute_reply.started":"2022-07-18T04:17:13.389279Z","shell.execute_reply":"2022-07-18T04:17:14.343381Z"},"trusted":true},"execution_count":null,"outputs":[]}]}