{"cells":[{"metadata":{"_cell_guid":"beb0f912-cc88-4a09-89aa-4f9857a01bfa","_uuid":"2f238a336215528b25de7777e90830389a6b0687","trusted":true},"cell_type":"code","source":"# warningsを無視する\nimport warnings\nwarnings.filterwarnings('ignore')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"09cbe766-c8d2-4b7b-9268-a058fb902f75","_uuid":"a5f6c9fe9959d489dc3739cb77167792b7a247e6"},"cell_type":"markdown","source":"# 4.1 ライブラリのインポートとデータの読み込み"},{"metadata":{"_cell_guid":"f8097b97-0374-41df-a5fd-bf91bdd7c1ae","_uuid":"16b5bcbe25409a757eb55e3fe5cc895d158cfa55","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport japanize_matplotlib ","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"b3915b89-a49d-49b3-a1a4-64a31f0e6ffa","_uuid":"ee0b52b63f9e4fa8116476bcc6cf8c71c06f24b6","trusted":true},"cell_type":"code","source":"df_train = pd.read_csv(\"../input/train.csv\")\ndf_test = pd.read_csv(\"../input/test.csv\")\ndf_gender_submission = pd.read_csv(\"../input/gender_submission.csv\")","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"2c0df772-2979-4198-b896-b50e39f9e8a1","_uuid":"471e8cc348801f18d706766dd826e3ed0d06fb66","trusted":true},"cell_type":"code","source":"# 本文にはない、レイアウト設定用\n# sns.set_palette(\"Blues_r\", 3) # 青３色のスタイル\n\n\n# fontsizeの設定\nplt.rcParams[\"font.size\"] = 18\n\n# サイズの設定\nplt.rcParams['figure.figsize'] = (8.0, 6.0)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"d9590a9a-69b7-442c-b5ad-c2b0f4769863","_uuid":"9496b513be2d62f715f98b7daba4a2c33eda32b7"},"cell_type":"markdown","source":"# 4.2 データの概要を確認する\n## 4.2.1 データフレームについて"},{"metadata":{"_cell_guid":"1208e90e-502e-43d1-ba2a-0086efc08fd6","_uuid":"6c98393c5e52a400f20c50d6939527202b330163","trusted":true},"cell_type":"code","source":"df_train.head(5)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"5a0090c7-b00b-4d38-8997-9e815fa32f3e","_uuid":"74bd6429f914af75dd439807b09e4539513ba6c6"},"cell_type":"markdown","source":"## 4.2.2 データフレームの行数と列数を確認"},{"metadata":{"_cell_guid":"eec45946-a64a-4e3a-aa6b-2eb95b03e650","_uuid":"e606c19efd1cd977512748ecb67aacd8915d80d6","trusted":true},"cell_type":"code","source":"print(df_train.shape) # 学習用データ\nprint(df_test.shape) # 本番予測用データ\nprint(df_gender_submission.shape) # 提出データのサンプル","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"dae4ccc5-2c92-4b4d-a711-142dd14963ff","_uuid":"f5e4dbe157dc7ddcd26cac667ce686d86e53e624"},"cell_type":"markdown","source":"## 4.2.3 列の名前の確認"},{"metadata":{"_cell_guid":"957a96fb-cc2b-489c-8006-4c04cc5c8f7b","scrolled":true,"_uuid":"acb2a356313873001a48a19fa489bec481811f49","trusted":true},"cell_type":"code","source":"print(df_train.columns) # トレーニングデータの列名\nprint('-'*10) # 区切りを挿入\nprint(df_test.columns) # テストデータの列名","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"3c450162-5b21-435b-8441-ad6f5d694e8e","_uuid":"d1e6b10d854fbc251ccc64a1b12c9170d0d31aff"},"cell_type":"markdown","source":"## 4.2.4 df.info()で概要の確認"},{"metadata":{"_cell_guid":"3ad5471b-5454-4eac-b30d-f6bf57aa3b2d","_uuid":"67e289d20f29947e31cb604081d56717c7f30b73","trusted":true},"cell_type":"code","source":"df_train.info()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"e532de39-9b08-42db-992e-c4318904936f","_uuid":"a87131b673744ab239eff9e498ef8691263d5fa7","trusted":true},"cell_type":"code","source":"df_test.info()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"6870e01f-3615-4fab-9a23-9f39b756edd2","_uuid":"f1819c4e25c104000f1bacb2998d3ca99f02adab"},"cell_type":"markdown","source":"## 4.2.5 df.head()で概要の確認"},{"metadata":{"_cell_guid":"f3886b0f-1fbc-4f46-b25b-0441d0795606","_uuid":"2524297c94bb3e70deb76119315dbc5f56c0f05a","trusted":true},"cell_type":"code","source":"df_train.head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"18317cbc-57e8-45fd-8f47-8b9e043a30c9","_uuid":"894ed22c1c8e046bf08522bc933e5397e86c5e3c"},"cell_type":"markdown","source":"## 4.2.6 欠損値がいくつあるか確認"},{"metadata":{"_cell_guid":"7ff6f309-2695-4926-aae7-a0b253043fbf","_uuid":"68d2868664e27c08278d00d68abd58e831c4bf36","trusted":true},"cell_type":"code","source":"df_train.isnull().sum() \n# isnull()は、欠損値に対しTrueを返し、欠損値以外にはFalseを返す\n# sum()は、Trueを1、Falseを0として合計する\n# よってdf.isnull().sum()で欠損値を算出することができる","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"ccf35268-0203-448e-84fa-07dd9c7c6890","_uuid":"0ad8cab93379f06e6ca52c59678743247236c7e6","trusted":true},"cell_type":"code","source":"df_test.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"007e4b96-2835-4ce6-a651-b4aa935549b4","_uuid":"ef8273b0cf5a437d514159ae4b3542658a2cedd9"},"cell_type":"markdown","source":"## 4.2.7 要約統計量の表示"},{"metadata":{"_cell_guid":"eb70bd18-6a21-4983-9e78-9b9c73ec2f40","_uuid":"3fbf9e0f4d3646c300ac4fb1ada8b4b39c6288ff","trusted":true},"cell_type":"code","source":"# df_trainとdf_Testを縦に連結\ndf_full = pd.concat([df_train, df_test], axis = 0, ignore_index=True)\n\nprint(df_full.shape) # df_fullの行数と列数を確認\n\ndf_full.describe() # df_fullの要約統計量","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"097b8b1eed6564cc641f12fa7f5ba8687ce5978d","trusted":true},"cell_type":"code","source":"df_full.describe(include = 'all')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4725d349c61071fd5bd56189fa54a699c18bbf0d","trusted":true},"cell_type":"code","source":"df_full.describe(include=['O'])","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"_uuid":"22dda6f419ea8720137cd55ea5606d3b3f5827a1","trusted":true},"cell_type":"code","source":"df_full.describe(percentiles=[.1, .2, .3, .4, .5, .6, .7, .8, .9, .99])","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"0c63d486-c220-42b3-b0ba-bd42cb47f1bd","_uuid":"96c48bbadd6b5e8ad952a0591b5aa495bc653012"},"cell_type":"markdown","source":"## 4.2.8 死亡者と生存者の可視化"},{"metadata":{"_cell_guid":"a60a6cc3-aa87-4f35-9667-5f0ae35fee7b","scrolled":false,"_uuid":"de4c290f88477b1f5c49dfd8eb45ef05393cff65","trusted":true},"cell_type":"code","source":"sns.countplot(x='Survived', data=df_train)\nplt.title('死亡者と生存者の数')\nplt.xticks([0,1],['死亡者', '生存者'])\nplt.show()\n\n# 死亡者と生存者数を表示する\ndisplay(df_train['Survived'].value_counts())\n\n# 死亡者と生存者割合を表示する\ndisplay(df_train['Survived'].value_counts()/len(df_train['Survived']))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a3e25b44024fef5a1d60b35436274e688197dcb8"},"cell_type":"markdown","source":"## 4.2.9 性別"},{"metadata":{"_cell_guid":"01149b8d-56d8-4563-9527-fa8b96e08fcb","_uuid":"37e1b76310399ab4a23b5a308873b6e8373bca87","trusted":true,"scrolled":false},"cell_type":"code","source":"# 男女別の生存者数を可視化\nsns.countplot(x='Sex', hue='Survived', data=df_train)\n# plt.xticks([0.0,1.0], ['死亡','生存'])\n# plt.title('男女別の死亡者と生存者の数')\n# plt.show()\n\nplt.title('男女別の死亡者と生存者の数')\nplt.legend(['死亡','生存'])\nplt.show()\n\n# SexとSurvivedをクロス集計する\ndisplay(pd.crosstab(df_train['Sex'], df_train['Survived']))\n\n# クロス集計しSexごとに正規化する\ndisplay(pd.crosstab(df_train['Sex'], df_train['Survived'],normalize = 'index'))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"cc348c80-a5cc-4a61-ad49-9ebd1dde8363","_uuid":"6d509af9d95ea21d35f833ceae90487d29a02f06"},"cell_type":"markdown","source":"## 4.2.10 チケットクラス"},{"metadata":{"_cell_guid":"6c1c6f1b-7898-41ed-96f7-4494c8836c55","scrolled":false,"_uuid":"bd4bd3a4815bf50523b1058b60dca4f4a17b6df1","trusted":true},"cell_type":"code","source":"# チケットクラス別の生存者数を可視化\nsns.countplot(x='Pclass', hue='Survived', data=df_train)\nplt.title('チケットクラス別の死亡者と生存者の数')\nplt.legend(['死亡','生存'])\nplt.show()\n\n# PclassとSurvivedをクロス集計する\ndisplay(pd.crosstab(df_train['Pclass'], df_train['Survived']))\n\n# クロス集計しPclassごとに正規化する\ndisplay(pd.crosstab(df_train['Pclass'], df_train['Survived'],normalize = 'index'))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"09032ef1-cea6-4743-81c5-4d4974f09b37","_uuid":"16f1883c4e41b50a2e6e2e969e303f2aee8bade0"},"cell_type":"markdown","source":"## 4.2.11 年齢の分布"},{"metadata":{"_cell_guid":"06617e12-b0aa-4a5a-ba58-7951ed7fb7a3","_uuid":"4694f274b7f86bf440dd465d1f65fab3066d052b","trusted":true},"cell_type":"code","source":"sns.set_palette('pastel')\n# 全体のヒストグラム\nsns.distplot(df_train['Age'].dropna(), kde=False, bins = 30 ,label='全体')\n\n# 死亡者のヒストグラム\nsns.distplot(df_train[df_train['Survived'] == 0].Age.dropna(), kde = False, bins=30, label='死亡')\n\n# 生存者のヒストグラム\nsns.distplot(df_train[df_train['Survived'] == 1].Age.dropna(), kde = False, bins=30, label='生存')\n\nplt.title('乗船者の年齢の分布') # タイトル\nplt.legend() # 凡例を表示;","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"0b218737-ea81-4ae2-80ce-e27d475b92cb","_uuid":"8b6252b5fff2fd3eb03bb67c87f1d01637491390","trusted":true},"cell_type":"code","source":"# 年齢を８等分し、CategoricalAgeという変数を作成\ndf_train['CategoricalAge'] = pd.cut(df_train['Age'], 8)\n\n# CategoricalAgeでグルーピングして、Survivedを平均\ndf_train[['CategoricalAge', 'Survived']].groupby(['CategoricalAge'], as_index=False).mean()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"c8c2b9d2-1afa-48e2-a6bc-2ae80fce7b91","_uuid":"e305dfe02098f61b77823ff6195d85ac9b23b6f2"},"cell_type":"markdown","source":"## 4.2.12 タイタニック号に乗っている兄弟・配偶者の数"},{"metadata":{"_cell_guid":"e0fed74e-35e7-433b-ba31-5718c47bbb6a","scrolled":false,"_uuid":"a68e2985e716e481e7e78b97d49e8393ffb64ca8","trusted":true},"cell_type":"code","source":"sns.countplot(x='SibSp', data = df_train, color='cornflowerblue')\n\nplt.title(\"同乗している兄弟・配偶者の数\", fontsize = 20);","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"82bfd253-3b45-4bb1-b9e5-6f8d199ee71d","_uuid":"eb31f601c53b13b4ead5746bd2479b8d42b0790e","trusted":true},"cell_type":"code","source":"# SibSpが0か1であればそのまま、2以上であれば2である特徴量SibSp_0_1_2overを作成\ndf_train['SibSp_0_1_2over'] = [i if i <=1 else 2 for i in df_train['SibSp']]\n\n# SibSp_0_1_2overごとに集計し、可視化 \nsns.countplot(x = 'SibSp_0_1_2over', hue = 'Survived', data = df_train)\nplt.legend(['死亡', '生存'])\nplt.xticks([0,1,2], ['0人', '1人', '2人以上'])\nplt.title('同乗している兄弟・配偶者の数別の死亡者と生存者の数')\nplt.show()\n\n# SibSpとSurvivedをクロス集計する\ndisplay(pd.crosstab(df_train['SibSp_0_1_2over'], df_train['Survived']))\n\n# クロス集計しSibSpごとに正規化する\ndisplay(pd.crosstab(df_train['SibSp_0_1_2over'], df_train['Survived'], normalize = 'index'))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"1e237f70-9ed6-4b91-ba5a-f26cc913db9e","_uuid":"563c3c0d94e3791693c2b5f92dd81f5a461e4693"},"cell_type":"markdown","source":"## 4.2.12 タイタニック号に乗っている両親・子供の数"},{"metadata":{"_cell_guid":"2e77cf2e-72bf-4c19-9680-acbb472ddce0","scrolled":false,"_uuid":"2905b00a2c17e74849ea6b1e7418a9dfac9cfb66","trusted":true},"cell_type":"code","source":"sns.countplot(x='Parch', data = df_train)\nplt.title('同乗している両親・子供の数');","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"41532c7c-cf50-4f6f-9097-013aae03ad57","_uuid":"7ed50b242ee39fed4c39a0a0a65129201abafc06","trusted":true},"cell_type":"code","source":"# 2以下であればそのままの数、3以上は3という変換を行う\ndf_train['Parch_0_1_2_3over'] = [i if i <=2 else 3 for i in df_train['Parch']]\n\n# Parch_0_1_2_3overごとに集計し可視化\nsns.countplot(x='Parch_0_1_2_3over',hue='Survived', data = df_train)\nplt.title('同乗している両親・子供の数別の死亡者と生存者の数')\nplt.legend(['死亡','生存'])\nplt.xticks([0, 1, 2, 3], ['0人', '1人', '2人', '3人以上'])\nplt.xlabel('Parch')\nplt.show()\n\n# ParchとSurvivedをクロス集計する\ndisplay(pd.crosstab(df_train['Parch_0_1_2_3over'], df_train['Survived']))\n\n# クロス集計しParchごとに正規化する\ndisplay(pd.crosstab(df_train['Parch_0_1_2_3over'], df_train['Survived'], normalize = 'index'))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"342ab50d-a358-47c4-8464-658eeccc1b65","_uuid":"d17ffa80dc32a6f80d372b558424c38d6b020e6e"},"cell_type":"markdown","source":"## 4.2.14 １人で乗船しているか２人以上で乗船しているか"},{"metadata":{"_cell_guid":"17bc8758-b01d-4d5a-b510-ce8e9b2af84e","scrolled":false,"_uuid":"4c7a60c5a8b687a61e8ecfd9aee379eb71d7710b","trusted":true},"cell_type":"code","source":"#SibSpとParchが同乗している家族の数。1を足すと家族の人数となる\ndf_train['FamilySize']=df_train['SibSp']+ df_train['Parch']+ 1\n\n# IsAloneを0とし、2行目でFamilySizeが2以上であれば1にしている\ndf_train['IsAlone'] = 0\ndf_train.loc[df_train['FamilySize'] >= 2, 'IsAlone'] = 1\n\n# IsAloneごとに可視化\nsns.countplot(x='IsAlone', hue = 'Survived', data = df_train)\nplt.xticks([0, 1], ['1人', '2人以上'])\n\nplt.legend(['死亡', '生存'])\nplt.title('１人or２人以上で乗船別の死亡者と生存者の数')\nplt.show()\n\n# IsAloneとSurvivedをクロス集計する\ndisplay(pd.crosstab(df_train['IsAlone'], df_train['Survived']))\n\n# クロス集計しIsAloneごとに正規化する\ndisplay(pd.crosstab(df_train['IsAlone'], df_train['Survived'], normalize = 'index'));","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"5b8202f5-21ad-41e8-bdca-25768e8624e2","_uuid":"9b9ff235b41cd99723eaf66d2d2193967bb38eab"},"cell_type":"markdown","source":"## 4.2.15 運賃の分布"},{"metadata":{"_cell_guid":"1a3a06c8-c278-4f3b-9dcc-c3e94dad0937","_uuid":"c1a86b485f4881ffd990526982c391a763c28884","trusted":true},"cell_type":"code","source":"sns.distplot(df_train['Fare'].dropna(), kde=False, hist=True)\nplt.title('運賃の分布');","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"d8151c9c-be7d-415f-8d06-f62c0c06c78f","scrolled":false,"_uuid":"c1e9bdd73d8dc8ab95ada80c05914045edf3b542","trusted":true},"cell_type":"code","source":"df_train['CategoricalFare'] = pd.qcut(df_train['Fare'], 4)\ndf_train[['CategoricalFare', 'Survived']].groupby(['CategoricalFare'], as_index=False).mean()\n\n# CategoricalFareとSurvivedをクロス集計する\ndisplay(pd.crosstab(df_train['CategoricalFare'], df_train['Survived']))\n\n# クロス集計しCategoricalFareごとに正規化する\ndisplay(pd.crosstab(df_train['CategoricalFare'], df_train['Survived'], normalize = 'index'))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"a5b72c4c-396a-48c6-b449-239c950d8ab9","_uuid":"4aa6433117daad0dc1db28cce3af8d9fc24ab993"},"cell_type":"markdown","source":"## 4.2.16 名前"},{"metadata":{"_cell_guid":"1d70af31-eb61-4466-922f-85673d0109b1","_uuid":"8d5374b6e12102672c147c3f69efb1b5efae72f9","trusted":true},"cell_type":"code","source":"df_test['Name'][0:5]","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"7cb6723e-6f61-4c75-ae61-5eb6d78718d4","_uuid":"81667b2740dc9b1439ed9a7b0d5b01769952dd24","trusted":true},"cell_type":"code","source":"# 敬称を抽出し、重複を省く\nset(df_train.Name.str.extract(' ([A-Za-z]+)\\.', expand=False))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"5a59ebf0-f449-406f-aea4-b1542a51dcd1","_uuid":"291a443d5e92949e91647b6488405f34baf9de03","trusted":true},"cell_type":"code","source":"# collections.Counterを使用して、数え上げる\nimport collections\ncollections.Counter(df_train.Name.str.extract(' ([A-Za-z]+)\\.', expand=False))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"721a5640-9f6b-42d4-aef0-1d61bd8399ae","_uuid":"228159516b525eb318e12ad13d884ae1cc0e0119","trusted":true},"cell_type":"code","source":"# df_trainにTitle列を作成、Title列の値は敬称\ndf_train['Title'] = df_train.Name.str.extract(' ([A-Za-z]+)\\.', expand=False)\n\n# df_testにTitle列を作成、Title列の値は敬称\ndf_test['Title'] = df_test.Name.str.extract(' ([A-Za-z]+)\\.', expand=False)\n\n# df_trainのTitle列の値ごとに平均値を算出\ndf_train.groupby('Title').mean()['Age']","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"88b170c9-436d-42fc-85cc-47d1993e2aa5","_uuid":"e5d7e4902a6a1d51602496b0e8d0b2dcbca3c383","trusted":true},"cell_type":"code","source":"# 変換するための関数を作成\ndef title_to_num(title):\n    if title == 'Master':\n        return 1\n    elif title == 'Miss':\n        return 2\n    elif title == 'Mr':\n        return 3\n    elif title == 'Mrs':\n        return 4\n    else:\n        return 5\n\n# リスト内包表記を用いて変換\ndf_train['Title_num'] = [title_to_num(i) for i in df_train['Title']]\ndf_test['Title_num'] = [title_to_num(i) for i in df_test['Title']]","execution_count":null,"outputs":[]}],"metadata":{"toc":{"navigate_menu":true,"nav_menu":{"width":"252px","height":"318px"},"toc_window_display":true,"threshold":4,"toc_cell":false,"sideBar":true,"toc_section_display":"block","moveMenuLeft":true,"colors":{"hover_highlight":"#DAA520","selected_highlight":"#FFD700","running_highlight":"#FF0000"},"number_sections":false,"widenNotebook":false},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"anaconda-cloud":{},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}