{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-23T01:13:50.205188Z","iopub.execute_input":"2022-07-23T01:13:50.205634Z","iopub.status.idle":"2022-07-23T01:13:50.217748Z","shell.execute_reply.started":"2022-07-23T01:13:50.205589Z","shell.execute_reply":"2022-07-23T01:13:50.216283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas_profiling\nimport statistics as st","metadata":{"execution":{"iopub.status.busy":"2022-07-23T01:13:50.219406Z","iopub.execute_input":"2022-07-23T01:13:50.219984Z","iopub.status.idle":"2022-07-23T01:13:50.228309Z","shell.execute_reply.started":"2022-07-23T01:13:50.219938Z","shell.execute_reply":"2022-07-23T01:13:50.226923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train = pd.read_csv('../input/titanic/train.csv')\ndata_test = pd.read_csv('../input/titanic/test.csv')\ndata_gender_submission = pd.read_csv('../input/titanic/gender_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T01:14:10.443648Z","iopub.execute_input":"2022-07-23T01:14:10.444106Z","iopub.status.idle":"2022-07-23T01:14:10.482460Z","shell.execute_reply.started":"2022-07-23T01:14:10.444072Z","shell.execute_reply":"2022-07-23T01:14:10.481184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-23T01:15:24.352813Z","iopub.execute_input":"2022-07-23T01:15:24.353209Z","iopub.status.idle":"2022-07-23T01:15:24.379013Z","shell.execute_reply.started":"2022-07-23T01:15:24.353175Z","shell.execute_reply":"2022-07-23T01:15:24.378003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train.profile_report()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T01:21:49.457859Z","iopub.execute_input":"2022-07-23T01:21:49.459046Z","iopub.status.idle":"2022-07-23T01:22:04.737347Z","shell.execute_reply.started":"2022-07-23T01:21:49.458994Z","shell.execute_reply":"2022-07-23T01:22:04.736126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_all = pd.concat([data_train, data_test], sort=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T01:30:20.144361Z","iopub.execute_input":"2022-07-23T01:30:20.145462Z","iopub.status.idle":"2022-07-23T01:30:20.156025Z","shell.execute_reply.started":"2022-07-23T01:30:20.145420Z","shell.execute_reply":"2022-07-23T01:30:20.154924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_all","metadata":{"execution":{"iopub.status.busy":"2022-07-23T01:30:37.975431Z","iopub.execute_input":"2022-07-23T01:30:37.975868Z","iopub.status.idle":"2022-07-23T01:30:38.003269Z","shell.execute_reply.started":"2022-07-23T01:30:37.975833Z","shell.execute_reply":"2022-07-23T01:30:38.001771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_all.drop(['PassengerId', 'Name', 'Parch', 'SibSp', 'Ticket', 'Cabin'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T01:31:37.500146Z","iopub.execute_input":"2022-07-23T01:31:37.500556Z","iopub.status.idle":"2022-07-23T01:31:37.508482Z","shell.execute_reply.started":"2022-07-23T01:31:37.500523Z","shell.execute_reply":"2022-07-23T01:31:37.507167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_all.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T01:31:58.060050Z","iopub.execute_input":"2022-07-23T01:31:58.060461Z","iopub.status.idle":"2022-07-23T01:31:58.075631Z","shell.execute_reply.started":"2022-07-23T01:31:58.060427Z","shell.execute_reply":"2022-07-23T01:31:58.074488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#年齢は分散が小さいので、平均値を補完\ndata_all['Age'].fillna(np.mean(data_all['Age']), inplace=True)\n#運賃は分散が大きいので、中央値を補完\ndata_all['Fare'].fillna(np.nanmedian(data_all['Fare']), inplace=True)\n#搭乗港は文字列なので、最頻値を補完\ndata_all['Embarked'].fillna(st.mode(data_all['Embarked']), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T01:34:43.219629Z","iopub.execute_input":"2022-07-23T01:34:43.220054Z","iopub.status.idle":"2022-07-23T01:34:43.230065Z","shell.execute_reply.started":"2022-07-23T01:34:43.220021Z","shell.execute_reply":"2022-07-23T01:34:43.228940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#文字列を数値へ変換\ndata_all['Sex'].replace(['male', 'female'], [0, 1], inplace=True)\ndata_all['Embarked'].replace(['S', 'C', 'Q'], [0, 1, 2], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T01:35:08.477503Z","iopub.execute_input":"2022-07-23T01:35:08.477954Z","iopub.status.idle":"2022-07-23T01:35:08.490716Z","shell.execute_reply.started":"2022-07-23T01:35:08.477922Z","shell.execute_reply":"2022-07-23T01:35:08.489183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train = data_all[~data_all['Survived'].isnull()]\ndata_test = data_all[data_all['Survived'].isnull()]","metadata":{"execution":{"iopub.status.busy":"2022-07-23T01:35:27.235235Z","iopub.execute_input":"2022-07-23T01:35:27.235645Z","iopub.status.idle":"2022-07-23T01:35:27.244373Z","shell.execute_reply.started":"2022-07-23T01:35:27.235610Z","shell.execute_reply":"2022-07-23T01:35:27.243162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#訓練データ\ny_train = data_train['Survived']\nX_train = data_train.drop('Survived', axis=1)\n#検証データ\nX_test = data_test.drop('Survived', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T01:36:20.513973Z","iopub.execute_input":"2022-07-23T01:36:20.514354Z","iopub.status.idle":"2022-07-23T01:36:20.522079Z","shell.execute_reply.started":"2022-07-23T01:36:20.514321Z","shell.execute_reply":"2022-07-23T01:36:20.521246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ランダムフォレストライブラリのインポート\nfrom sklearn.ensemble import RandomForestClassifier\n#ランダムフォレストのパラメータ指定(デフォルト)\nclf = RandomForestClassifier()\n#訓練データを元にモデルを作成\nclf.fit(X_train, y_train)\n#検証データの予測を出力\ny_test = clf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T01:37:16.530423Z","iopub.execute_input":"2022-07-23T01:37:16.530860Z","iopub.status.idle":"2022-07-23T01:37:17.120007Z","shell.execute_reply.started":"2022-07-23T01:37:16.530824Z","shell.execute_reply":"2022-07-23T01:37:17.118845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result = pd.DataFrame(y_test).astype(int)\nresult.columns = ['Survived']\nsubmit = pd.concat([data_gender_submission['PassengerId'].astype(int), result],axis=1)\nsubmit.to_csv('submit_v1.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T01:41:32.590865Z","iopub.execute_input":"2022-07-23T01:41:32.591260Z","iopub.status.idle":"2022-07-23T01:41:32.604402Z","shell.execute_reply.started":"2022-07-23T01:41:32.591228Z","shell.execute_reply":"2022-07-23T01:41:32.603189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}