{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-24T07:47:13.175616Z","iopub.execute_input":"2022-06-24T07:47:13.176028Z","iopub.status.idle":"2022-06-24T07:47:13.184958Z","shell.execute_reply.started":"2022-06-24T07:47:13.175995Z","shell.execute_reply":"2022-06-24T07:47:13.183606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 訓練データの読み込み\nTitanic号データの訓練用サブセットを読み込む。","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/titanic/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:48:20.494938Z","iopub.execute_input":"2022-06-24T07:48:20.49535Z","iopub.status.idle":"2022-06-24T07:48:20.515686Z","shell.execute_reply.started":"2022-06-24T07:48:20.495316Z","shell.execute_reply":"2022-06-24T07:48:20.513948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 特徴量を確認する","metadata":{}},{"cell_type":"code","source":"train.describe","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:49:47.121862Z","iopub.execute_input":"2022-06-24T07:49:47.122301Z","iopub.status.idle":"2022-06-24T07:49:47.148427Z","shell.execute_reply.started":"2022-06-24T07:49:47.122269Z","shell.execute_reply":"2022-06-24T07:49:47.146944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.corr()","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:50:18.697776Z","iopub.execute_input":"2022-06-24T07:50:18.698159Z","iopub.status.idle":"2022-06-24T07:50:18.720769Z","shell.execute_reply.started":"2022-06-24T07:50:18.698131Z","shell.execute_reply":"2022-06-24T07:50:18.719955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nsns.countplot(x = 'SibSp', hue = \"Survived\", data = train)\nplt.legend(loc = \"upper right\", title = \"Survived ~ Sibsp\")","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:52:01.108951Z","iopub.execute_input":"2022-06-24T07:52:01.109483Z","iopub.status.idle":"2022-06-24T07:52:01.85984Z","shell.execute_reply.started":"2022-06-24T07:52:01.10944Z","shell.execute_reply":"2022-06-24T07:52:01.85876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nsns.displot(data=train,x='Fare', hue='Survived', kde=False,rug=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:52:58.837254Z","iopub.execute_input":"2022-06-24T07:52:58.837947Z","iopub.status.idle":"2022-06-24T07:52:59.815795Z","shell.execute_reply.started":"2022-06-24T07:52:58.83791Z","shell.execute_reply":"2022-06-24T07:52:59.81473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:53:07.796851Z","iopub.execute_input":"2022-06-24T07:53:07.797262Z","iopub.status.idle":"2022-06-24T07:53:07.808191Z","shell.execute_reply.started":"2022-06-24T07:53:07.79723Z","shell.execute_reply":"2022-06-24T07:53:07.807094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(['PassengerId','Name','Cabin','Ticket'], axis=1, inplace=True)\ntrain['Age'].fillna(train['Age'].median(skipna=True), inplace=True)\ntrain['Embarked'].fillna(train['Embarked'].value_counts().idxmax(), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:53:18.558808Z","iopub.execute_input":"2022-06-24T07:53:18.559181Z","iopub.status.idle":"2022-06-24T07:53:18.568998Z","shell.execute_reply.started":"2022-06-24T07:53:18.559152Z","shell.execute_reply":"2022-06-24T07:53:18.568188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\ntrain['Alone']=np.where((train['SibSp'] + train['Parch'])>0, 0, 1)\ntrain.drop(['SibSp', 'Parch'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:53:29.773672Z","iopub.execute_input":"2022-06-24T07:53:29.774078Z","iopub.status.idle":"2022-06-24T07:53:29.782683Z","shell.execute_reply.started":"2022-06-24T07:53:29.774044Z","shell.execute_reply":"2022-06-24T07:53:29.781769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.get_dummies(train['Sex'])","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:53:39.70922Z","iopub.execute_input":"2022-06-24T07:53:39.709623Z","iopub.status.idle":"2022-06-24T07:53:39.72561Z","shell.execute_reply.started":"2022-06-24T07:53:39.709591Z","shell.execute_reply":"2022-06-24T07:53:39.724742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training = pd.get_dummies(train, columns=['Pclass', 'Embarked', 'Sex'], drop_first=True)\ntraining","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:53:47.965249Z","iopub.execute_input":"2022-06-24T07:53:47.965595Z","iopub.status.idle":"2022-06-24T07:53:47.99434Z","shell.execute_reply.started":"2022-06-24T07:53:47.965568Z","shell.execute_reply":"2022-06-24T07:53:47.993284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\ntrain_standard = StandardScaler()\ntrain_copied = training.copy()\ntrain_standard.fit(train_copied[['Age','Fare']])\ntrain_std = pd.DataFrame(train_standard.transform(train_copied[['Age','Fare']]))\ntrain_std","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:53:58.509335Z","iopub.execute_input":"2022-06-24T07:53:58.509694Z","iopub.status.idle":"2022-06-24T07:53:58.596913Z","shell.execute_reply.started":"2022-06-24T07:53:58.509665Z","shell.execute_reply":"2022-06-24T07:53:58.595833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\ncols = [\"Age\",\"Fare\",\"Alone\",\"Pclass_2\",\"Pclass_2\",\"Embarked_Q\",\"Embarked_S\",\"Sex_male\"] \nX = training[cols]\ny = training['Survived']\n# Build a logreg and compute the feature importances\nmodel = LogisticRegression()\n# create the RFE model and select 8 attributes\nmodel.fit(X,y)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:54:15.028402Z","iopub.execute_input":"2022-06-24T07:54:15.028783Z","iopub.status.idle":"2022-06-24T07:54:15.226791Z","shell.execute_reply.started":"2022-06-24T07:54:15.028752Z","shell.execute_reply":"2022-06-24T07:54:15.225573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\n\ntrain_predicted = model.predict(X)\naccuracy_score(train_predicted, y)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:54:27.213511Z","iopub.execute_input":"2022-06-24T07:54:27.213842Z","iopub.status.idle":"2022-06-24T07:54:27.222642Z","shell.execute_reply.started":"2022-06-24T07:54:27.213816Z","shell.execute_reply":"2022-06-24T07:54:27.221747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/titanic/test.csv')\ntest.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:55:31.34297Z","iopub.execute_input":"2022-06-24T07:55:31.343367Z","iopub.status.idle":"2022-06-24T07:55:31.359815Z","shell.execute_reply.started":"2022-06-24T07:55:31.343336Z","shell.execute_reply":"2022-06-24T07:55:31.35883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.drop(['PassengerId','Name','Cabin','Ticket'], axis=1, inplace=True)\ntest[\"Age\"].fillna(28, inplace=True)\ntest[\"Embarked\"].fillna(test['Embarked'].value_counts().idxmax(), inplace=True)\ntest[\"Fare\"].fillna(train.Fare.median(), inplace=True)\ntest['Alone']=np.where((test[\"SibSp\"]+test[\"Parch\"])>0, 0, 1)\ntest.drop(['SibSp', 'Parch'], axis=1, inplace=True)\ntesting=pd.get_dummies(test, columns=[\"Pclass\",\"Embarked\",\"Sex\"], drop_first=True)\nprint(testing.dtypes)\ntest_copied = testing.copy()\ntest_std = train_standard.transform(test_copied[['Age','Fare']])\ntest_std\ntesting[['Age','Fare']] = test_std\ntesting","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:55:44.21504Z","iopub.execute_input":"2022-06-24T07:55:44.215432Z","iopub.status.idle":"2022-06-24T07:55:44.251479Z","shell.execute_reply.started":"2022-06-24T07:55:44.215398Z","shell.execute_reply":"2022-06-24T07:55:44.250549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = [\"Age\",\"Fare\",\"Alone\",\"Pclass_2\",\"Pclass_2\",\"Embarked_Q\",\"Embarked_S\",\"Sex_male\"] \nX_test=testing[cols]\nprint(X_test.dtypes)\ntest_predicted = model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:55:53.927539Z","iopub.execute_input":"2022-06-24T07:55:53.927901Z","iopub.status.idle":"2022-06-24T07:55:53.939425Z","shell.execute_reply.started":"2022-06-24T07:55:53.927857Z","shell.execute_reply":"2022-06-24T07:55:53.938477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('../input/titanic/gender_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:56:29.943761Z","iopub.execute_input":"2022-06-24T07:56:29.944175Z","iopub.status.idle":"2022-06-24T07:56:29.955335Z","shell.execute_reply.started":"2022-06-24T07:56:29.944143Z","shell.execute_reply":"2022-06-24T07:56:29.954166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['Survived'] = list(map(int, test_predicted))\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T07:56:32.124247Z","iopub.execute_input":"2022-06-24T07:56:32.12463Z","iopub.status.idle":"2022-06-24T07:56:32.134603Z","shell.execute_reply.started":"2022-06-24T07:56:32.124601Z","shell.execute_reply":"2022-06-24T07:56:32.133468Z"},"trusted":true},"execution_count":null,"outputs":[]}]}