{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-07T07:09:34.793802Z","iopub.execute_input":"2022-07-07T07:09:34.794346Z","iopub.status.idle":"2022-07-07T07:09:34.836470Z","shell.execute_reply.started":"2022-07-07T07:09:34.794232Z","shell.execute_reply":"2022-07-07T07:09:34.835664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 訓練データの読み込み\nTitanic号データの訓練用サブセットを読み込む","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/titanic/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:34.838492Z","iopub.execute_input":"2022-07-07T07:09:34.838874Z","iopub.status.idle":"2022-07-07T07:09:34.863226Z","shell.execute_reply.started":"2022-07-07T07:09:34.838843Z","shell.execute_reply":"2022-07-07T07:09:34.861960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 特徴量を確認","metadata":{}},{"cell_type":"code","source":"train.describe","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:34.864664Z","iopub.execute_input":"2022-07-07T07:09:34.865472Z","iopub.status.idle":"2022-07-07T07:09:34.894831Z","shell.execute_reply.started":"2022-07-07T07:09:34.865430Z","shell.execute_reply":"2022-07-07T07:09:34.893533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:34.897955Z","iopub.execute_input":"2022-07-07T07:09:34.898678Z","iopub.status.idle":"2022-07-07T07:09:34.921968Z","shell.execute_reply.started":"2022-07-07T07:09:34.898630Z","shell.execute_reply":"2022-07-07T07:09:34.920693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ntrain = pd.read_csv(\"../input/titanic/train.csv\")\ntest = pd.read_csv(\"../input/titanic/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:34.923850Z","iopub.execute_input":"2022-07-07T07:09:34.924304Z","iopub.status.idle":"2022-07-07T07:09:34.945161Z","shell.execute_reply.started":"2022-07-07T07:09:34.924262Z","shell.execute_reply":"2022-07-07T07:09:34.943438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"../input/titanic/gender_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:34.947595Z","iopub.execute_input":"2022-07-07T07:09:34.948438Z","iopub.status.idle":"2022-07-07T07:09:34.961357Z","shell.execute_reply.started":"2022-07-07T07:09:34.948387Z","shell.execute_reply":"2022-07-07T07:09:34.960034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nsns.countplot(x = \"SibSp\", hue = \"Survived\", data = train)\nplt.legend(loc = \"upper right\", title = \"Survived~Sibsp\")","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:34.962989Z","iopub.execute_input":"2022-07-07T07:09:34.963971Z","iopub.status.idle":"2022-07-07T07:09:36.728234Z","shell.execute_reply.started":"2022-07-07T07:09:34.963932Z","shell.execute_reply":"2022-07-07T07:09:36.726905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nsns.displot(data = train, x = \"Fare\", hue = \"Survived\", kde=False, rug=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:36.730738Z","iopub.execute_input":"2022-07-07T07:09:36.731267Z","iopub.status.idle":"2022-07-07T07:09:38.101493Z","shell.execute_reply.started":"2022-07-07T07:09:36.731220Z","shell.execute_reply":"2022-07-07T07:09:38.100069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 欠損値の処理","metadata":{}},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:38.106668Z","iopub.execute_input":"2022-07-07T07:09:38.107091Z","iopub.status.idle":"2022-07-07T07:09:38.119122Z","shell.execute_reply.started":"2022-07-07T07:09:38.107056Z","shell.execute_reply":"2022-07-07T07:09:38.117703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop([\"PassengerId\", \"Name\", \"Cabin\", \"Ticket\"], axis=1, inplace=True)\ntrain[\"Age\"].fillna(train[\"Age\"].median(skipna=True), inplace=True)\ntrain[\"Embarked\"].fillna(train[\"Embarked\"].value_counts().idxmax(), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:38.120994Z","iopub.execute_input":"2022-07-07T07:09:38.122274Z","iopub.status.idle":"2022-07-07T07:09:38.140284Z","shell.execute_reply.started":"2022-07-07T07:09:38.122216Z","shell.execute_reply":"2022-07-07T07:09:38.138639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 特徴量の追加","metadata":{}},{"cell_type":"code","source":"import numpy as np\ntrain[\"Alone\"] = np.where((train[\"SibSp\"]+train[\"Parch\"]) > 0, 0, 1)\ntrain.drop([\"SibSp\", \"Parch\"], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:38.141730Z","iopub.execute_input":"2022-07-07T07:09:38.142571Z","iopub.status.idle":"2022-07-07T07:09:38.159207Z","shell.execute_reply.started":"2022-07-07T07:09:38.142530Z","shell.execute_reply":"2022-07-07T07:09:38.157376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ダミー変数","metadata":{}},{"cell_type":"code","source":"pd.get_dummies(train[\"Sex\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:38.165355Z","iopub.execute_input":"2022-07-07T07:09:38.166926Z","iopub.status.idle":"2022-07-07T07:09:38.184855Z","shell.execute_reply.started":"2022-07-07T07:09:38.166881Z","shell.execute_reply":"2022-07-07T07:09:38.183414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training = pd.get_dummies(train, columns = [\"Pclass\", \"Embarked\", \"Sex\"], drop_first=True)\ntraining","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:38.186793Z","iopub.execute_input":"2022-07-07T07:09:38.187647Z","iopub.status.idle":"2022-07-07T07:09:38.217083Z","shell.execute_reply.started":"2022-07-07T07:09:38.187566Z","shell.execute_reply":"2022-07-07T07:09:38.215726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## AgeとFare標準化","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\ntrain_standard = StandardScaler()\ntrain_copied = training.copy()\ntrain_standard.fit(train_copied[[\"Age\",\"Fare\"]])\ntrain_std = pd.DataFrame(train_standard.transform(train_copied[[\"Age\", \"Fare\"]]))\ntraining[[\"Age\", \"Fare\"]] = train_std","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:38.219489Z","iopub.execute_input":"2022-07-07T07:09:38.220361Z","iopub.status.idle":"2022-07-07T07:09:38.428436Z","shell.execute_reply.started":"2022-07-07T07:09:38.220305Z","shell.execute_reply":"2022-07-07T07:09:38.426980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## モデル適用","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\ncols = [\"Age\",\"Fare\",\"Alone\",\"Pclass_2\",\"Pclass_2\",\"Embarked_Q\",\"Embarked_S\",\"Sex_male\"] \nX = training[cols]\ny = training['Survived']\n# Build a logreg and compute the feature importances\nmodel = LogisticRegression()\n# create the RFE model and select 8 attributes\nmodel.fit(X,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:38.430693Z","iopub.execute_input":"2022-07-07T07:09:38.431564Z","iopub.status.idle":"2022-07-07T07:09:38.650825Z","shell.execute_reply.started":"2022-07-07T07:09:38.431509Z","shell.execute_reply":"2022-07-07T07:09:38.649490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\ntrain_predicted = model.predict(X)\naccuracy_score(train_predicted, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:38.652905Z","iopub.execute_input":"2022-07-07T07:09:38.653730Z","iopub.status.idle":"2022-07-07T07:09:38.671481Z","shell.execute_reply.started":"2022-07-07T07:09:38.653675Z","shell.execute_reply":"2022-07-07T07:09:38.669282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## テストデータ","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv(\"../input/titanic/test.csv\")\ntest.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:38.673876Z","iopub.execute_input":"2022-07-07T07:09:38.674901Z","iopub.status.idle":"2022-07-07T07:09:38.696860Z","shell.execute_reply.started":"2022-07-07T07:09:38.674844Z","shell.execute_reply":"2022-07-07T07:09:38.695667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.drop([\"PassengerId\", \"Name\", \"Cabin\", \"Ticket\"], axis=1, inplace=True)\ntest[\"Age\"].fillna(28, inplace=True)\ntest[\"Embarked\"].fillna(test[\"Embarked\"].value_counts().idxmax(), inplace=True)\ntest[\"Fare\"].fillna(train.Fare.median(), inplace=True)\n\ntest['Alone']=np.where((test['SibSp'] + test['Parch'])>0, 0, 1)\ntest.drop(['SibSp', 'Parch'], axis=1, inplace=True)\n\ntesting = pd.get_dummies(test, columns=['Pclass', 'Embarked', 'Sex'], drop_first=True)\ntesting","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:38.698822Z","iopub.execute_input":"2022-07-07T07:09:38.699562Z","iopub.status.idle":"2022-07-07T07:09:38.745186Z","shell.execute_reply.started":"2022-07-07T07:09:38.699514Z","shell.execute_reply":"2022-07-07T07:09:38.744199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_copied = testing.copy()\ntest_std = pd.DataFrame(train_standard.transform(test_copied[['Age','Fare']]))\ntesting [['Age','Fare']] = test_std","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:38.746560Z","iopub.execute_input":"2022-07-07T07:09:38.747149Z","iopub.status.idle":"2022-07-07T07:09:38.757949Z","shell.execute_reply.started":"2022-07-07T07:09:38.747115Z","shell.execute_reply":"2022-07-07T07:09:38.756622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = [\"Age\",\"Fare\",\"Alone\",\"Pclass_2\",\"Pclass_2\",\"Embarked_Q\",\"Embarked_S\",\"Sex_male\"] \nX_test = testing[cols]\nprint(X_test.dtypes)\ntest_predicted = model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:38.759705Z","iopub.execute_input":"2022-07-07T07:09:38.761324Z","iopub.status.idle":"2022-07-07T07:09:38.777634Z","shell.execute_reply.started":"2022-07-07T07:09:38.761256Z","shell.execute_reply":"2022-07-07T07:09:38.776293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 結果をコミット","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv('../input/titanic/gender_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:38.782904Z","iopub.execute_input":"2022-07-07T07:09:38.783979Z","iopub.status.idle":"2022-07-07T07:09:38.793652Z","shell.execute_reply.started":"2022-07-07T07:09:38.783927Z","shell.execute_reply":"2022-07-07T07:09:38.792108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['Survived'] = list(map(int, test_predicted))\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T07:09:38.795199Z","iopub.execute_input":"2022-07-07T07:09:38.795866Z","iopub.status.idle":"2022-07-07T07:09:38.821614Z","shell.execute_reply.started":"2022-07-07T07:09:38.795825Z","shell.execute_reply":"2022-07-07T07:09:38.820410Z"},"trusted":true},"execution_count":null,"outputs":[]}]}