{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T07:05:03.501692Z","iopub.execute_input":"2022-07-12T07:05:03.502077Z","iopub.status.idle":"2022-07-12T07:05:03.509397Z","shell.execute_reply.started":"2022-07-12T07:05:03.502043Z","shell.execute_reply":"2022-07-12T07:05:03.508712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/tabular-playground-series-apr-2021/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:05:03.516305Z","iopub.execute_input":"2022-07-12T07:05:03.516598Z","iopub.status.idle":"2022-07-12T07:05:03.701551Z","shell.execute_reply.started":"2022-07-12T07:05:03.516571Z","shell.execute_reply":"2022-07-12T07:05:03.700393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:05:03.703624Z","iopub.execute_input":"2022-07-12T07:05:03.703965Z","iopub.status.idle":"2022-07-12T07:05:03.718842Z","shell.execute_reply.started":"2022-07-12T07:05:03.703938Z","shell.execute_reply":"2022-07-12T07:05:03.717911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:05:03.720205Z","iopub.execute_input":"2022-07-12T07:05:03.720481Z","iopub.status.idle":"2022-07-12T07:05:03.754345Z","shell.execute_reply.started":"2022-07-12T07:05:03.720454Z","shell.execute_reply":"2022-07-12T07:05:03.753360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\ntrain = pd.read_csv(\"/kaggle/input/tabular-playground-series-apr-2021/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/tabular-playground-series-apr-2021/test.csv\")\n\nsub = pd.read_csv(\"/kaggle/input/tabular-playground-series-apr-2021/sample_submission.csv\")\n\nsns.countplot(x = 'SibSp', hue = \"Survived\", data = train)\nplt.legend(loc = \"upper right\", title = \"Survived ~ Sibsp\")","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:05:03.757249Z","iopub.execute_input":"2022-07-12T07:05:03.757657Z","iopub.status.idle":"2022-07-12T07:05:04.354233Z","shell.execute_reply.started":"2022-07-12T07:05:03.757616Z","shell.execute_reply":"2022-07-12T07:05:04.352487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"SibspとSurvivedの影響あり","metadata":{}},{"cell_type":"code","source":"sns.countplot(x = 'Parch', hue = \"Survived\", data = train)\nplt.legend(loc = \"upper right\", title = \"Survived ~ Parch\")","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:05:04.355789Z","iopub.execute_input":"2022-07-12T07:05:04.356228Z","iopub.status.idle":"2022-07-12T07:05:04.587682Z","shell.execute_reply.started":"2022-07-12T07:05:04.356187Z","shell.execute_reply":"2022-07-12T07:05:04.586415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x = 'Embarked', hue = \"Survived\", data = train)\nplt.legend(loc = \"upper right\", title = \"Survived ~ Emberked\")","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:05:04.589261Z","iopub.execute_input":"2022-07-12T07:05:04.589767Z","iopub.status.idle":"2022-07-12T07:05:04.835622Z","shell.execute_reply.started":"2022-07-12T07:05:04.589723Z","shell.execute_reply":"2022-07-12T07:05:04.834660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.displot(data=train,x='Fare', hue='Survived', kde=False,rug=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:05:04.836938Z","iopub.execute_input":"2022-07-12T07:05:04.837530Z","iopub.status.idle":"2022-07-12T07:05:09.137439Z","shell.execute_reply.started":"2022-07-12T07:05:04.837492Z","shell.execute_reply":"2022-07-12T07:05:09.136556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:05:09.138573Z","iopub.execute_input":"2022-07-12T07:05:09.138856Z","iopub.status.idle":"2022-07-12T07:05:09.176935Z","shell.execute_reply.started":"2022-07-12T07:05:09.138830Z","shell.execute_reply":"2022-07-12T07:05:09.175892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(['PassengerId','Name','Cabin','Ticket'], axis=1, inplace=True)\ntrain['Age'].fillna(train['Age'].median(skipna=True), inplace=True)\ntrain['Fare'].fillna(train['Fare'].median(skipna=True), inplace=True)\ntrain['Embarked'].fillna(train['Embarked'].value_counts().idxmax(), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:05:09.178382Z","iopub.execute_input":"2022-07-12T07:05:09.178694Z","iopub.status.idle":"2022-07-12T07:05:09.209067Z","shell.execute_reply.started":"2022-07-12T07:05:09.178667Z","shell.execute_reply":"2022-07-12T07:05:09.208172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:05:09.211952Z","iopub.execute_input":"2022-07-12T07:05:09.212733Z","iopub.status.idle":"2022-07-12T07:05:09.231953Z","shell.execute_reply.started":"2022-07-12T07:05:09.212702Z","shell.execute_reply":"2022-07-12T07:05:09.230827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\ntrain['Alone']=np.where((train['SibSp'] + train['Parch'])>0, 0, 1)\ntrain.drop(['SibSp', 'Parch'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:05:09.233257Z","iopub.execute_input":"2022-07-12T07:05:09.233538Z","iopub.status.idle":"2022-07-12T07:05:09.245846Z","shell.execute_reply.started":"2022-07-12T07:05:09.233513Z","shell.execute_reply":"2022-07-12T07:05:09.245033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.get_dummies(train['Sex'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:05:34.605244Z","iopub.execute_input":"2022-07-12T07:05:34.605604Z","iopub.status.idle":"2022-07-12T07:05:34.626370Z","shell.execute_reply.started":"2022-07-12T07:05:34.605575Z","shell.execute_reply":"2022-07-12T07:05:34.625292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training = pd.get_dummies(train, columns=['Pclass', 'Embarked', 'Sex'], drop_first=True)\ntraining","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:05:49.660083Z","iopub.execute_input":"2022-07-12T07:05:49.660442Z","iopub.status.idle":"2022-07-12T07:05:49.703423Z","shell.execute_reply.started":"2022-07-12T07:05:49.660413Z","shell.execute_reply":"2022-07-12T07:05:49.702578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\ntrain_standard = StandardScaler()\ntrain_copied = training.copy()\ntrain_standard.fit(train_copied[['Age','Fare']])\ntrain_std = pd.DataFrame(train_standard.transform(train_copied[['Age','Fare']]))\ntraining [['Age','Fare']] = train_std","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:06:16.983905Z","iopub.execute_input":"2022-07-12T07:06:16.984270Z","iopub.status.idle":"2022-07-12T07:06:17.000806Z","shell.execute_reply.started":"2022-07-12T07:06:16.984241Z","shell.execute_reply":"2022-07-12T07:06:17.000038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:06:36.486542Z","iopub.execute_input":"2022-07-12T07:06:36.486907Z","iopub.status.idle":"2022-07-12T07:06:36.505237Z","shell.execute_reply.started":"2022-07-12T07:06:36.486876Z","shell.execute_reply":"2022-07-12T07:06:36.504421Z"},"trusted":true},"execution_count":null,"outputs":[]}]}