{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))  \n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8d23d26f38d440d89c7036416823aeb16a2c98fc"},"cell_type":"code","source":"import matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"#load the data\ndf_train = pd.read_csv('../input/train.csv')\ndf_test = pd.read_csv('../input/test.csv')\ndf_all = [df_train, df_test]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f3f832fd608c0643943a27ccd7dfab350e734d0f"},"cell_type":"code","source":"#show first(or last) 5 rows of the train data\ndf_train.head() #train_df.tail()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8cb3c5db549dafc569464314d54d57728b188d45"},"cell_type":"code","source":"df_train.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a2331600627a14f567330f6e1b5bd6fa8a7b3a9a"},"cell_type":"code","source":"df_all[:891].info()\nprint('*'*40)\ndf_all[891:].info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"750681741b3f390e59011b6f51b2a53c0f26a84a"},"cell_type":"code","source":"## sex pivot\nsex_pivot=df_train.pivot_table(index=\"Sex\",values=\"Survived\")\nsex_pivot.plot.bar()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4687f9ade79642dbfcf3492f49f17a23f046d8bb"},"cell_type":"code","source":"# class pivot\nclass_pivot = df_train.pivot_table(index=\"Pclass\",values=\"Survived\")\nclass_pivot.plot.bar()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"d3150e2eb4dce7ac07ca8d218b2742f5f3a6aecb"},"cell_type":"code","source":"## fare and sex pivot\nfare_cut_train=pd.cut(df_train[\"Fare\"],[0,5,10,25,50,75,100,10000])\ndf_train[\"fare_cut\"]=fare_cut_train\nfare_cut_test=pd.cut(df_test[\"Fare\"],[0,5,10,25,50,75,100,10000])\ndf_test[\"fare_cut\"]=fare_cut_test\ndf_train.pivot_table(\"Survived\",index=\"fare_cut\",columns='Sex',aggfunc='mean')\ndf_train.pivot_table(\"Survived\",index=\"fare_cut\",columns='Sex',aggfunc='count')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b95b82bd5278510dde499a40fadd19e22ccbc423"},"cell_type":"code","source":"## Age investigation\ndf_train[\"Age\"].describe()\nsurvived = df_train[df_train[\"Survived\"] == 1]\ndied = df_train[df_train[\"Survived\"] == 0]\nsurvived[\"Age\"].plot.hist(alpha=0.5,color='red',bins=50)\ndied[\"Age\"].plot.hist(alpha=0.5,color='blue',bins=50)\nplt.legend(['Survived','Died'])\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3b3adb1fa52f3e40f6c1566c1d8dbceab0358575"},"cell_type":"code","source":"## create age categories variable based on age\ndef process_age(df,cut_points,label_names):\n    df[\"Age\"] = df[\"Age\"].fillna(-0.5)\n    df[\"Age_categories\"] = pd.cut(df[\"Age\"],cut_points,labels=label_names)\n    return df\n\ncut_points = [-1,0,5,12,18,35,60,100]\nlabel_names = [\"Missing\",\"Infant\",\"Child\",\"Teenager\",\"Young Adult\",\"Adult\",\"Senior\"]\n\ndf_train = process_age(df_train,cut_points,label_names)\ndf_test = process_age(df_test,cut_points,label_names)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0b6a99086d722299bd2e0c2c67d6d4555272deb7"},"cell_type":"code","source":"## show survival rate by age categories\npivot = df_train.pivot_table(index=\"Age_categories\",values='Survived')\npivot.plot.bar()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7ab1aea0928439b4942d54ca81a4459ed4147a6f"},"cell_type":"code","source":"df_train.pivot_table(index=\"Parch\",values=\"Survived\")\ndf_train.pivot_table(index=\"Parch\",columns='Sex',values=\"Survived\")\ndf_train.pivot_table(index=\"Parch\",columns='Age_categories',values=\"Survived\")\ndf_train.pivot_table(index=\"Age_categories\",columns='Parch',values=\"Survived\")\ndf_train.pivot_table(index=\"Age_categories\",columns='Parch',values=\"Survived\",aggfunc='count')\ndf_train.pivot_table(index=\"Parch\",values=\"Survived\",aggfunc=\"count\")\ndf_train.pivot_table(index=\"SibSp\",values=\"Survived\")\ndf_train.pivot_table(index=\"SibSp\",values=\"Survived\",aggfunc=\"count\")","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}