{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Data Analysis to see what features make sense\nPlotting\n\nBased on \"A Data Science Framework: To Achieve 99% Accuracy\" https://www.kaggle.com/code/ldfreeman3/a-data-science-framework-to-achieve-99-accuracy for Titanic competition","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2022-08-19T23:21:54.589572Z","iopub.execute_input":"2022-08-19T23:21:54.590044Z","iopub.status.idle":"2022-08-19T23:21:54.623817Z","shell.execute_reply.started":"2022-08-19T23:21:54.589935Z","shell.execute_reply":"2022-08-19T23:21:54.622930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_DIR = '/kaggle/input/titanic/'\ndata_raw = pd.read_csv(_DIR+'train.csv')\n\ndata_val = pd.read_csv(_DIR+'test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-19T23:25:22.543954Z","iopub.execute_input":"2022-08-19T23:25:22.544449Z","iopub.status.idle":"2022-08-19T23:25:22.565837Z","shell.execute_reply.started":"2022-08-19T23:25:22.544409Z","shell.execute_reply":"2022-08-19T23:25:22.564354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#1st way to find out null values of each column\ndata_raw.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-19T23:21:54.649787Z","iopub.execute_input":"2022-08-19T23:21:54.650111Z","iopub.status.idle":"2022-08-19T23:21:54.677970Z","shell.execute_reply.started":"2022-08-19T23:21:54.650081Z","shell.execute_reply":"2022-08-19T23:21:54.676719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#2nd way to find out null value count of the each column\ndata_raw.isnull().sum()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-19T23:21:54.680951Z","iopub.execute_input":"2022-08-19T23:21:54.681615Z","iopub.status.idle":"2022-08-19T23:21:54.694451Z","shell.execute_reply.started":"2022-08-19T23:21:54.681576Z","shell.execute_reply":"2022-08-19T23:21:54.693056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Impute the missing value","metadata":{}},{"cell_type":"code","source":"#Simply using fillna of pd\ndata1 = data_raw.copy(deep = True)\ndata1['Age'].fillna(data1['Age'].median(), inplace = True)\ndata1['Embarked'].fillna(data1['Embarked'].mode()[0], inplace = True)\n\n\nprint(data1.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-19T23:21:54.695913Z","iopub.execute_input":"2022-08-19T23:21:54.696508Z","iopub.status.idle":"2022-08-19T23:21:54.709446Z","shell.execute_reply.started":"2022-08-19T23:21:54.696475Z","shell.execute_reply":"2022-08-19T23:21:54.708265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#using SimpleImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nmedian_columns_list = ['Age', 'Fare']\nmode_columns_list = ['Embarked']\nct = ColumnTransformer(\n    [(\"median_imp\", SimpleImputer(strategy = 'median'), median_columns_list),\n     (\"mode_imp\", SimpleImputer(strategy = 'most_frequent'), mode_columns_list)])\ndata1 = data_raw.copy(deep = True)\ndata_cleaner = [data1, data_val]\nfor dataset in data_cleaner:\n    ###COMPLETING: complete or delete missing values in train and test/validation dataset\n    dataset[median_columns_list+mode_columns_list] = pd.DataFrame(ct.fit_transform(dataset), index=dataset.index, columns=median_columns_list+mode_columns_list)\n\n\nprint(data1.isnull().sum())\n","metadata":{"execution":{"iopub.status.busy":"2022-08-19T23:30:21.592809Z","iopub.execute_input":"2022-08-19T23:30:21.593261Z","iopub.status.idle":"2022-08-19T23:30:21.628110Z","shell.execute_reply.started":"2022-08-19T23:30:21.593202Z","shell.execute_reply":"2022-08-19T23:30:21.626872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nfor dataset in data_cleaner:    \n    #Discrete variables\n    dataset['FamilySize'] = dataset ['SibSp'] + dataset['Parch'] + 1\n\n    dataset['IsAlone'] = 1 #initialize to yes/1 is alone\n    dataset['IsAlone'].loc[dataset['FamilySize'] > 1] = 0 # now update to no/0 if family size is greater than 1\n\n    #quick and dirty code split title from name: http://www.pythonforbeginners.com/dictionary/python-split\n    dataset['Title'] = dataset['Name'].str.split(\", \", expand=True)[1].str.split(\".\", expand=True)[0]\n\n\n    #Continuous variable bins; qcut vs cut: https://stackoverflow.com/questions/30211923/what-is-the-difference-between-pandas-qcut-and-pandas-cut\n    #Fare Bins/Buckets using qcut or frequency bins: https://pandas.pydata.org/pandas-docs/stable/generated/pandas.qcut.html\n    dataset['FareBin'] = pd.qcut(dataset['Fare'], 4)\n\n    #Age Bins/Buckets using cut or value bins: https://pandas.pydata.org/pandas-docs/stable/generated/pandas.cut.html\n    dataset['AgeBin'] = pd.cut(dataset['Age'].astype(int), 5)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T23:30:25.477894Z","iopub.execute_input":"2022-08-19T23:30:25.478381Z","iopub.status.idle":"2022-08-19T23:30:25.516738Z","shell.execute_reply.started":"2022-08-19T23:30:25.478338Z","shell.execute_reply":"2022-08-19T23:30:25.515669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data1.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T23:31:23.339834Z","iopub.execute_input":"2022-08-19T23:31:23.340308Z","iopub.status.idle":"2022-08-19T23:31:23.347212Z","shell.execute_reply.started":"2022-08-19T23:31:23.340266Z","shell.execute_reply":"2022-08-19T23:31:23.346009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Exploration of data using groupby","metadata":{}},{"cell_type":"code","source":"drop_column = ['PassengerId','Cabin', 'Ticket']\ndata1.drop(drop_column, axis=1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T23:21:56.072411Z","iopub.execute_input":"2022-08-19T23:21:56.073173Z","iopub.status.idle":"2022-08-19T23:21:56.081163Z","shell.execute_reply.started":"2022-08-19T23:21:56.073128Z","shell.execute_reply":"2022-08-19T23:21:56.080028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nTarget = ['Survived']\nfor x in data1:\n    if x != Target[0] and data1[x].dtype != 'float64' :\n        print('Survival Correlation by:', x)\n        print(data1[[x, Target[0]]].groupby(x, as_index=False).mean())\n        print('-'*10, '\\n')","metadata":{"execution":{"iopub.status.busy":"2022-08-19T23:21:56.082706Z","iopub.execute_input":"2022-08-19T23:21:56.083388Z","iopub.status.idle":"2022-08-19T23:21:56.152110Z","shell.execute_reply.started":"2022-08-19T23:21:56.083354Z","shell.execute_reply":"2022-08-19T23:21:56.150670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plotting and see the correlations","metadata":{}},{"cell_type":"code","source":"\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfig, saxis = plt.subplots(2, 3,figsize=(16,12))\n\nsns.barplot(x = 'Embarked', y = 'Survived', data=data1, ax = saxis[0,0])\nsns.barplot(x = 'Pclass', y = 'Survived', order=[1,2,3], data=data1, ax = saxis[0,1])\nsns.barplot(x = 'IsAlone', y = 'Survived', order=[1,0], data=data1, ax = saxis[0,2])\n\nsns.pointplot(x = 'FareBin', y = 'Survived',  data=data1, ax = saxis[1,0])\nsns.pointplot(x = 'AgeBin', y = 'Survived',  data=data1, ax = saxis[1,1])\nsns.pointplot(x = 'FamilySize', y = 'Survived', data=data1, ax = saxis[1,2])","metadata":{"execution":{"iopub.status.busy":"2022-08-19T23:32:04.037848Z","iopub.execute_input":"2022-08-19T23:32:04.038305Z","iopub.status.idle":"2022-08-19T23:32:05.389786Z","shell.execute_reply.started":"2022-08-19T23:32:04.038266Z","shell.execute_reply":"2022-08-19T23:32:05.388610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#graph distribution of qualitative data: Sex\n#we know sex mattered in survival, now let's compare sex and a 2nd feature\nfig, qaxis = plt.subplots(1,3,figsize=(14,12))\n\nsns.barplot(x = 'Sex', y = 'Survived', hue = 'Embarked', data=data1, ax = qaxis[0])\nqaxis[0].set_title('Sex vs Embarked Survival Comparison')\n\nsns.barplot(x = 'Sex', y = 'Survived', hue = 'Pclass', data=data1, ax  = qaxis[1])\nqaxis[1].set_title('Sex vs Pclass Survival Comparison')\n\nsns.barplot(x = 'Sex', y = 'Survived', hue = 'IsAlone', data=data1, ax  = qaxis[2])\nqaxis[2].set_title('Sex vs IsAlone Survival Comparison')","metadata":{"execution":{"iopub.status.busy":"2022-08-19T23:39:07.945452Z","iopub.execute_input":"2022-08-19T23:39:07.945920Z","iopub.status.idle":"2022-08-19T23:39:08.913110Z","shell.execute_reply.started":"2022-08-19T23:39:07.945882Z","shell.execute_reply":"2022-08-19T23:39:08.911624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#more side-by-side comparisons\nfig, (maxis1, maxis2) = plt.subplots(1, 2,figsize=(14,12))\n\n#how does family size factor with sex & survival compare\nsns.pointplot(x=\"FamilySize\", y=\"Survived\", hue=\"Sex\", data=data1,\n              palette={\"male\": \"blue\", \"female\": \"pink\"},\n              markers=[\"*\", \"o\"], linestyles=[\"-\", \"--\"], ax = maxis1)\n\n#how does class factor with sex & survival compare\nsns.pointplot(x=\"Pclass\", y=\"Survived\", hue=\"Sex\", data=data1,\n              palette={\"male\": \"blue\", \"female\": \"pink\"},\n              markers=[\"*\", \"o\"], linestyles=[\"-\", \"--\"], ax = maxis2)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T23:40:26.658404Z","iopub.execute_input":"2022-08-19T23:40:26.659800Z","iopub.status.idle":"2022-08-19T23:40:27.759506Z","shell.execute_reply.started":"2022-08-19T23:40:26.659747Z","shell.execute_reply":"2022-08-19T23:40:27.758413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#how does embark port factor with class, sex, and survival compare\n#facetgrid: https://seaborn.pydata.org/generated/seaborn.FacetGrid.html\ne = sns.FacetGrid(data1, col = 'Embarked')\ne.map(sns.pointplot, 'Pclass', 'Survived', 'Sex', ci=95.0, palette = 'deep')\ne.add_legend()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-19T23:40:49.124424Z","iopub.execute_input":"2022-08-19T23:40:49.125211Z","iopub.status.idle":"2022-08-19T23:40:50.202163Z","shell.execute_reply.started":"2022-08-19T23:40:49.125170Z","shell.execute_reply":"2022-08-19T23:40:50.200985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#correlation heatmap of dataset\ndef correlation_heatmap(df):\n    _ , ax = plt.subplots(figsize =(14, 12))\n    colormap = sns.diverging_palette(220, 10, as_cmap = True)\n    \n    _ = sns.heatmap(\n        df.corr(), \n        cmap = colormap,\n        square=True, \n        cbar_kws={'shrink':.9 }, \n        ax=ax,\n        annot=True, \n        linewidths=0.1,vmax=1.0, linecolor='white',\n        annot_kws={'fontsize':12 }\n    )\n    \n    plt.title('Pearson Correlation of Features', y=1.05, size=15)\n\ncorrelation_heatmap(data1)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T23:41:49.998492Z","iopub.execute_input":"2022-08-19T23:41:49.999005Z","iopub.status.idle":"2022-08-19T23:41:50.541830Z","shell.execute_reply.started":"2022-08-19T23:41:49.998965Z","shell.execute_reply":"2022-08-19T23:41:50.540697Z"},"trusted":true},"execution_count":null,"outputs":[]}]}