{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T20:45:31.371875Z","iopub.execute_input":"2022-07-12T20:45:31.372389Z","iopub.status.idle":"2022-07-12T20:45:31.382958Z","shell.execute_reply.started":"2022-07-12T20:45:31.372353Z","shell.execute_reply":"2022-07-12T20:45:31.381827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's remove the visualization limitation with these codes:\npd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', None)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport joblib\nimport sys\n%matplotlib inline\nfrom mlxtend.feature_selection import ExhaustiveFeatureSelector as EFS","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:31.514582Z","iopub.execute_input":"2022-07-12T20:45:31.515048Z","iopub.status.idle":"2022-07-12T20:45:31.524237Z","shell.execute_reply.started":"2022-07-12T20:45:31.515012Z","shell.execute_reply":"2022-07-12T20:45:31.523400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading the train data\ndf_train = pd.read_csv('/kaggle/input/titanic/train.csv')\ndf_test = pd.read_csv('/kaggle/input/titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:31.653070Z","iopub.execute_input":"2022-07-12T20:45:31.653830Z","iopub.status.idle":"2022-07-12T20:45:31.671789Z","shell.execute_reply.started":"2022-07-12T20:45:31.653782Z","shell.execute_reply":"2022-07-12T20:45:31.670398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saying hello to the df\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:31.789039Z","iopub.execute_input":"2022-07-12T20:45:31.789527Z","iopub.status.idle":"2022-07-12T20:45:31.807149Z","shell.execute_reply.started":"2022-07-12T20:45:31.789487Z","shell.execute_reply":"2022-07-12T20:45:31.805958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Exploring it ...\ndf_train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:31.920048Z","iopub.execute_input":"2022-07-12T20:45:31.920547Z","iopub.status.idle":"2022-07-12T20:45:31.955193Z","shell.execute_reply.started":"2022-07-12T20:45:31.920506Z","shell.execute_reply":"2022-07-12T20:45:31.954046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a function to transform the male and female to 0,1\ndef survived(s):\n    if s == 'male':\n        sex = 0\n    elif s == 'female':\n        sex = 1\n    return sex","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:32.044921Z","iopub.execute_input":"2022-07-12T20:45:32.045383Z","iopub.status.idle":"2022-07-12T20:45:32.051062Z","shell.execute_reply.started":"2022-07-12T20:45:32.045347Z","shell.execute_reply":"2022-07-12T20:45:32.049858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Applying in the dataset\ndf_train['Sex'] = df_train['Sex'].apply(survived)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:32.167169Z","iopub.execute_input":"2022-07-12T20:45:32.167950Z","iopub.status.idle":"2022-07-12T20:45:32.174471Z","shell.execute_reply.started":"2022-07-12T20:45:32.167901Z","shell.execute_reply":"2022-07-12T20:45:32.173586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:32.297926Z","iopub.execute_input":"2022-07-12T20:45:32.298662Z","iopub.status.idle":"2022-07-12T20:45:32.315836Z","shell.execute_reply.started":"2022-07-12T20:45:32.298622Z","shell.execute_reply":"2022-07-12T20:45:32.314955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now I'll investigate the missing values\nmissing_values_percent = df_train.isnull().sum() / len(df_train['Survived']) * 100\nmissing_values_percent\n# At first, I'll analyse the complete features and after handle with missing data","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:32.425666Z","iopub.execute_input":"2022-07-12T20:45:32.426722Z","iopub.status.idle":"2022-07-12T20:45:32.439227Z","shell.execute_reply.started":"2022-07-12T20:45:32.426667Z","shell.execute_reply":"2022-07-12T20:45:32.437966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Droping the feature \"Cabin\" because I think it won't help me for now.\ndf_train.drop('Cabin', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:32.551783Z","iopub.execute_input":"2022-07-12T20:45:32.552981Z","iopub.status.idle":"2022-07-12T20:45:32.560077Z","shell.execute_reply.started":"2022-07-12T20:45:32.552932Z","shell.execute_reply":"2022-07-12T20:45:32.558790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.drop('Ticket', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:32.677998Z","iopub.execute_input":"2022-07-12T20:45:32.679205Z","iopub.status.idle":"2022-07-12T20:45:32.685199Z","shell.execute_reply.started":"2022-07-12T20:45:32.679158Z","shell.execute_reply":"2022-07-12T20:45:32.684307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I'll put intervals in age to observate the relation with the variables\nage_ranges = [0, 6, 12, 18, 50, 100]\ndf_train['age_range'] = pd.cut(df_train['Age'], age_ranges)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:32.800006Z","iopub.execute_input":"2022-07-12T20:45:32.800835Z","iopub.status.idle":"2022-07-12T20:45:32.809599Z","shell.execute_reply.started":"2022-07-12T20:45:32.800796Z","shell.execute_reply":"2022-07-12T20:45:32.808364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:32.981850Z","iopub.execute_input":"2022-07-12T20:45:32.982355Z","iopub.status.idle":"2022-07-12T20:45:33.002142Z","shell.execute_reply.started":"2022-07-12T20:45:32.982316Z","shell.execute_reply":"2022-07-12T20:45:33.000962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"___","metadata":{}},{"cell_type":"markdown","source":"## Missing Values","metadata":{}},{"cell_type":"markdown","source":"#### 'Age' is 19.9% missing data.\n##### I won't just replace with the mean or median because I think that I can investigate and replace with a better approach.\n    \n#### 'Cabin' 77% - removed\n\n","metadata":{}},{"cell_type":"code","source":"def titanic_analysis(df, feature, count = True):\n    print(f'Value Counts \\n{df[feature].value_counts()}')\n    print(f'Null Values >>> {df[feature].isnull().sum()}')\n    plt.figure(figsize=(5,4))\n    if count == True:\n        sns.countplot(x=df[feature], hue=df['Survived'])\n    else:\n        sns.displot(df[feature], kde = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:33.132151Z","iopub.execute_input":"2022-07-12T20:45:33.132604Z","iopub.status.idle":"2022-07-12T20:45:33.140041Z","shell.execute_reply.started":"2022-07-12T20:45:33.132566Z","shell.execute_reply":"2022-07-12T20:45:33.138799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"___","metadata":{}},{"cell_type":"markdown","source":"# Title","metadata":{}},{"cell_type":"code","source":"# Now, it's necessary to separate de \"Title\" from the Feature \"Name\" .\n# It will help me to fill in the age of the people with the smallest error,\n#   because I don'find feature with a decent correlation with \"age\" (previous code)\ndf_train['Title'] = df_train['Name'].apply(lambda name: name.split(',')[1].split('.')[0].strip())\ndf_train['Title'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:33.259838Z","iopub.execute_input":"2022-07-12T20:45:33.261205Z","iopub.status.idle":"2022-07-12T20:45:33.273390Z","shell.execute_reply.started":"2022-07-12T20:45:33.261159Z","shell.execute_reply":"2022-07-12T20:45:33.272320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The values that not are inserted in Mr, Miss, Mrs and Master, will be replaced in one value (Others)\ndf_train['Title'] = [n if n in ['Mr', 'Miss', 'Mrs', 'Master'] else 'Others' for n in df_train['Title']]\ndf_train['Title'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:33.383758Z","iopub.execute_input":"2022-07-12T20:45:33.385019Z","iopub.status.idle":"2022-07-12T20:45:33.398493Z","shell.execute_reply.started":"2022-07-12T20:45:33.384958Z","shell.execute_reply":"2022-07-12T20:45:33.397488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_analysis(df_train, 'Title')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:33.515013Z","iopub.execute_input":"2022-07-12T20:45:33.515829Z","iopub.status.idle":"2022-07-12T20:45:33.729063Z","shell.execute_reply.started":"2022-07-12T20:45:33.515788Z","shell.execute_reply":"2022-07-12T20:45:33.727856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PClass","metadata":{}},{"cell_type":"code","source":"# Analyzing the feature\ntitanic_analysis(df_train, 'Pclass')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:33.730740Z","iopub.execute_input":"2022-07-12T20:45:33.731113Z","iopub.status.idle":"2022-07-12T20:45:33.931154Z","shell.execute_reply.started":"2022-07-12T20:45:33.731078Z","shell.execute_reply":"2022-07-12T20:45:33.929824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sex","metadata":{}},{"cell_type":"code","source":"# Analyzing the feature\ntitanic_analysis(df_train, 'Sex')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:33.933856Z","iopub.execute_input":"2022-07-12T20:45:33.934623Z","iopub.status.idle":"2022-07-12T20:45:34.110782Z","shell.execute_reply.started":"2022-07-12T20:45:33.934573Z","shell.execute_reply":"2022-07-12T20:45:34.109623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Age","metadata":{}},{"cell_type":"code","source":"# Analyzing the feature\ntitanic_analysis(df_train, 'Age', False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:34.111999Z","iopub.execute_input":"2022-07-12T20:45:34.112315Z","iopub.status.idle":"2022-07-12T20:45:34.436910Z","shell.execute_reply.started":"2022-07-12T20:45:34.112284Z","shell.execute_reply":"2022-07-12T20:45:34.435595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Here I observed a crucial point.\n##### There are a lot of missing values.\n##### First, I'll create a new dataset just containing complete data.","metadata":{}},{"cell_type":"code","source":"df_train[df_train['Age'].notnull()].head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:34.439939Z","iopub.execute_input":"2022-07-12T20:45:34.440397Z","iopub.status.idle":"2022-07-12T20:45:34.462267Z","shell.execute_reply.started":"2022-07-12T20:45:34.440352Z","shell.execute_reply":"2022-07-12T20:45:34.461356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I'll use this values to complete the NaN data\ndf_train_age_notnull = df_train[df_train['Age'].notnull()]\n\ndf_ft_notnulls = df_train_age_notnull[['Title', 'Age']].groupby(['Title']).agg(['count', 'mean', 'std'])\ndisplay(df_ft_notnulls.index)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:34.463683Z","iopub.execute_input":"2022-07-12T20:45:34.464066Z","iopub.status.idle":"2022-07-12T20:45:34.478561Z","shell.execute_reply.started":"2022-07-12T20:45:34.464029Z","shell.execute_reply":"2022-07-12T20:45:34.477785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I'm sure must be a better to fill it\n# Can you teach me? hehehe\nfor i in df_train.index:\n    if pd.isnull(df_train['Age'][i]):\n        if df_train['Title'][i] == 'Master':\n            df_train['Age'][i] = 4.57\n        elif df_train['Title'][i] == 'Miss':\n            df_train['Age'][i] = 21.77\n        elif df_train['Title'][i] == 'Mr':\n            df_train['Age'][i] = 32.36\n        elif df_train['Title'][i] == 'Mrs':\n            df_train['Age'][i] = 35.89\n        elif df_train['Title'][i] == 'Others':\n            df_train['Age'][i] = 42.38\n    else:\n        continue\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:34.480000Z","iopub.execute_input":"2022-07-12T20:45:34.481067Z","iopub.status.idle":"2022-07-12T20:45:34.517822Z","shell.execute_reply.started":"2022-07-12T20:45:34.481020Z","shell.execute_reply":"2022-07-12T20:45:34.516586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SibSp","metadata":{}},{"cell_type":"code","source":"titanic_analysis(df_train, 'SibSp')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:34.519465Z","iopub.execute_input":"2022-07-12T20:45:34.520167Z","iopub.status.idle":"2022-07-12T20:45:34.761472Z","shell.execute_reply.started":"2022-07-12T20:45:34.520122Z","shell.execute_reply":"2022-07-12T20:45:34.760397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Parch","metadata":{}},{"cell_type":"code","source":"titanic_analysis(df_train, 'Parch')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:34.763805Z","iopub.execute_input":"2022-07-12T20:45:34.764251Z","iopub.status.idle":"2022-07-12T20:45:35.317451Z","shell.execute_reply.started":"2022-07-12T20:45:34.764214Z","shell.execute_reply":"2022-07-12T20:45:35.316387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fare","metadata":{}},{"cell_type":"code","source":"titanic_analysis(df_train, 'Fare', False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:35.318992Z","iopub.execute_input":"2022-07-12T20:45:35.319278Z","iopub.status.idle":"2022-07-12T20:45:35.796550Z","shell.execute_reply.started":"2022-07-12T20:45:35.319250Z","shell.execute_reply":"2022-07-12T20:45:35.795366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Embarked","metadata":{}},{"cell_type":"code","source":"titanic_analysis(df_train, 'Embarked')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:35.801112Z","iopub.execute_input":"2022-07-12T20:45:35.801460Z","iopub.status.idle":"2022-07-12T20:45:36.005483Z","shell.execute_reply.started":"2022-07-12T20:45:35.801429Z","shell.execute_reply":"2022-07-12T20:45:36.004148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['Embarked'].fillna('C', inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:36.007338Z","iopub.execute_input":"2022-07-12T20:45:36.007767Z","iopub.status.idle":"2022-07-12T20:45:36.013593Z","shell.execute_reply.started":"2022-07-12T20:45:36.007711Z","shell.execute_reply":"2022-07-12T20:45:36.012763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_values_percent","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:36.014916Z","iopub.execute_input":"2022-07-12T20:45:36.015237Z","iopub.status.idle":"2022-07-12T20:45:36.030575Z","shell.execute_reply.started":"2022-07-12T20:45:36.015207Z","shell.execute_reply":"2022-07-12T20:45:36.029431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Age Range","metadata":{}},{"cell_type":"code","source":"titanic_analysis(df_train, 'age_range')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:36.032387Z","iopub.execute_input":"2022-07-12T20:45:36.032995Z","iopub.status.idle":"2022-07-12T20:45:36.245951Z","shell.execute_reply.started":"2022-07-12T20:45:36.032961Z","shell.execute_reply":"2022-07-12T20:45:36.244952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:36.247107Z","iopub.execute_input":"2022-07-12T20:45:36.247411Z","iopub.status.idle":"2022-07-12T20:45:36.263981Z","shell.execute_reply.started":"2022-07-12T20:45:36.247384Z","shell.execute_reply":"2022-07-12T20:45:36.262814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['Sex'] = df_test['Sex'].apply(survived)\ndf_test.drop('Cabin', axis=1, inplace=True)\ndf_test.drop('Ticket', axis=1, inplace=True)\ndf_test['age_range'] = pd.cut(df_test['Age'], age_ranges)\ndf_test['Title'] = df_test['Name'].apply(lambda name: name.split(',')[1].split('.')[0].strip())\ndf_test['Title'] = [n if n in ['Mr', 'Miss', 'Mrs', 'Master'] else 'Others' for n in df_test['Title']]\nfor i in df_test.index:\n    if pd.isnull(df_test['Age'][i]):\n        if df_test['Title'][i] == 'Master':\n            df_test['Age'][i] = 4.57\n        elif df_test['Title'][i] == 'Miss':\n            df_test['Age'][i] = 21.77\n        elif df_test['Title'][i] == 'Mr':\n            df_test['Age'][i] = 32.36\n        elif df_test['Title'][i] == 'Mrs':\n            df_test['Age'][i] = 35.89\n        elif df_test['Title'][i] == 'Others':\n            df_test['Age'][i] = 42.38\n    else:\n        continue","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:36.268000Z","iopub.execute_input":"2022-07-12T20:45:36.268762Z","iopub.status.idle":"2022-07-12T20:45:36.305607Z","shell.execute_reply.started":"2022-07-12T20:45:36.268715Z","shell.execute_reply":"2022-07-12T20:45:36.304617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['age_range'] = pd.cut(df_test['Age'], age_ranges)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:36.307170Z","iopub.execute_input":"2022-07-12T20:45:36.308432Z","iopub.status.idle":"2022-07-12T20:45:36.317101Z","shell.execute_reply.started":"2022-07-12T20:45:36.308385Z","shell.execute_reply":"2022-07-12T20:45:36.315958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test[df_test['age_range'].isnull()]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:36.318196Z","iopub.execute_input":"2022-07-12T20:45:36.319009Z","iopub.status.idle":"2022-07-12T20:45:36.338099Z","shell.execute_reply.started":"2022-07-12T20:45:36.318973Z","shell.execute_reply":"2022-07-12T20:45:36.336948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:36.341271Z","iopub.execute_input":"2022-07-12T20:45:36.342019Z","iopub.status.idle":"2022-07-12T20:45:36.355125Z","shell.execute_reply.started":"2022-07-12T20:45:36.341985Z","shell.execute_reply":"2022-07-12T20:45:36.353928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['Fare'].fillna(0, inplace=True)\ndf_train['age_range'] = pd.cut(df_train['Age'], age_ranges)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:36.418576Z","iopub.execute_input":"2022-07-12T20:45:36.419002Z","iopub.status.idle":"2022-07-12T20:45:36.427392Z","shell.execute_reply.started":"2022-07-12T20:45:36.418967Z","shell.execute_reply":"2022-07-12T20:45:36.426488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Age_range have some nan values.\ndisplay(df_train.info())\ndisplay(df_test.info())","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:36.599480Z","iopub.execute_input":"2022-07-12T20:45:36.600541Z","iopub.status.idle":"2022-07-12T20:45:36.624589Z","shell.execute_reply.started":"2022-07-12T20:45:36.600503Z","shell.execute_reply":"2022-07-12T20:45:36.623481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encode_pclass = pd.get_dummies(df_train['Pclass'])\nencode_embarked = pd.get_dummies(df_train['Embarked'])\nencode_title = pd.get_dummies(df_train['Title'])\ndf_train_concat = pd.concat([df_train, encode_pclass, encode_embarked, encode_title], axis=1)\ndf_train_concat.drop(['Pclass', 'Embarked', 'Title', 'Name', 'age_range', 'PassengerId'], axis=1, inplace=True)\ndf_train_concat.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:36.872122Z","iopub.execute_input":"2022-07-12T20:45:36.872799Z","iopub.status.idle":"2022-07-12T20:45:36.898287Z","shell.execute_reply.started":"2022-07-12T20:45:36.872761Z","shell.execute_reply":"2022-07-12T20:45:36.897489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encode_pclass = pd.get_dummies(df_test['Pclass'])\nencode_embarked = pd.get_dummies(df_test['Embarked'])\nencode_title = pd.get_dummies(df_test['Title'])\ndf_test_concat = pd.concat([df_test, encode_pclass, encode_embarked, encode_title], axis=1)\ndf_test_concat.drop(['Pclass', 'Embarked', 'Title', 'Name', 'age_range', 'PassengerId'], axis=1, inplace=True)\ndf_test_concat.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:37.263131Z","iopub.execute_input":"2022-07-12T20:45:37.263809Z","iopub.status.idle":"2022-07-12T20:45:37.288549Z","shell.execute_reply.started":"2022-07-12T20:45:37.263771Z","shell.execute_reply":"2022-07-12T20:45:37.287671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16,6))\nsns.heatmap(df_train_concat.corr(), annot=True, cmap='BrBG')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:37.527307Z","iopub.execute_input":"2022-07-12T20:45:37.528072Z","iopub.status.idle":"2022-07-12T20:45:38.937451Z","shell.execute_reply.started":"2022-07-12T20:45:37.528033Z","shell.execute_reply":"2022-07-12T20:45:38.936054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, MinMaxScaler\nnorm = MinMaxScaler(feature_range=(-1,1))\nx = df_train_concat.drop(['Survived', 'Fare'], axis=1)\ny = df_train_concat['Survived']\n\nx_norm = norm.fit_transform(x)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:38.939509Z","iopub.execute_input":"2022-07-12T20:45:38.939865Z","iopub.status.idle":"2022-07-12T20:45:38.954595Z","shell.execute_reply.started":"2022-07-12T20:45:38.939830Z","shell.execute_reply":"2022-07-12T20:45:38.953359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.linear_model import LogisticRegression\n\nmodel = LogisticRegression()\nskfold = StratifiedKFold(n_splits=3)\nresult = cross_val_score(model, x_norm, y, cv=skfold, scoring='accuracy')\n\nprint(f'{result} --> {result.mean():.2f}')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:38.956111Z","iopub.execute_input":"2022-07-12T20:45:38.957180Z","iopub.status.idle":"2022-07-12T20:45:39.016248Z","shell.execute_reply.started":"2022-07-12T20:45:38.957127Z","shell.execute_reply":"2022-07-12T20:45:39.014814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def model_acc(a, b):\n    '''Return the accuracy from some classification models'''\n    \n    from sklearn.linear_model import LogisticRegression\n    from sklearn.naive_bayes import GaussianNB\n    from sklearn.tree import DecisionTreeClassifier\n    from sklearn.ensemble import RandomForestClassifier, ExtraTreesClassifier, AdaBoostClassifier, GradientBoostingClassifier\n    from sklearn.svm import SVC\n        \n    a = x_norm\n    b = y\n    \n    skfold = StratifiedKFold(n_splits=3)\n    \n    log_reg = LogisticRegression()\n    naive_bayes = GaussianNB()\n    dec_tree = DecisionTreeClassifier(min_samples_split=6)\n    random_forest = RandomForestClassifier(min_samples_split=6, n_estimators=50)\n    extra_trees = ExtraTreesClassifier(min_samples_split=6)\n    ada_boost = AdaBoostClassifier()\n    gradient_boost = GradientBoostingClassifier()\n    svm = SVC()\n    \n    result_log_reg = cross_val_score(log_reg, x_norm, y, cv=skfold, scoring='accuracy')\n    result_naive_bayes = cross_val_score(naive_bayes, x_norm, y, cv=skfold, scoring='accuracy')\n    result_dec_tree = cross_val_score(dec_tree, x_norm, y, cv=skfold, scoring='accuracy')\n    result_random_forest = cross_val_score(random_forest, x_norm, y, cv=skfold, scoring='accuracy')\n    result_extra_trees = cross_val_score(extra_trees, x_norm, y, cv=skfold, scoring='accuracy')\n    result_ada_boost = cross_val_score(ada_boost, x_norm, y, cv=skfold, scoring='accuracy')\n    result_gradient_boost = cross_val_score(gradient_boost, x_norm, y, cv=skfold, scoring='accuracy')\n    result_svm = cross_val_score(svm, x_norm, y, cv=skfold, scoring='accuracy')\n    \n    print('Model --> Mean, Stdev')\n    print(f'Logistic Regression: {result_log_reg} --> {result_log_reg.mean():.3f}, {result_log_reg.std():.2f}')\n    print(f'Naive Bayes: {result_naive_bayes} --> {result_naive_bayes.mean():.3f}, {result_naive_bayes.std():.2f}')\n    print(f'Decision Tree: {result_dec_tree} --> {result_dec_tree.mean():.3f}, {result_dec_tree.std():.2f}')\n    print(f'Random Forest: {result_random_forest} --> {result_random_forest.mean():.3f}, {result_random_forest.std():.2f}')\n    print(f'Extra Trees: {result_extra_trees} --> {result_extra_trees.mean():.3f}, {result_extra_trees.std():.2f}')\n    print(f'Ada Boost: {result_ada_boost} --> {result_ada_boost.mean():.3f}, {result_ada_boost.std():.2f}')\n    print(f'Gradient Boosting: {result_gradient_boost} --> {result_gradient_boost.mean():.3f}, {result_gradient_boost.std():.2f}')\n    print(f'SVM: {result_svm} --> {result_svm.mean():.3f}, {result_svm.std():.2f}')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:39.020320Z","iopub.execute_input":"2022-07-12T20:45:39.020651Z","iopub.status.idle":"2022-07-12T20:45:39.036434Z","shell.execute_reply.started":"2022-07-12T20:45:39.020614Z","shell.execute_reply":"2022-07-12T20:45:39.034807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_acc(x_norm, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:39.037931Z","iopub.execute_input":"2022-07-12T20:45:39.038371Z","iopub.status.idle":"2022-07-12T20:45:40.598393Z","shell.execute_reply.started":"2022-07-12T20:45:39.038335Z","shell.execute_reply":"2022-07-12T20:45:40.596898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_acc(x_norm, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:40.601169Z","iopub.execute_input":"2022-07-12T20:45:40.601566Z","iopub.status.idle":"2022-07-12T20:45:42.169640Z","shell.execute_reply.started":"2022-07-12T20:45:40.601534Z","shell.execute_reply":"2022-07-12T20:45:42.168378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from sklearn.svm import SVC\n#svc = SVC()\n#efs = EFS(estimator=svc, n_jobs=-1, cv=3, max_features=6)\n#efs = efs.fit(x, y)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:42.171197Z","iopub.execute_input":"2022-07-12T20:45:42.171499Z","iopub.status.idle":"2022-07-12T20:45:42.175496Z","shell.execute_reply.started":"2022-07-12T20:45:42.171469Z","shell.execute_reply":"2022-07-12T20:45:42.174465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#efs.best_idx_","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:42.177327Z","iopub.execute_input":"2022-07-12T20:45:42.177764Z","iopub.status.idle":"2022-07-12T20:45:42.187046Z","shell.execute_reply.started":"2022-07-12T20:45:42.177720Z","shell.execute_reply":"2022-07-12T20:45:42.185936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#efs.best_score_","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:42.188499Z","iopub.execute_input":"2022-07-12T20:45:42.189100Z","iopub.status.idle":"2022-07-12T20:45:42.197813Z","shell.execute_reply.started":"2022-07-12T20:45:42.189063Z","shell.execute_reply":"2022-07-12T20:45:42.196955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df_visualization_features = pd.DataFrame.from_dict(efs.get_metric_dict()).T\n#df_visualization_features.sort_values('avg_score', inplace=True, ascending=False)\n#df_visualization_features.head(30)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:42.201038Z","iopub.execute_input":"2022-07-12T20:45:42.203005Z","iopub.status.idle":"2022-07-12T20:45:42.208583Z","shell.execute_reply.started":"2022-07-12T20:45:42.202973Z","shell.execute_reply":"2022-07-12T20:45:42.207445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import KFold\nfrom sklearn.svm import SVC\n\nc = np.array([1.0, 0.95, 1.05, 1.10, 1.20, 2, 0.9, 0.8])\nkernel = ['linear', 'poly', 'rbf', 'sigmoid']\npolynom = np.array([3, 4, 5])\ngamma = ['scale', 'auto']\ncoef = np.array([2, 3, 5])\nvalues_grid = {'C':c,\n              'kernel': kernel,\n              'degree': polynom,\n              'gamma': gamma,\n              'coef0': coef}\n\nmodel = SVC()\n\nkfold = KFold(n_splits=5, shuffle=True)\ngridSVC = GridSearchCV(estimator=model, param_grid=values_grid, cv=kfold, n_jobs=-1)\ngridSVC.fit(x_norm, y)\n\nprint('Best Constant: ', gridSVC.best_estimator_.C)\nprint('Best Kernel: ', gridSVC.best_estimator_.kernel)\nprint('Best polynom: ', gridSVC.best_estimator_.degree)\nprint('Best gamma: ', gridSVC.best_estimator_.gamma)\nprint('Best coef0: ', gridSVC.best_estimator_.coef0)\nprint('Best Accuracy: ', gridSVC.best_score_)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:45:42.211747Z","iopub.execute_input":"2022-07-12T20:45:42.212377Z","iopub.status.idle":"2022-07-12T20:47:04.983943Z","shell.execute_reply.started":"2022-07-12T20:45:42.212332Z","shell.execute_reply":"2022-07-12T20:47:04.982537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_concat.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:47:04.985771Z","iopub.execute_input":"2022-07-12T20:47:04.986261Z","iopub.status.idle":"2022-07-12T20:47:05.007032Z","shell.execute_reply.started":"2022-07-12T20:47:04.986214Z","shell.execute_reply":"2022-07-12T20:47:05.005923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_concat.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:47:05.008321Z","iopub.execute_input":"2022-07-12T20:47:05.009026Z","iopub.status.idle":"2022-07-12T20:47:05.027976Z","shell.execute_reply.started":"2022-07-12T20:47:05.008992Z","shell.execute_reply":"2022-07-12T20:47:05.027169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = df_test_concat.drop('Fare', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:47:05.028867Z","iopub.execute_input":"2022-07-12T20:47:05.029191Z","iopub.status.idle":"2022-07-12T20:47:05.041476Z","shell.execute_reply.started":"2022-07-12T20:47:05.029161Z","shell.execute_reply":"2022-07-12T20:47:05.040453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = SVC(kernel='poly', degree=4, gamma='auto', coef0=2)\n\nmodel.fit(x_norm, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:47:05.042643Z","iopub.execute_input":"2022-07-12T20:47:05.042972Z","iopub.status.idle":"2022-07-12T20:47:05.120296Z","shell.execute_reply.started":"2022-07-12T20:47:05.042935Z","shell.execute_reply":"2022-07-12T20:47:05.119213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:47:05.121600Z","iopub.execute_input":"2022-07-12T20:47:05.122041Z","iopub.status.idle":"2022-07-12T20:47:05.135101Z","shell.execute_reply.started":"2022-07-12T20:47:05.122010Z","shell.execute_reply":"2022-07-12T20:47:05.134283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(norm.fit_transform(x_test))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:52:14.320913Z","iopub.execute_input":"2022-07-12T20:52:14.321416Z","iopub.status.idle":"2022-07-12T20:52:14.341640Z","shell.execute_reply.started":"2022-07-12T20:52:14.321369Z","shell.execute_reply":"2022-07-12T20:52:14.340220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/titanic/gender_submission.csv')\nsubmission['Survived'] = y_pred\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T20:54:23.688728Z","iopub.execute_input":"2022-07-12T20:54:23.689707Z","iopub.status.idle":"2022-07-12T20:54:23.708457Z","shell.execute_reply.started":"2022-07-12T20:54:23.689649Z","shell.execute_reply":"2022-07-12T20:54:23.707565Z"},"trusted":true},"execution_count":null,"outputs":[]}]}