{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import classification_report\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import cross_val_score\nimport matplotlib.pyplot as plt\nimport string\n!pip install catboost\nfrom catboost import CatBoostClassifier\n!pip install missingno\nimport missingno as msno\nfrom datetime import date\n\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.neighbors import LocalOutlierFactor\nfrom sklearn.preprocessing import MinMaxScaler, LabelEncoder, StandardScaler, RobustScaler\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-03T15:08:55.066292Z","iopub.execute_input":"2022-08-03T15:08:55.066725Z","iopub.status.idle":"2022-08-03T15:09:16.550638Z","shell.execute_reply.started":"2022-08-03T15:08:55.066691Z","shell.execute_reply":"2022-08-03T15:09:16.549325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', None)\npd.set_option('display.float_format', lambda x: '%.3f' % x)\npd.set_option('display.width', 500)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:16.554326Z","iopub.execute_input":"2022-08-03T15:09:16.554751Z","iopub.status.idle":"2022-08-03T15:09:16.561327Z","shell.execute_reply.started":"2022-08-03T15:09:16.554713Z","shell.execute_reply":"2022-08-03T15:09:16.560171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train= pd.read_csv('../input/titanic/train.csv')\ntrain = df_train.copy()\ndf_test= pd.read_csv('../input/titanic/test.csv')\ntest = df_test.copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:16.563062Z","iopub.execute_input":"2022-08-03T15:09:16.563640Z","iopub.status.idle":"2022-08-03T15:09:16.590694Z","shell.execute_reply.started":"2022-08-03T15:09:16.563594Z","shell.execute_reply":"2022-08-03T15:09:16.589739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pclass: Class of the passenger\n# Sibsp: Any parents, siblings, etc.\n# Parch: Any distant relatives\n# Embarked: a place where the passenger go on board.","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:16.592527Z","iopub.execute_input":"2022-08-03T15:09:16.592865Z","iopub.status.idle":"2022-08-03T15:09:16.597301Z","shell.execute_reply.started":"2022-08-03T15:09:16.592834Z","shell.execute_reply":"2022-08-03T15:09:16.596477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def check_df(data, x=5):\n    print('################################# shape ##########################')\n    print(data.shape)\n    print('################################# type ##########################')\n    print(data.dtypes)\n    print('################################# head ##########################')\n    print(data.head(x))\n    print('################################# tail ##########################')\n    print(data.tail(x))\n    print('################################# null ##########################')\n    print(data.isnull().sum().sort_values(ascending=False))\n    print('################################# quantiles #####################')\n    print(data.describe([0,0.05, 0.5, 0.95, 0.99, 1 ]).T)\n    \ncheck_df(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:16.598642Z","iopub.execute_input":"2022-08-03T15:09:16.598991Z","iopub.status.idle":"2022-08-03T15:09:17.258145Z","shell.execute_reply.started":"2022-08-03T15:09:16.598958Z","shell.execute_reply":"2022-08-03T15:09:17.257119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\" \nFirst look to the train set:\n###########################################################\nCategorical Variables: Name, Sex, Ticket, Cabin, Embarked\nNumerical Variables : PassengerId, Survived (Target Var), Pclass, Age, SibSp, Parch, Fare.\nTarget Var: Survived.\n###########################################################\nIt seems that Cabin variable has NaN values the most. Age and Fare also has some NaN values.\n###########################################################\nDescriptive stats seems normal for now.\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.259636Z","iopub.execute_input":"2022-08-03T15:09:17.260047Z","iopub.status.idle":"2022-08-03T15:09:17.266883Z","shell.execute_reply.started":"2022-08-03T15:09:17.260013Z","shell.execute_reply":"2022-08-03T15:09:17.265892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def check_df(data, x=5):\n    print('################################# shape ##########################')\n    print(data.shape)\n    print('################################# type ##########################')\n    print(data.dtypes)\n    print('################################# head ##########################')\n    print(data.head(x))\n    print('################################# tail ##########################')\n    print(data.tail(x))\n    print('################################# null ##########################')\n    print(data.isnull().sum().sort_values(ascending=False))\n    print('################################# quantiles #####################')\n    print(data.describe([0,0.05, 0.5, 0.95, 0.99, 1 ]).T)\n    \ncheck_df(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.268493Z","iopub.execute_input":"2022-08-03T15:09:17.269405Z","iopub.status.idle":"2022-08-03T15:09:17.310302Z","shell.execute_reply.started":"2022-08-03T15:09:17.269357Z","shell.execute_reply":"2022-08-03T15:09:17.309250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"Test set follows the same pattern as train set in terms of descriptive stats, null values and the datatypes.\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.311749Z","iopub.execute_input":"2022-08-03T15:09:17.312078Z","iopub.status.idle":"2022-08-03T15:09:17.318765Z","shell.execute_reply.started":"2022-08-03T15:09:17.312047Z","shell.execute_reply":"2022-08-03T15:09:17.317523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#############################################################\n# Analysis for Categorical and Numerical Variables\n#############################################################\n\n# First we need to analyze the variables which seems numerical but actually a categorical variable. \n# The cardinal categoric variables which seems categoric variable. (cat_but_card : a categoric variable which has more than 20 categories. Measuring it will be a problem...)\n# Defining categorical variables list as cat_cols-cat_but_card+ num_but_cat\n# num_but_cat : if a numerical seemed variable has less than 10 unique values.\n\ndef grab_col_names(dataframe, cat_th= 10, car_th= 20):\n    cat_cols = [col for col in dataframe.columns if dataframe[col].dtype == 'O']\n    num_but_cat = [col for col in dataframe.columns if dataframe[col].dtypes != \"O\" and\n                    dataframe[col].nunique() < cat_th]\n    cat_but_card = [col for col in dataframe.columns if (dataframe[col].dtype == 'O') and\n                    (dataframe[col].nunique() > car_th)]\n    cat_cols = cat_cols+num_but_cat\n    cat_cols = [col for col in cat_cols if col not in cat_but_card]\n    \n    \n    num_cols = [col for col in dataframe.columns if dataframe[col].dtypes != \"O\"]\n    num_cols = [col for col in num_cols if col not in num_but_cat]            \n                \n    print(f\"Observations, {dataframe.shape[0]}\")\n    print(f\"Variables, {dataframe.shape[1]}\")\n    print(f'cat_cols, {len(cat_cols)}')\n    print(f'cat_but_card, {len(cat_but_card)}')\n    print(f'num_but_cat, {len(num_but_cat)}')\n    print(f'num_cols, {len(num_cols)}')\n                \n    return cat_cols, num_cols, cat_but_card","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.320577Z","iopub.execute_input":"2022-08-03T15:09:17.321267Z","iopub.status.idle":"2022-08-03T15:09:17.337448Z","shell.execute_reply.started":"2022-08-03T15:09:17.321209Z","shell.execute_reply":"2022-08-03T15:09:17.335172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grab_col_names(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.342716Z","iopub.execute_input":"2022-08-03T15:09:17.343779Z","iopub.status.idle":"2022-08-03T15:09:17.362461Z","shell.execute_reply.started":"2022-08-03T15:09:17.343723Z","shell.execute_reply":"2022-08-03T15:09:17.360972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#grab_col_names(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.363665Z","iopub.execute_input":"2022-08-03T15:09:17.364436Z","iopub.status.idle":"2022-08-03T15:09:17.369492Z","shell.execute_reply.started":"2022-08-03T15:09:17.364398Z","shell.execute_reply":"2022-08-03T15:09:17.368316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols, num_cols, cat_but_card = grab_col_names(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.371484Z","iopub.execute_input":"2022-08-03T15:09:17.372703Z","iopub.status.idle":"2022-08-03T15:09:17.392362Z","shell.execute_reply.started":"2022-08-03T15:09:17.372616Z","shell.execute_reply":"2022-08-03T15:09:17.388188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cols\nnum_cols = [col for col in num_cols if col not in 'PassengerId']\nnum_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.394452Z","iopub.execute_input":"2022-08-03T15:09:17.395460Z","iopub.status.idle":"2022-08-03T15:09:17.404791Z","shell.execute_reply.started":"2022-08-03T15:09:17.395409Z","shell.execute_reply":"2022-08-03T15:09:17.404045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's see some outliers of the numerical variables (num_cols). To do that you can either use a boxplot or a histogram.\nfor i in num_cols:\n    sns.boxplot(x = train[i])\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.405966Z","iopub.execute_input":"2022-08-03T15:09:17.406342Z","iopub.status.idle":"2022-08-03T15:09:17.761753Z","shell.execute_reply.started":"2022-08-03T15:09:17.406302Z","shell.execute_reply":"2022-08-03T15:09:17.760632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#########################################################################################\n# Checking for Outliers\n#########################################################################################\n\ndef outlier_thresholds(dataframe, col_name, q1 = 0.25, q3 = 0.75):\n    quartile1 = dataframe[col_name].quantile(q1)\n    quartile3 = dataframe[col_name].quantile(q3)\n    interquantile_range = quartile3 - quartile1\n    up_limit = quartile3 + 1.5 * interquantile_range\n    low_limit = quartile1 - 1.5 * interquantile_range\n    return low_limit, up_limit\n\noutlier_thresholds(train, 'Age')\noutlier_thresholds(train, 'Fare')\n\ndef check_outliers(dataframe, col_name):\n    low_limit, up_limit = outlier_thresholds(dataframe, col_name)\n    if dataframe[(dataframe[col_name] > up_limit) | (dataframe[col_name] < low_limit)].any(axis = None):\n        return print(f\"{col_name.upper()} feature CONTAINS at least one outlier\")\n    else:\n        return print(f\"{col_name.upper()} feature DOES NOT CONTAIN any outlier\")\n\n# Checking outliers for NUMERIC FEATURES.\nfor col in num_cols:\n    check_outliers(train,col)    \n\nfor col in num_cols:\n    check_outliers(test,col) \n#print(f\"Does Age feature have any outliers? , {(check_outliers(train, 'Age'))}\")\n#print(f\"Does Age feature have any outliers? , {(check_outliers(train, 'Fare'))}\")\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.763238Z","iopub.execute_input":"2022-08-03T15:09:17.763610Z","iopub.status.idle":"2022-08-03T15:09:17.795071Z","shell.execute_reply.started":"2022-08-03T15:09:17.763577Z","shell.execute_reply":"2022-08-03T15:09:17.794071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#########################################################################################\n# Accessing to the outliers\n#########################################################################################\ndef grab_outliers(dataframe, col_name, index = False):\n    low, up = outlier_thresholds(dataframe, col_name)\n    if dataframe[((dataframe[col_name] < low) | (dataframe[col_name] > up))].shape[0] > 10:\n        print(dataframe[((dataframe[col_name] < low) | (dataframe[col_name]> up))].head())\n    else:\n        print(dataframe[((dataframe(col_name)<low) | (dataframe[col_name] > up))])\n    if index:\n        outlier_index = dataframe[((dataframe[col_name]<low) | (dataframe[col_name] > up))].index\n        return outlier_index","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.796669Z","iopub.execute_input":"2022-08-03T15:09:17.797323Z","iopub.status.idle":"2022-08-03T15:09:17.806134Z","shell.execute_reply.started":"2022-08-03T15:09:17.797276Z","shell.execute_reply":"2022-08-03T15:09:17.805212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#grab_outliers(train, 'Age')\ngrab_outliers(train, 'Age', True)\nage_index = grab_outliers(train, 'Age', True)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.807697Z","iopub.execute_input":"2022-08-03T15:09:17.808360Z","iopub.status.idle":"2022-08-03T15:09:17.841770Z","shell.execute_reply.started":"2022-08-03T15:09:17.808314Z","shell.execute_reply":"2022-08-03T15:09:17.840895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"age_index","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.843044Z","iopub.execute_input":"2022-08-03T15:09:17.843610Z","iopub.status.idle":"2022-08-03T15:09:17.849236Z","shell.execute_reply.started":"2022-08-03T15:09:17.843573Z","shell.execute_reply":"2022-08-03T15:09:17.848325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#########################################################################################\n# Solving The Outlier Problem\n#########################################################################################\n\n## 1. Removing the outliers.\n\ndef remove_outlier(dataframe,col_name):\n    low_limit, up_limit = outlier_thresholds(dataframe, col_name)\n    df_without_outliers = dataframe[~((dataframe[col_name]<low_limit) | (dataframe[col_name]>up_limit))]\n    return df_without_outliers\n\ntrain.shape\n\nfor col in num_cols:\n    new_df = remove_outlier(train, col)\n\ntrain.shape[0]-new_df.shape[0] \n\n# Bir tane bir hucredeki verileri silme islemi yaptigimizda, diger tam olan gozlemdeki verilerdende oluyoruz. Baskilama da tercih edilebilir.\n\n# 2. Re-assigning the outlier values with thresholds ( BASKILAMA)\n\nlow, high = outlier_thresholds(train,'Fare')\ntrain[((train['Fare'] < low) | (train['Fare']> high))]['Fare']\n#train.loc[((train['Fare'] < low) | (train['Fare']> high)), 'Fare']\n\n#train.loc[(train['Fare'] > high), 'Fare']\n#train.loc[(train['Fare'] > high), 'Fare'] = high\n#train.loc[(train['Fare'] > high), 'Fare']\n#train.loc[(train['Fare'] < low), 'Fare'] \n\n# Now let's apply these steps in a programatic way by defining replace_with_thresholds function.\n\ndef replace_with_thresholds(dataframe, variable):\n    low_limit, up_limit = outlier_thresholds(dataframe, variable)\n    dataframe.loc[(dataframe[variable] < low_limit), variable] = low_limit\n    dataframe.loc[(dataframe[variable] > up_limit), variable] = up_limit\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.850703Z","iopub.execute_input":"2022-08-03T15:09:17.851398Z","iopub.status.idle":"2022-08-03T15:09:17.873842Z","shell.execute_reply.started":"2022-08-03T15:09:17.851349Z","shell.execute_reply":"2022-08-03T15:09:17.872899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in num_cols:\n    check_outliers(train,col)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.875175Z","iopub.execute_input":"2022-08-03T15:09:17.875982Z","iopub.status.idle":"2022-08-03T15:09:17.891161Z","shell.execute_reply.started":"2022-08-03T15:09:17.875946Z","shell.execute_reply":"2022-08-03T15:09:17.890316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in num_cols:\n    check_outliers(test,col)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.892419Z","iopub.execute_input":"2022-08-03T15:09:17.892925Z","iopub.status.idle":"2022-08-03T15:09:17.907048Z","shell.execute_reply.started":"2022-08-03T15:09:17.892888Z","shell.execute_reply":"2022-08-03T15:09:17.906053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in num_cols:\n    replace_with_thresholds(train, col)\nfor col in num_cols:\n    replace_with_thresholds(test, col)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.910893Z","iopub.execute_input":"2022-08-03T15:09:17.911340Z","iopub.status.idle":"2022-08-03T15:09:17.929534Z","shell.execute_reply.started":"2022-08-03T15:09:17.911297Z","shell.execute_reply":"2022-08-03T15:09:17.928741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in num_cols:\n    check_outliers(train,col)\nfor col in num_cols:\n    check_outliers(train,col)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.930920Z","iopub.execute_input":"2022-08-03T15:09:17.931349Z","iopub.status.idle":"2022-08-03T15:09:17.956513Z","shell.execute_reply.started":"2022-08-03T15:09:17.931311Z","shell.execute_reply":"2022-08-03T15:09:17.955454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#########################################################################################\n# 2. Missing Values\n#########################################################################################\n# Eksikligin Rassalligi.\n\ntrain.isnull().sum().sort_values(ascending = False)\n(train.isnull().sum()/ train.shape[0] * 100).sort_values(ascending = False)\nna_columns = [col for col in train.columns if train[col].isnull().sum()>0]\nna_columns\n\ndef missing_values_table(dataframe, na_name = False):\n    na_columns = [col for col in train.columns if train[col].isnull().sum()>0]\n    number_missing = train[na_columns].isnull().sum().sort_values(ascending = False)\n    ratio = (train[na_columns].isnull().sum()/ train.shape[0] * 100).sort_values(ascending = False)\n    missing_df = pd.concat([number_missing, np.round(ratio,2)], axis =1, keys = ['number_missing', 'ratio'])\n    print(missing_df, end = '\\n')\n    if na_name:\n        return na_columns\n    \nmissing_values_table(train)\nmissing_values_table(train, True)\n\n#########################################################################################\n# Handling The Missing Values\n#########################################################################################\n\n# Approach 1 : Delete\n\ntrain.dropna()\n\n# Approach 2 :  Filling the NaN with Basic Assignment Methods\n\ntrain['Age'].fillna(train['Age'].mean()).isnull().sum() # Filling it using mean value.\n\ntrain['Age'].fillna(train['Age'].median()).isnull().sum() # Filling it using mean value.\n\ntrain['Age'].fillna(0).isnull().sum() # Filling it with zeros.\n\n# Approach 3: Filling it with ML.","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:17.958180Z","iopub.execute_input":"2022-08-03T15:09:17.958555Z","iopub.status.idle":"2022-08-03T15:09:18.005808Z","shell.execute_reply.started":"2022-08-03T15:09:17.958519Z","shell.execute_reply":"2022-08-03T15:09:18.004831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#########################################################################################\n# Nan- Target Value Analysis\n#########################################################################################\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.007085Z","iopub.execute_input":"2022-08-03T15:09:18.007497Z","iopub.status.idle":"2022-08-03T15:09:18.011928Z","shell.execute_reply.started":"2022-08-03T15:09:18.007462Z","shell.execute_reply":"2022-08-03T15:09:18.010935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#########################################################################################\n# Feature Extraction\n#########################################################################################\n\n#Binary Features\n\n# Normally Cabin Feature is a garbage feature.\n\ntrain['NEW_CABIN_FEATURE'] = train['Cabin'].notnull().astype('int') # Notnull will give True-False values and astype will cast them into 1-0s.\ntrain.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.013493Z","iopub.execute_input":"2022-08-03T15:09:18.013850Z","iopub.status.idle":"2022-08-03T15:09:18.038192Z","shell.execute_reply.started":"2022-08-03T15:09:18.013816Z","shell.execute_reply":"2022-08-03T15:09:18.037315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply same thing for test.\ntest['NEW_CABIN_FEATURE'] = test['Cabin'].notnull().astype('int') # Notnull will give True-False values and astype will cast them into 1-0s.","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.039349Z","iopub.execute_input":"2022-08-03T15:09:18.040234Z","iopub.status.idle":"2022-08-03T15:09:18.055199Z","shell.execute_reply.started":"2022-08-03T15:09:18.040196Z","shell.execute_reply":"2022-08-03T15:09:18.054373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.groupby('NEW_CABIN_FEATURE').agg({'Survived': 'mean'}) \n# In average, 66% of the people whom we know their cabin name is survived. For those we do not\n# know cabin numbers, 30% of them survived.\n# Before we thought cabin feature needs to be dropped due to its high NAN rate. Now, it seems like this\n# feature has significance.\n# But, hold on a second... How do we test the relationship between 2 variables? PROPORTION ZTEST.\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.060661Z","iopub.execute_input":"2022-08-03T15:09:18.061077Z","iopub.status.idle":"2022-08-03T15:09:18.076043Z","shell.execute_reply.started":"2022-08-03T15:09:18.061040Z","shell.execute_reply":"2022-08-03T15:09:18.074977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from statsmodels.stats.proportion import proportions_ztest\ntest_stat, pvalue = proportions_ztest( count = [train.loc[train['NEW_CABIN_FEATURE'] == 1, \"Survived\"].sum(),\n                                               train.loc[train['NEW_CABIN_FEATURE'] == 0, \"Survived\"].sum()],\nnobs = [train.loc[train['NEW_CABIN_FEATURE'] == 1, \"Survived\"].shape[0],\n        train.loc[train['NEW_CABIN_FEATURE'] == 0, \"Survived\"].shape[0]])\n    \nprint('Test Stat = %.4f , p-value = %.4f' % (test_stat, pvalue))\n# p-value is lower than 0.05. We reject the h0 hyphothesis. So, there is a statistically meaningful difference between two ratios. \n#(h0: M1 == M2. There is not dif between two ratios.)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.077751Z","iopub.execute_input":"2022-08-03T15:09:18.079168Z","iopub.status.idle":"2022-08-03T15:09:18.093398Z","shell.execute_reply.started":"2022-08-03T15:09:18.079115Z","shell.execute_reply":"2022-08-03T15:09:18.092414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.loc[((train['SibSp']+train['Parch'])>0), \"NEW_IS_ALONE\"] = 'NO'\ntrain.loc[((train['SibSp']+train['Parch'])==0), \"NEW_IS_ALONE\"] = 'YES'\n# Maybe person's survival rate is about the number of their parents, childrens on board. We do not know it yet. We are just trying to generate features.\n# And see its relatability with the target feauture.","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.094731Z","iopub.execute_input":"2022-08-03T15:09:18.095739Z","iopub.status.idle":"2022-08-03T15:09:18.106360Z","shell.execute_reply.started":"2022-08-03T15:09:18.095678Z","shell.execute_reply":"2022-08-03T15:09:18.105108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.loc[((test['SibSp']+test['Parch'])>0), \"NEW_IS_ALONE\"] = 'NO'\ntest.loc[((test['SibSp']+test['Parch'])==0), \"NEW_IS_ALONE\"] = 'YES'","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.110176Z","iopub.execute_input":"2022-08-03T15:09:18.111032Z","iopub.status.idle":"2022-08-03T15:09:18.123490Z","shell.execute_reply.started":"2022-08-03T15:09:18.110981Z","shell.execute_reply":"2022-08-03T15:09:18.122364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.groupby('NEW_IS_ALONE').agg({'Survived' : 'mean'})\n# It seems that alone passenger's survive rate is lower than the ones who has at least one relative on board.\n# Let's check it's statistical relevance","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.124913Z","iopub.execute_input":"2022-08-03T15:09:18.125315Z","iopub.status.idle":"2022-08-03T15:09:18.145129Z","shell.execute_reply.started":"2022-08-03T15:09:18.125270Z","shell.execute_reply":"2022-08-03T15:09:18.144135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_stat, pvalue = proportions_ztest( count = [train.loc[train['NEW_IS_ALONE'] == 'YES', \"Survived\"].sum(),\n                                               train.loc[train['NEW_IS_ALONE'] == 'NO', \"Survived\"].sum()],\nnobs = [train.loc[train['NEW_IS_ALONE'] == 'YES', \"Survived\"].shape[0],\n        train.loc[train['NEW_IS_ALONE'] =='NO', \"Survived\"].shape[0]])\n    \nprint('Test Stat = %.4f , p-value = %.4f' % (test_stat, pvalue))\n#Out: p-value = 0.0. This means the h0 is rejected. This means there is a statistically meaningful difference between two groups' ratios.","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.147070Z","iopub.execute_input":"2022-08-03T15:09:18.147925Z","iopub.status.idle":"2022-08-03T15:09:18.161996Z","shell.execute_reply.started":"2022-08-03T15:09:18.147883Z","shell.execute_reply":"2022-08-03T15:09:18.161192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Text Features\n# Name feature has high cardinality. We couldh have seen it as useless feature and dropped it but, now that we think it can create another feature.\n\n# Letter Count : Maybe someone is from the royal family and has too many names/prefixes. Their name will be long.\n\ntrain['NEW_NAME_COUNT'] = train['Name'].str.len()\ntest['NEW_NAME_COUNT'] = test['Name'].str.len()\n\n\n# Word Count: \n\ntrain['NEW_NAME_WORD_COUNT'] = train['Name'].apply(lambda x: len(str(x).split(\" \")))\ntest['NEW_NAME_WORD_COUNT'] = test['Name'].apply(lambda x: len(str(x).split(\" \")))\n\n\n# Catching Special Structures in the Name Feature:\n\ntrain['NEW_NAME_DR'] = train['Name'].apply(lambda x: len([x for x in x.split() if x.startswith(\"Dr\")]))\ntest['NEW_NAME_DR'] = test['Name'].apply(lambda x: len([x for x in x.split() if x.startswith(\"Dr\")]))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.163303Z","iopub.execute_input":"2022-08-03T15:09:18.164260Z","iopub.status.idle":"2022-08-03T15:09:18.185862Z","shell.execute_reply.started":"2022-08-03T15:09:18.164222Z","shell.execute_reply":"2022-08-03T15:09:18.184633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.groupby('NEW_NAME_DR').agg({'Survived': ['mean', 'sum']})","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.187517Z","iopub.execute_input":"2022-08-03T15:09:18.188676Z","iopub.status.idle":"2022-08-03T15:09:18.204849Z","shell.execute_reply.started":"2022-08-03T15:09:18.188626Z","shell.execute_reply":"2022-08-03T15:09:18.204016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generating Features using REGEX.\n# Inside the NAME column, there are some patterns one needs to realize. It always starts with upper letter, then there is a comma, then upper letter again,\n# then a point, then an upper letter. We can split these names into words using REGEX feature.\n\ntrain['NEW_TITLE'] = train.Name.str.extract(' ([A-Za-z]+)\\.', expand = False)\ntest['NEW_TITLE'] = test.Name.str.extract(' ([A-Za-z]+)\\.', expand = False)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.205842Z","iopub.execute_input":"2022-08-03T15:09:18.206642Z","iopub.status.idle":"2022-08-03T15:09:18.215399Z","shell.execute_reply.started":"2022-08-03T15:09:18.206607Z","shell.execute_reply":"2022-08-03T15:09:18.214328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(20)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.216784Z","iopub.execute_input":"2022-08-03T15:09:18.217286Z","iopub.status.idle":"2022-08-03T15:09:18.243331Z","shell.execute_reply.started":"2022-08-03T15:09:18.217249Z","shell.execute_reply":"2022-08-03T15:09:18.242468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['SURNAME'] = train.Name.str.extract('([A-Za-z]+)\\,', expand = False)\ntest['SURNAME'] = test.Name.str.extract('([A-Za-z]+)\\,', expand = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.244543Z","iopub.execute_input":"2022-08-03T15:09:18.245481Z","iopub.status.idle":"2022-08-03T15:09:18.257903Z","shell.execute_reply.started":"2022-08-03T15:09:18.245437Z","shell.execute_reply":"2022-08-03T15:09:18.256942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SURNAME_DF = train.groupby('SURNAME').agg({'Survived': 'sum'})\nSURNAME_DF = SURNAME_DF.reset_index()\nx1 = SURNAME_DF[SURNAME_DF['Survived'] == 1]['SURNAME'].to_list()\nx2 = SURNAME_DF[SURNAME_DF['Survived'] == 2]['SURNAME'].to_list()\nx3 = SURNAME_DF[SURNAME_DF['Survived'] == 3]['SURNAME'].to_list()\nx4 = SURNAME_DF[SURNAME_DF['Survived'] == 4]['SURNAME'].to_list()\n\ntrain['SURNAME_MORE_ONE'] = np.where(train.SURNAME.isin(x1),True, False)\ntrain['SURNAME_MORE_TWO'] = np.where(train.SURNAME.isin(x2),True, False)\ntrain['SURNAME_MORE_THREE'] = np.where(train.SURNAME.isin(x3),True, False)\ntrain['SURNAME_MORE_FOUR'] = np.where(train.SURNAME.isin(x4),True, False)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.259177Z","iopub.execute_input":"2022-08-03T15:09:18.259690Z","iopub.status.idle":"2022-08-03T15:09:18.282357Z","shell.execute_reply.started":"2022-08-03T15:09:18.259646Z","shell.execute_reply":"2022-08-03T15:09:18.281297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['SURNAME_MORE_ONE'] = np.where(test.SURNAME.isin(x1),True, False)\ntest['SURNAME_MORE_TWO'] = np.where(test.SURNAME.isin(x2),True, False)\ntest['SURNAME_MORE_THREE'] = np.where(test.SURNAME.isin(x3),True, False)\ntest['SURNAME_MORE_FOUR'] = np.where(test.SURNAME.isin(x4),True, False)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.283881Z","iopub.execute_input":"2022-08-03T15:09:18.284326Z","iopub.status.idle":"2022-08-03T15:09:18.294464Z","shell.execute_reply.started":"2022-08-03T15:09:18.284282Z","shell.execute_reply":"2022-08-03T15:09:18.293517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.295976Z","iopub.execute_input":"2022-08-03T15:09:18.296999Z","iopub.status.idle":"2022-08-03T15:09:18.325128Z","shell.execute_reply.started":"2022-08-03T15:09:18.296952Z","shell.execute_reply":"2022-08-03T15:09:18.323981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def target_summary_with_cat(dataframe, target, categorical_col):\n    print(pd.DataFrame({\"TARGET_MEAN\": dataframe.groupby(categorical_col)[target].mean()}), end=\"\\n\\n\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.327157Z","iopub.execute_input":"2022-08-03T15:09:18.327662Z","iopub.status.idle":"2022-08-03T15:09:18.333198Z","shell.execute_reply.started":"2022-08-03T15:09:18.327613Z","shell.execute_reply":"2022-08-03T15:09:18.332162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_summary_with_cat(train, 'Survived', 'SURNAME_MORE_ONE')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.334842Z","iopub.execute_input":"2022-08-03T15:09:18.335570Z","iopub.status.idle":"2022-08-03T15:09:18.349147Z","shell.execute_reply.started":"2022-08-03T15:09:18.335520Z","shell.execute_reply":"2022-08-03T15:09:18.348126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_summary_with_cat(train, 'Survived', 'SURNAME_MORE_TWO')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.350880Z","iopub.execute_input":"2022-08-03T15:09:18.351684Z","iopub.status.idle":"2022-08-03T15:09:18.361254Z","shell.execute_reply.started":"2022-08-03T15:09:18.351638Z","shell.execute_reply":"2022-08-03T15:09:18.359947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_summary_with_cat(train, 'Survived', 'SURNAME_MORE_THREE')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.362668Z","iopub.execute_input":"2022-08-03T15:09:18.363031Z","iopub.status.idle":"2022-08-03T15:09:18.372508Z","shell.execute_reply.started":"2022-08-03T15:09:18.362998Z","shell.execute_reply":"2022-08-03T15:09:18.371535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_summary_with_cat(train, 'Survived', 'SURNAME_MORE_FOUR')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.374189Z","iopub.execute_input":"2022-08-03T15:09:18.374937Z","iopub.status.idle":"2022-08-03T15:09:18.385501Z","shell.execute_reply.started":"2022-08-03T15:09:18.374893Z","shell.execute_reply":"2022-08-03T15:09:18.384699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Feature Extractions\n\ntrain['NEW_AGE_PCLASS'] = train['Age'] * train['Pclass']\n\n# To reveal the relationship between age and the social status and the target variable.\n# For example, his age might be higher, then we expect him to stay as upper class. But if he is staying in a lower class then we might think if he has\n# low welfare status.\n\ntrain['NEW_FAMILY_SIZE'] = train['SibSp'] + train['Parch']\n\ntrain.loc[(train['Sex'] == 'male') & (train['Age'] <= 21), 'NEW_SEX_CAT'] = 'youngmale'\n\ntrain.loc[(train['Sex'] == 'male') & (train['Age'] > 21) & (train['Age'] < 50), 'NEW_SEX_CAT'] = 'maturemale'\n\ntrain.loc[(train['Sex'] == 'male') & (train['Age'] >= 50), 'NEW_SEX_CAT'] = 'seniormale'\n\ntrain.loc[(train['Sex'] == 'female') & (train['Age'] <= 21), 'NEW_SEX_CAT'] = 'youngfemale'\n\ntrain.loc[(train['Sex'] == 'female') & (train['Age'] > 21) & (train['Age'] < 50), 'NEW_SEX_CAT'] = 'maturefemale'\n\ntrain.loc[(train['Sex'] == 'female') & (train['Age'] >= 50), 'NEW_SEX_CAT'] = 'seniorfemale'\n\ntrain.groupby(\"NEW_SEX_CAT\")[\"Survived\"].mean()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.386622Z","iopub.execute_input":"2022-08-03T15:09:18.387054Z","iopub.status.idle":"2022-08-03T15:09:18.410247Z","shell.execute_reply.started":"2022-08-03T15:09:18.387024Z","shell.execute_reply":"2022-08-03T15:09:18.409549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Feature Extractions for test\n\ntest['NEW_AGE_PCLASS'] = test['Age'] * test['Pclass']\n\n# To reveal the relationship between age and the social status and the target variable.\n# For example, his age might be higher, then we expect him to stay as upper class. But if he is staying in a lower class then we might think if he has\n# low welfare status.\n\ntest['NEW_FAMILY_SIZE'] = test['SibSp'] + test['Parch']\n\ntest.loc[(train['Sex'] == 'male') & (test['Age'] <= 21), 'NEW_SEX_CAT'] = 'youngmale'\n\ntest.loc[(test['Sex'] == 'male') & (test['Age'] > 21) & (test['Age'] < 50), 'NEW_SEX_CAT'] = 'maturemale'\n\ntest.loc[(test['Sex'] == 'male') & (test['Age'] >= 50), 'NEW_SEX_CAT'] = 'seniormale'\n\ntest.loc[(test['Sex'] == 'female') & (test['Age'] <= 21), 'NEW_SEX_CAT'] = 'youngfemale'\n\ntest.loc[(test['Sex'] == 'female') & (test['Age'] > 21) & (test['Age'] < 50), 'NEW_SEX_CAT'] = 'maturefemale'\n\ntest.loc[(test['Sex'] == 'female') & (test['Age'] >= 50), 'NEW_SEX_CAT'] = 'seniorfemale'\n\n#test.groupby(\"NEW_SEX_CAT\")[\"Survived\"].mean()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.411798Z","iopub.execute_input":"2022-08-03T15:09:18.412313Z","iopub.status.idle":"2022-08-03T15:09:18.433447Z","shell.execute_reply.started":"2022-08-03T15:09:18.412280Z","shell.execute_reply":"2022-08-03T15:09:18.432330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.434733Z","iopub.execute_input":"2022-08-03T15:09:18.435257Z","iopub.status.idle":"2022-08-03T15:09:18.442058Z","shell.execute_reply.started":"2022-08-03T15:09:18.435222Z","shell.execute_reply":"2022-08-03T15:09:18.441347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data Preparation:\n\ncat_cols, num_cols, cat_but_car = grab_col_names(train)\n\nnum_cols = [col for col in num_cols if \"PassengerId\" not in col]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.443119Z","iopub.execute_input":"2022-08-03T15:09:18.443858Z","iopub.status.idle":"2022-08-03T15:09:18.460960Z","shell.execute_reply.started":"2022-08-03T15:09:18.443810Z","shell.execute_reply":"2022-08-03T15:09:18.459845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking Outliers:\nfor col in num_cols:\n    print(check_outliers(train,col))\n    \nfor col in num_cols:\n    replace_with_thresholds(train,col)\n\nfor col in num_cols:\n    print(check_outliers(train,col))\n    \nfor col in num_cols:\n    print(check_outliers(test,col))\n    \nfor col in num_cols:\n    replace_with_thresholds(test,col)\n\nfor col in num_cols:\n    print(check_outliers(test,col))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.462306Z","iopub.execute_input":"2022-08-03T15:09:18.462681Z","iopub.status.idle":"2022-08-03T15:09:18.568562Z","shell.execute_reply.started":"2022-08-03T15:09:18.462648Z","shell.execute_reply":"2022-08-03T15:09:18.567549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Handling Missing Values:\n\nmissing_values_table(train)\n\n# We already generated a new feature for cabin (['CABIN_NEW']). So, we drop Cabin.\n\ntrain.drop('Cabin',inplace= True, axis=1)\ntest.drop('Cabin',inplace= True, axis=1)\n\nremove_cols = ['Ticket', 'Name']\n\ntrain.drop(remove_cols,inplace= True, axis=1)\ntest.drop(remove_cols,inplace= True, axis=1)\n\n\ntrain['Age'] = train['Age'].fillna(train.groupby('NEW_TITLE')['Age'].transform('median'))\ntest['Age'] = test['Age'].fillna(test.groupby('NEW_TITLE')['Age'].transform('median'))\n\n\n#############################################################################################################\n# Since we deleted Age Column, now we need to RE-DEFINE the features that we generated from AGE.\n#############################################################################################################\n\ntrain['NEW_AGE_PCLASS'] = train['Age'] * train['Pclass']\n\n# To reveal the relationship between age and the social status and the target variable.\n# For example, his age might be higher, then we expect him to stay as upper class. But if he is staying in a lower class then we might think if he has\n# low welfare status.\n\ntrain.loc[(train['Sex'] == 'male') & (train['Age'] <= 21), 'NEW_SEX_CAT'] = 'youngmale'\n\ntrain.loc[(train['Sex'] == 'male') & (train['Age'] > 21) & (train['Age'] < 50), 'NEW_SEX_CAT'] = 'maturemale'\n\ntrain.loc[(train['Sex'] == 'male') & (train['Age'] >= 50), 'NEW_SEX_CAT'] = 'seniormale'\n\ntrain.loc[(train['Sex'] == 'female') & (train['Age'] <= 21), 'NEW_SEX_CAT'] = 'youngfemale'\n\ntrain.loc[(train['Sex'] == 'female') & (train['Age'] > 21) & (train['Age'] < 50), 'NEW_SEX_CAT'] = 'maturefemale'\n\ntrain.loc[(train['Sex'] == 'female') & (train['Age'] >= 50), 'NEW_SEX_CAT'] = 'seniorfemale'\n\ntrain.groupby(\"NEW_SEX_CAT\")[\"Survived\"].mean()\n\ntest['NEW_AGE_PCLASS'] = test['Age'] * test['Pclass']\n\n# To reveal the relationship between age and the social status and the target variable.\n# For example, his age might be higher, then we expect him to stay as upper class. But if he is staying in a lower class then we might think if he has\n# low welfare status.\n\ntest['NEW_FAMILY_SIZE'] = test['SibSp'] + test['Parch']\n\ntest.loc[(train['Sex'] == 'male') & (test['Age'] <= 21), 'NEW_SEX_CAT'] = 'youngmale'\n\ntest.loc[(test['Sex'] == 'male') & (test['Age'] > 21) & (test['Age'] < 50), 'NEW_SEX_CAT'] = 'maturemale'\n\ntest.loc[(test['Sex'] == 'male') & (test['Age'] >= 50), 'NEW_SEX_CAT'] = 'seniormale'\n\ntest.loc[(test['Sex'] == 'female') & (test['Age'] <= 21), 'NEW_SEX_CAT'] = 'youngfemale'\n\ntest.loc[(test['Sex'] == 'female') & (test['Age'] > 21) & (test['Age'] < 50), 'NEW_SEX_CAT'] = 'maturefemale'\n\ntest.loc[(test['Sex'] == 'female') & (test['Age'] >= 50), 'NEW_SEX_CAT'] = 'seniorfemale'\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.569828Z","iopub.execute_input":"2022-08-03T15:09:18.570186Z","iopub.status.idle":"2022-08-03T15:09:18.624868Z","shell.execute_reply.started":"2022-08-03T15:09:18.570151Z","shell.execute_reply":"2022-08-03T15:09:18.623972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"[col for col in train.columns if col not in test.columns]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.626002Z","iopub.execute_input":"2022-08-03T15:09:18.626662Z","iopub.status.idle":"2022-08-03T15:09:18.635199Z","shell.execute_reply.started":"2022-08-03T15:09:18.626620Z","shell.execute_reply":"2022-08-03T15:09:18.634151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.636410Z","iopub.execute_input":"2022-08-03T15:09:18.637327Z","iopub.status.idle":"2022-08-03T15:09:18.664219Z","shell.execute_reply.started":"2022-08-03T15:09:18.637281Z","shell.execute_reply":"2022-08-03T15:09:18.663194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.665536Z","iopub.execute_input":"2022-08-03T15:09:18.665900Z","iopub.status.idle":"2022-08-03T15:09:18.691777Z","shell.execute_reply.started":"2022-08-03T15:09:18.665860Z","shell.execute_reply":"2022-08-03T15:09:18.690420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.apply(lambda x: x.fillna(x.mode()[0]) if (x.dtype == \"O\" and len(x.unique()) <= 10) else x, axis=0)\ntest = test.apply(lambda x: x.fillna(x.mode()[0]) if (x.dtype == \"O\" and len(x.unique()) <= 10) else x, axis=0)\n\n\nmissing_values_table(train)\nmissing_values_table(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.693701Z","iopub.execute_input":"2022-08-03T15:09:18.694088Z","iopub.status.idle":"2022-08-03T15:09:18.733464Z","shell.execute_reply.started":"2022-08-03T15:09:18.694054Z","shell.execute_reply":"2022-08-03T15:09:18.732475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"binary_cols = [col for col in train.columns if train[col].dtype not in [int, float]\n               and train[col].nunique() == 2]\n\nbinary_cols2 = [col for col in test.columns if test[col].dtype not in [int, float]\n               and test[col].nunique() == 2]\nbinary_cols\nbinary_cols2","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.734706Z","iopub.execute_input":"2022-08-03T15:09:18.735026Z","iopub.status.idle":"2022-08-03T15:09:18.750478Z","shell.execute_reply.started":"2022-08-03T15:09:18.734997Z","shell.execute_reply":"2022-08-03T15:09:18.749361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.751651Z","iopub.execute_input":"2022-08-03T15:09:18.752363Z","iopub.status.idle":"2022-08-03T15:09:18.775924Z","shell.execute_reply.started":"2022-08-03T15:09:18.752318Z","shell.execute_reply":"2022-08-03T15:09:18.775146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#############################################################################################################\n# Binary Encoding\n#############################################################################################################\nfrom sklearn.preprocessing import MinMaxScaler, LabelEncoder, StandardScaler, RobustScaler\n\ndef label_encoder(dataframe, binary_col):\n    labelencoder = LabelEncoder()\n    dataframe[binary_col] = labelencoder.fit_transform(dataframe[binary_col])\n    return dataframe\n\nbinary_cols = [col for col in train.columns if train[col].dtype not in [int, float]\n               and train[col].nunique() == 2]\n\nfor col in binary_cols:\n    train = label_encoder(train, col)\n    test = label_encoder(test,col)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.777044Z","iopub.execute_input":"2022-08-03T15:09:18.777431Z","iopub.status.idle":"2022-08-03T15:09:18.792306Z","shell.execute_reply.started":"2022-08-03T15:09:18.777396Z","shell.execute_reply":"2022-08-03T15:09:18.790922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.793927Z","iopub.execute_input":"2022-08-03T15:09:18.794518Z","iopub.status.idle":"2022-08-03T15:09:18.816490Z","shell.execute_reply.started":"2022-08-03T15:09:18.794480Z","shell.execute_reply":"2022-08-03T15:09:18.815455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.817921Z","iopub.execute_input":"2022-08-03T15:09:18.818463Z","iopub.status.idle":"2022-08-03T15:09:18.838376Z","shell.execute_reply.started":"2022-08-03T15:09:18.818419Z","shell.execute_reply":"2022-08-03T15:09:18.837627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#############################################################################################################\n# Rare Analyzer\n#############################################################################################################\n\nfrom sklearn.preprocessing import MinMaxScaler, LabelEncoder, StandardScaler, RobustScaler\ndef rare_analyser(dataframe, target, cat_cols):\n    for col in cat_cols:\n        print(col, \":\", len(dataframe[col].value_counts()))\n        print(pd.DataFrame({\"COUNT\": dataframe[col].value_counts(),\n                            \"RATIO\": dataframe[col].value_counts() / len(dataframe),\n                            \"TARGET_MEAN\": dataframe.groupby(col)[target].mean()}), end=\"\\n\\n\\n\")\n\nrare_analyser(train, 'Survived',cat_cols) ","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.839446Z","iopub.execute_input":"2022-08-03T15:09:18.839976Z","iopub.status.idle":"2022-08-03T15:09:18.926909Z","shell.execute_reply.started":"2022-08-03T15:09:18.839942Z","shell.execute_reply":"2022-08-03T15:09:18.925963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rare_encoder(dataframe, rare_perc):\n    temp_df = dataframe.copy()\n\n    rare_columns = [col for col in temp_df.columns if temp_df[col].dtypes == 'O'\n                    and (temp_df[col].value_counts() / len(temp_df) < rare_perc).any(axis=None)]\n\n    for var in rare_columns:\n        tmp = temp_df[var].value_counts() / len(temp_df)\n        rare_labels = tmp[tmp < rare_perc].index\n        temp_df[var] = np.where(temp_df[var].isin(rare_labels), 'Rare', temp_df[var])\n\n    return temp_df\ntrain = rare_encoder(train, 0.01)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.928266Z","iopub.execute_input":"2022-08-03T15:09:18.928714Z","iopub.status.idle":"2022-08-03T15:09:18.945790Z","shell.execute_reply.started":"2022-08-03T15:09:18.928679Z","shell.execute_reply":"2022-08-03T15:09:18.944744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = rare_encoder(train, 0.01)\ntest = rare_encoder(test, 0.01)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.946822Z","iopub.execute_input":"2022-08-03T15:09:18.947540Z","iopub.status.idle":"2022-08-03T15:09:18.967998Z","shell.execute_reply.started":"2022-08-03T15:09:18.947503Z","shell.execute_reply":"2022-08-03T15:09:18.967066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.969713Z","iopub.execute_input":"2022-08-03T15:09:18.970199Z","iopub.status.idle":"2022-08-03T15:09:18.976984Z","shell.execute_reply.started":"2022-08-03T15:09:18.970148Z","shell.execute_reply":"2022-08-03T15:09:18.975995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# One Hot Encoding:\n\ndef one_hot_encoder(dataframe, categorical_cols, drop_first=True):\n    dataframe = pd.get_dummies(dataframe, columns=categorical_cols, drop_first=drop_first)\n    return dataframe\n\nohe_cols = [col for col in train.columns if 17 >= train[col].nunique() > 2]\nohe_cols2 = [col for col in test.columns if 17 >= test[col].nunique() > 2]\n\n\ntrain = one_hot_encoder(train, ohe_cols)\ntest = one_hot_encoder(test, ohe_cols2)\n\n\ntrain.head()\n#train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:18.978393Z","iopub.execute_input":"2022-08-03T15:09:18.978893Z","iopub.status.idle":"2022-08-03T15:09:19.037162Z","shell.execute_reply.started":"2022-08-03T15:09:18.978859Z","shell.execute_reply":"2022-08-03T15:09:19.036442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.drop('SURNAME',axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:19.038466Z","iopub.execute_input":"2022-08-03T15:09:19.038973Z","iopub.status.idle":"2022-08-03T15:09:19.044885Z","shell.execute_reply.started":"2022-08-03T15:09:19.038938Z","shell.execute_reply":"2022-08-03T15:09:19.044214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test =  test.drop('SURNAME',axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:19.046404Z","iopub.execute_input":"2022-08-03T15:09:19.047040Z","iopub.status.idle":"2022-08-03T15:09:19.057625Z","shell.execute_reply.started":"2022-08-03T15:09:19.046996Z","shell.execute_reply":"2022-08-03T15:09:19.056665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols, num_cols, cat_but_car = grab_col_names(train)\n\nnum_cols = [col for col in num_cols if \"PassengerId\" not in col]\n\nrare_analyser(train, \"Survived\", cat_cols)\n#rare_analyser(test, \"Survived\", cat_cols)\n\nuseless_cols = [col for col in train.columns if train[col].nunique() == 2 and\n                (train[col].value_counts() / len(train) < 0.01).any(axis=None)]\n                \n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:11:01.739215Z","iopub.execute_input":"2022-08-03T15:11:01.739660Z","iopub.status.idle":"2022-08-03T15:11:01.981152Z","shell.execute_reply.started":"2022-08-03T15:11:01.739624Z","shell.execute_reply":"2022-08-03T15:11:01.980044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"useless_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:11:14.806138Z","iopub.execute_input":"2022-08-03T15:11:14.806597Z","iopub.status.idle":"2022-08-03T15:11:14.813792Z","shell.execute_reply.started":"2022-08-03T15:11:14.806557Z","shell.execute_reply":"2022-08-03T15:11:14.812833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:19.345008Z","iopub.status.idle":"2022-08-03T15:09:19.345738Z","shell.execute_reply.started":"2022-08-03T15:09:19.345468Z","shell.execute_reply":"2022-08-03T15:09:19.345497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler = StandardScaler()\ntrain[num_cols] = scaler.fit_transform(train[num_cols])\ntest[num_cols] = scaler.fit_transform(test[num_cols])\n\n\ntrain[num_cols].head()\n\ntrain.head()\ntrain.shape\n#test.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:11:43.563947Z","iopub.execute_input":"2022-08-03T15:11:43.564599Z","iopub.status.idle":"2022-08-03T15:11:43.581829Z","shell.execute_reply.started":"2022-08-03T15:11:43.564549Z","shell.execute_reply":"2022-08-03T15:11:43.581133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:11:37.366763Z","iopub.execute_input":"2022-08-03T15:11:37.367250Z","iopub.status.idle":"2022-08-03T15:11:37.373918Z","shell.execute_reply.started":"2022-08-03T15:11:37.367211Z","shell.execute_reply":"2022-08-03T15:11:37.373183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train[\"Survived\"]\nX = train.drop([\"PassengerId\", \"Survived\"], axis=1)\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.30, random_state=17)\n\nfrom sklearn.ensemble import RandomForestClassifier\n\nrf_model = RandomForestClassifier(random_state=46).fit(X_train, y_train)\ny_pred = rf_model.predict(X_test)\naccuracy_score(y_pred, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:11:50.116477Z","iopub.execute_input":"2022-08-03T15:11:50.117012Z","iopub.status.idle":"2022-08-03T15:11:50.342994Z","shell.execute_reply.started":"2022-08-03T15:11:50.116960Z","shell.execute_reply":"2022-08-03T15:11:50.342314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.fillna(0,inplace =True)\na= test.copy()\ntest = test.drop('PassengerId',axis=1)\ny_pred2 = rf_model.predict(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:11:55.232241Z","iopub.execute_input":"2022-08-03T15:11:55.232640Z","iopub.status.idle":"2022-08-03T15:11:55.264609Z","shell.execute_reply.started":"2022-08-03T15:11:55.232607Z","shell.execute_reply":"2022-08-03T15:11:55.263431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_submission = pd.DataFrame({'PassengerId': a.PassengerId, 'Survived': y_pred2})\n# you could use any filename. We choose submission here\nmy_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:12:29.155750Z","iopub.execute_input":"2022-08-03T15:12:29.156175Z","iopub.status.idle":"2022-08-03T15:12:29.164788Z","shell.execute_reply.started":"2022-08-03T15:12:29.156139Z","shell.execute_reply":"2022-08-03T15:12:29.163630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_importance(model, features, num=len(X), save=False):\n    feature_imp = pd.DataFrame({'Value': model.feature_importances_, 'Feature': features.columns})\n    plt.figure(figsize=(10, 10))\n    sns.set(font_scale=1)\n    sns.barplot(x=\"Value\", y=\"Feature\", data=feature_imp.sort_values(by=\"Value\",\n                                                                      ascending=False)[0:num])\n    plt.title('Features')\n    plt.tight_layout()\n    plt.show()\n    if save:\n        plt.savefig('importances.png')\n\n\nplot_importance(rf_model, X_train)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:12:20.255282Z","iopub.execute_input":"2022-08-03T15:12:20.255687Z","iopub.status.idle":"2022-08-03T15:12:21.409907Z","shell.execute_reply.started":"2022-08-03T15:12:20.255655Z","shell.execute_reply":"2022-08-03T15:12:21.408870Z"},"trusted":true},"execution_count":null,"outputs":[]}]}