{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Score With Vs Without Outliers (Dramatic Increase):","metadata":{"execution":{"iopub.status.busy":"2022-07-25T12:06:01.724149Z","iopub.execute_input":"2022-07-25T12:06:01.724605Z","iopub.status.idle":"2022-07-25T12:06:01.730999Z","shell.execute_reply.started":"2022-07-25T12:06:01.724574Z","shell.execute_reply":"2022-07-25T12:06:01.729450Z"}}},{"cell_type":"markdown","source":"Hello kagglers .. \n\nin this notebook we are going to see the difference in score with and without outliers.\n\n**Note:** You can skip the processing steps like: Data Cleaning, Data Engineering, ... And focus on the Modeling part.\n\nIf you find this notebook helpful, Please press the **UPVOTE** button up there, This help me a lot ^-^.","metadata":{}},{"cell_type":"code","source":"# =================================================================================================\n# Importing the Libraries:\n# =================================================================================================\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\nfrom collections import Counter\nfrom sklearn.ensemble import RandomForestClassifier, AdaBoostClassifier, GradientBoostingClassifier, ExtraTreesClassifier, VotingClassifier\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.model_selection import GridSearchCV, cross_val_score, StratifiedKFold, learning_curve\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n\nsns.set(style='white', context='notebook', palette='deep')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-25T14:32:02.055116Z","iopub.execute_input":"2022-07-25T14:32:02.055517Z","iopub.status.idle":"2022-07-25T14:32:02.068836Z","shell.execute_reply.started":"2022-07-25T14:32:02.055471Z","shell.execute_reply":"2022-07-25T14:32:02.067884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# =================================================================================================\n# Importing the Data:\n# =================================================================================================\n\ntrain = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\nIDtest = test[\"PassengerId\"]\n## Join train and test datasets in order to obtain the same number of features during categorical conversion\ntrain_len = len(train)\ndataset =  pd.concat(objs=[train, test], axis=0).reset_index(drop=True)\nprint(\"Data imported successfully!!\")","metadata":{"execution":{"iopub.status.busy":"2022-07-25T14:32:02.961481Z","iopub.execute_input":"2022-07-25T14:32:02.961873Z","iopub.status.idle":"2022-07-25T14:32:02.980788Z","shell.execute_reply.started":"2022-07-25T14:32:02.961840Z","shell.execute_reply":"2022-07-25T14:32:02.980132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# =================================================================================================\n# Data Cleaning:\n# =================================================================================================\n# You can Skip this cell !!\n# ---------------------------------------------\n\n#Fill Fare missing values with the median value\ndataset[\"Fare\"] = dataset[\"Fare\"].fillna(dataset[\"Fare\"].median())\n\n# Apply log to Fare to reduce skewness distribution\ndataset[\"Fare\"] = dataset[\"Fare\"].map(lambda i: np.log(i) if i > 0 else 0)\n\n#Fill Embarked nan values of dataset set with 'S' most frequent value\ndataset[\"Embarked\"] = dataset[\"Embarked\"].fillna(\"S\")\n\n# convert Sex into categorical value 0 for male and 1 for female\ndataset[\"Sex\"] = dataset[\"Sex\"].map({\"male\": 0, \"female\":1})\n\ndataset.drop(columns = [\"Cabin\" , \"Ticket\"] , inplace = True)\n# Filling missing value of Age \n\n## Fill Age with the median age of similar rows according to Pclass, Parch and SibSp\n# Index of NaN age rows\nindex_NaN_age = list(dataset[\"Age\"][dataset[\"Age\"].isnull()].index)\n\nfor i in index_NaN_age :\n    age_med = dataset[\"Age\"].median()\n    age_pred = dataset[\"Age\"][((dataset['SibSp'] == dataset.iloc[i][\"SibSp\"]) & (dataset['Parch'] == dataset.iloc[i][\"Parch\"]) & (dataset['Pclass'] == dataset.iloc[i][\"Pclass\"]))].median()\n    if not np.isnan(age_pred) :\n        dataset['Age'].iloc[i] = age_pred\n    else :\n        dataset['Age'].iloc[i] = age_med\n        \nprint(\"Data cleaned successfully!!\")","metadata":{"execution":{"iopub.status.busy":"2022-07-25T14:32:04.099667Z","iopub.execute_input":"2022-07-25T14:32:04.100368Z","iopub.status.idle":"2022-07-25T14:32:04.611034Z","shell.execute_reply.started":"2022-07-25T14:32:04.100333Z","shell.execute_reply":"2022-07-25T14:32:04.610003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# =================================================================================================\n# Data Engineering:\n# =================================================================================================\n# You can Skip this cell !!\n# ---------------------------------------------\n\n\n# Get Title from Name\ndataset_title = [i.split(\",\")[1].split(\".\")[0].strip() for i in dataset[\"Name\"]]\ndataset[\"Title\"] = pd.Series(dataset_title)\n\n# Convert to categorical values Title \ndataset[\"Title\"] = dataset[\"Title\"].replace(['Lady', 'the Countess','Countess','Capt', 'Col','Don', 'Dr', 'Major', 'Rev', 'Sir', 'Jonkheer', 'Dona'], 'Rare')\ndataset[\"Title\"] = dataset[\"Title\"].map({\"Master\":0, \"Miss\":1, \"Ms\" : 1 , \"Mme\":1, \"Mlle\":1, \"Mrs\":1, \"Mr\":2, \"Rare\":3})\ndataset[\"Title\"] = dataset[\"Title\"].astype(int)\n\n# Drop Name variable\ndataset.drop(labels = [\"Name\"], axis = 1, inplace = True)\n\n# Create a family size descriptor from SibSp and Parch\ndataset[\"Fsize\"] = dataset[\"SibSp\"] + dataset[\"Parch\"] + 1\n\n# Create new feature of family size\ndataset['Single'] = dataset['Fsize'].map(lambda s: 1 if s == 1 else 0)\ndataset['SmallF'] = dataset['Fsize'].map(lambda s: 1 if  s == 2  else 0)\ndataset['MedF'] = dataset['Fsize'].map(lambda s: 1 if 3 <= s <= 4 else 0)\ndataset['LargeF'] = dataset['Fsize'].map(lambda s: 1 if s >= 5 else 0)\n\n# convert to indicator values Title and Embarked \ndataset = pd.get_dummies(dataset, columns = [\"Title\"])\ndataset = pd.get_dummies(dataset, columns = [\"Embarked\"], prefix=\"Em\")\n\n# Create categorical values for Pclass\ndataset[\"Pclass\"] = dataset[\"Pclass\"].astype(\"category\")\ndataset = pd.get_dummies(dataset, columns = [\"Pclass\"],prefix=\"Pc\")\n\n# Drop useless variables \ndataset.drop(labels = [\"PassengerId\"], axis = 1, inplace = True)\n\nprint(\"The operation is done successfully!!\")","metadata":{"execution":{"iopub.status.busy":"2022-07-25T14:32:05.332173Z","iopub.execute_input":"2022-07-25T14:32:05.332570Z","iopub.status.idle":"2022-07-25T14:32:05.370224Z","shell.execute_reply.started":"2022-07-25T14:32:05.332534Z","shell.execute_reply":"2022-07-25T14:32:05.369524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modeling:","metadata":{}},{"cell_type":"markdown","source":"#### Train With Outliers:","metadata":{}},{"cell_type":"code","source":"# ===========================================================================\n# Separate train dataset and test dataset\n# ===========================================================================\n\n# Train:\ntrain_with_outliers = dataset[:train_len]\ntrain_with_outliers[\"Survived\"] = train[\"Survived\"].astype(int)\n\nY_train_with_outliers = train_with_outliers[\"Survived\"]\n\nX_train_with_outliers = train_with_outliers.drop(columns = [\"Survived\"])\n\n# Test:\ntest = dataset[train_len:]\ntest.drop(labels=[\"Survived\"],axis = 1,inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-25T14:32:07.357510Z","iopub.execute_input":"2022-07-25T14:32:07.358519Z","iopub.status.idle":"2022-07-25T14:32:07.367179Z","shell.execute_reply.started":"2022-07-25T14:32:07.358470Z","shell.execute_reply":"2022-07-25T14:32:07.366392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_with_outliers.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T14:32:11.464353Z","iopub.execute_input":"2022-07-25T14:32:11.464965Z","iopub.status.idle":"2022-07-25T14:32:11.483009Z","shell.execute_reply.started":"2022-07-25T14:32:11.464921Z","shell.execute_reply":"2022-07-25T14:32:11.482376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_with_outliers.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-25T14:32:14.376300Z","iopub.execute_input":"2022-07-25T14:32:14.376636Z","iopub.status.idle":"2022-07-25T14:32:14.381657Z","shell.execute_reply.started":"2022-07-25T14:32:14.376609Z","shell.execute_reply":"2022-07-25T14:32:14.380919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will use these models:\n- Random Forest\n- SVC\n- GradientBoosting \n- AdaBoost\n","metadata":{}},{"cell_type":"markdown","source":"#### Parameters tuning:","metadata":{}},{"cell_type":"code","source":"# Cross validate model with Kfold stratified cross val\nkfold = StratifiedKFold(n_splits=10)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T14:32:17.902001Z","iopub.execute_input":"2022-07-25T14:32:17.902378Z","iopub.status.idle":"2022-07-25T14:32:17.907153Z","shell.execute_reply.started":"2022-07-25T14:32:17.902347Z","shell.execute_reply":"2022-07-25T14:32:17.906416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# =====================================================================================\n# SVC classifier\n# =====================================================================================\ndef Best_SVM(X_train , Y_train):\n    SVMC = SVC(probability=True)\n    svc_param_grid = {'kernel': ['rbf'], \n                      'gamma': [ 0.001, 0.01, 0.1, 1],\n                      'C': [1, 10, 50, 100,200,300, 1000]}\n    gsSVMC = GridSearchCV(SVMC,param_grid = svc_param_grid, cv=kfold, scoring=\"accuracy\", n_jobs= 4, verbose = 1)\n    gsSVMC.fit(X_train,Y_train)\n    return gsSVMC.best_estimator_ , gsSVMC.best_score_\n# =====================================================================================\n# RFC Parameters tunning \n# =====================================================================================\ndef Best_Random_Forest(X_train , Y_train):\n    RFC = RandomForestClassifier()\n    rf_param_grid = {\"max_depth\": [None],\n                  \"max_features\": [1, 3, 10],\n                  \"min_samples_split\": [2, 3, 10],\n                  \"min_samples_leaf\": [1, 3, 10],\n                  \"bootstrap\": [False],\n                  \"n_estimators\" :[100,300],\n                  \"criterion\": [\"gini\"]}\n    gsRFC = GridSearchCV(RFC,param_grid = rf_param_grid, cv=kfold, scoring=\"accuracy\", n_jobs= 4, verbose = 1)\n    gsRFC.fit(X_train,Y_train)\n    return  gsRFC.best_estimator_ , gsRFC.best_score_\n# =====================================================================================\n# Adaboost\n# =====================================================================================\ndef Best_ADABoosting(X_train , Y_train):\n    DTC = DecisionTreeClassifier()\n    adaDTC = AdaBoostClassifier(DTC, random_state=7)\n    ada_param_grid = {\"base_estimator__criterion\" : [\"gini\", \"entropy\"],\n                  \"base_estimator__splitter\" :   [\"best\", \"random\"],\n                  \"algorithm\" : [\"SAMME\",\"SAMME.R\"],\n                  \"n_estimators\" :[1,2],\n                  \"learning_rate\":  [0.0001, 0.001, 0.01, 0.1, 0.2, 0.3,1.5]}\n    gsadaDTC = GridSearchCV(adaDTC,param_grid = ada_param_grid, cv=kfold, scoring=\"accuracy\", n_jobs= 4, verbose = 1)\n    gsadaDTC.fit(X_train,Y_train)\n    return gsadaDTC.best_estimator_ , gsadaDTC.best_score_\n# =====================================================================================\n# Gradient boosting tunning\n# =====================================================================================\ndef Best_Gradient_Boosting(X_train , Y_train):\n    GBC = GradientBoostingClassifier()\n    gb_param_grid = {'loss' : [\"deviance\"],\n                  'n_estimators' : [100,200,300],\n                  'learning_rate': [0.1, 0.05, 0.01],\n                  'max_depth': [4, 8],\n                  'min_samples_leaf': [100,150],\n                  'max_features': [0.3, 0.1] \n                  }\n    gsGBC = GridSearchCV(GBC,param_grid = gb_param_grid, cv=kfold, scoring=\"accuracy\", n_jobs= 4, verbose = 1)\n    gsGBC.fit(X_train,Y_train)\n    return gsGBC.best_estimator_ , gsGBC.best_score_","metadata":{"execution":{"iopub.status.busy":"2022-07-25T14:32:20.180710Z","iopub.execute_input":"2022-07-25T14:32:20.181099Z","iopub.status.idle":"2022-07-25T14:32:20.197756Z","shell.execute_reply.started":"2022-07-25T14:32:20.181067Z","shell.execute_reply":"2022-07-25T14:32:20.196786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"-\"*100 , \"\\n\" , \"SVM:\" )\nSVMC_best_with_outlier , SVMC_score_with_outlier = Best_SVM(X_train_with_outliers , Y_train_with_outliers)\nprint(\"SVM Score: \" , SVMC_score_with_outlier)\nprint(\"-\"*100 , \"\\n\" , \"AdaBoost:\" )\nada_best_with_outlier , ada_score_with_outlier = Best_ADABoosting(X_train_with_outliers , Y_train_with_outliers)\nprint(\"ADABoost Score: \" , ada_score_with_outlier)\nprint(\"-\"*100 , \"\\n\" , \"Random Forest:\" )\nRFC_best_with_outlier , RFC_score_with_outlier = Best_Random_Forest(X_train_with_outliers , Y_train_with_outliers)\nprint(\"Random Forest Score: \" , RFC_score_with_outlier)\nprint(\"-\"*100 , \"\\n\" , \"Gradient Boosting:\" )\nGBC_best_with_outlier , GBC_score_with_outlier = Best_Gradient_Boosting(X_train_with_outliers , Y_train_with_outliers)\nprint(\"Gradient Boosting Score: \" , GBC_score_with_outlier)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T14:32:25.197703Z","iopub.execute_input":"2022-07-25T14:32:25.198070Z","iopub.status.idle":"2022-07-25T14:35:06.237856Z","shell.execute_reply.started":"2022-07-25T14:32:25.198040Z","shell.execute_reply":"2022-07-25T14:35:06.236683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Train without Outlier:","metadata":{}},{"cell_type":"code","source":"# Outlier detection \n\ndef detect_outliers(df,n,features):\n    \"\"\"\n    Takes a dataframe df of features and returns a list of the indices\n    corresponding to the observations containing more than n outliers according\n    to the Tukey method.\n    \"\"\"\n    outlier_indices = []\n    \n    # iterate over features(columns)\n    for col in features:\n        # 1st quartile (25%)\n        Q1 = np.percentile(df[col], 25)\n        # 3rd quartile (75%)\n        Q3 = np.percentile(df[col],75)\n        # Interquartile range (IQR)\n        IQR = Q3 - Q1\n        \n        # outlier step\n        outlier_step = 1.5 * IQR\n        \n        # Determine a list of indices of outliers for feature col\n        outlier_list_col = df[(df[col] < Q1 - outlier_step) | (df[col] > Q3 + outlier_step )].index\n        \n        # append the found outlier indices for col to the list of outlier indices \n        outlier_indices.extend(outlier_list_col)\n        \n    # select observations containing more than 2 outliers\n    outlier_indices = Counter(outlier_indices)        \n    multiple_outliers = list( k for k, v in outlier_indices.items() if v > n )\n    \n    return multiple_outliers   \n\n# detect outliers from Age, SibSp , Parch and Fare\nOutliers_to_drop = detect_outliers(X_train_with_outliers,2,[\"Age\",\"SibSp\",\"Parch\",\"Fare\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-25T14:35:11.616619Z","iopub.execute_input":"2022-07-25T14:35:11.617835Z","iopub.status.idle":"2022-07-25T14:35:11.633042Z","shell.execute_reply.started":"2022-07-25T14:35:11.617781Z","shell.execute_reply":"2022-07-25T14:35:11.631971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_with_outliers.loc[Outliers_to_drop] # Show the outliers rows","metadata":{"execution":{"iopub.status.busy":"2022-07-25T14:35:16.160158Z","iopub.execute_input":"2022-07-25T14:35:16.160613Z","iopub.status.idle":"2022-07-25T14:35:16.179473Z","shell.execute_reply.started":"2022-07-25T14:35:16.160565Z","shell.execute_reply":"2022-07-25T14:35:16.178460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop outliers\ntrain_without_outliers = train_with_outliers.drop(Outliers_to_drop, axis = 0).reset_index(drop=True)\nY_train_without_outliers = train_without_outliers[\"Survived\"]\nX_train_without_outliers = train_without_outliers.drop(columns = [\"Survived\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-25T14:35:27.588168Z","iopub.execute_input":"2022-07-25T14:35:27.588563Z","iopub.status.idle":"2022-07-25T14:35:27.596447Z","shell.execute_reply.started":"2022-07-25T14:35:27.588531Z","shell.execute_reply":"2022-07-25T14:35:27.595750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"-\"*100 , \"\\n\" , \"SVM:\" )\nSVMC_best_without_outlier , SVMC_score_without_outlier = Best_SVM(X_train_without_outliers , Y_train_without_outliers)\nprint(\"SVM Score: \" , SVMC_score_without_outlier)\nprint(\"-\"*100 , \"\\n\" , \"AdaBoost:\" )\nada_best_without_outlier , ada_score_without_outlier = Best_ADABoosting(X_train_without_outliers , Y_train_without_outliers)\nprint(\"ADABoost Score: \" , ada_score_without_outlier)\nprint(\"-\"*100 , \"\\n\" , \"Random Forest:\" )\nRFC_best_without_outlier , RFC_score_without_outlier = Best_Random_Forest(X_train_without_outliers , Y_train_without_outliers)\nprint(\"Random Forest Score: \" , RFC_score_without_outlier)\nprint(\"-\"*100 , \"\\n\" , \"Gradient Boosting:\" )\nGBC_best_without_outlier , GBC_score_without_outlier = Best_Gradient_Boosting(X_train_without_outliers , Y_train_without_outliers)\nprint(\"Gradient Boosting Score: \" , GBC_score_without_outlier)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T14:35:47.683909Z","iopub.execute_input":"2022-07-25T14:35:47.684264Z","iopub.status.idle":"2022-07-25T14:38:27.438125Z","shell.execute_reply.started":"2022-07-25T14:35:47.684234Z","shell.execute_reply":"2022-07-25T14:38:27.437020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result = pd.DataFrame({\"Algorithm\":[\"SVM\" , \"AdaBoost\" , \"Random Forest\" , \"Geadient Boosting\"],\n           \"With outliers\" : [SVMC_score_with_outlier , ada_score_with_outlier , RFC_score_with_outlier , GBC_score_with_outlier], \n           \"Without outliers\":[SVMC_score_without_outlier , ada_score_without_outlier , RFC_score_without_outlier , GBC_score_without_outlier],\n          })\nresult[\"difference\"] = result[\"Without outliers\"] - result[\"With outliers\"]\nresult","metadata":{"execution":{"iopub.status.busy":"2022-07-25T14:38:29.741795Z","iopub.execute_input":"2022-07-25T14:38:29.742146Z","iopub.status.idle":"2022-07-25T14:38:29.756989Z","shell.execute_reply.started":"2022-07-25T14:38:29.742102Z","shell.execute_reply":"2022-07-25T14:38:29.755998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}