{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import Necessary Libraries","metadata":{"_kg_hide-input":true}},{"cell_type":"code","source":"# Data Wrangling libraries\nimport numpy as np\nimport pandas as pd\nimport scipy.stats as stats\n\n# Visualization Libraries\nfrom IPython.display import display,HTML\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Preprocessing libraries\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import LabelEncoder, OneHotEncoder\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler\nfrom sklearn.model_selection import train_test_split\n\n# Machine Learning Estimators\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nimport xgboost as xgb\n\n# Metrics\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import accuracy_score, precision_recall_fscore_support\n\n# Visualization Library\nfrom visualizationfunctions import *","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.259106Z","iopub.execute_input":"2022-07-31T09:43:09.259564Z","iopub.status.idle":"2022-07-31T09:43:09.271996Z","shell.execute_reply.started":"2022-07-31T09:43:09.259522Z","shell.execute_reply":"2022-07-31T09:43:09.270283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing the dataset","metadata":{}},{"cell_type":"code","source":"# Importing the dataset\nspaceship_train_data = pd.read_csv('../input/spaceship-titanic/train.csv')\nspaceship_test_data = pd.read_csv('../input/spaceship-titanic/test.csv')\nspaceship_train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.273969Z","iopub.execute_input":"2022-07-31T09:43:09.274364Z","iopub.status.idle":"2022-07-31T09:43:09.344432Z","shell.execute_reply.started":"2022-07-31T09:43:09.274329Z","shell.execute_reply":"2022-07-31T09:43:09.343277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{}},{"cell_type":"markdown","source":"## Filling in missing values","metadata":{}},{"cell_type":"code","source":"# Check for missing values\nspaceship_test_data.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.346058Z","iopub.execute_input":"2022-07-31T09:43:09.347693Z","iopub.status.idle":"2022-07-31T09:43:09.360833Z","shell.execute_reply.started":"2022-07-31T09:43:09.347641Z","shell.execute_reply":"2022-07-31T09:43:09.359963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function for filling up missing values\ndef fill_null_values(data, strategy='constant', fill_value=None):\n    imputer = SimpleImputer(missing_values=np.nan, strategy=strategy, fill_value=fill_value)\n    return imputer.fit_transform(data)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.363263Z","iopub.execute_input":"2022-07-31T09:43:09.364626Z","iopub.status.idle":"2022-07-31T09:43:09.374746Z","shell.execute_reply.started":"2022-07-31T09:43:09.364588Z","shell.execute_reply":"2022-07-31T09:43:09.373553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### CryoSleep and VIP Column (Boolean Columns)","metadata":{}},{"cell_type":"code","source":"COLUMN_NAME = 'CryoSleep'\n\n# Check for null values\nprint(f\"Missing Values: {spaceship_train_data[COLUMN_NAME].isna().sum()}\")\n\n# Check for most frequent values \nprint(f\"\\nValue Counts:\\n{spaceship_train_data[COLUMN_NAME].value_counts()}\")\n\n# Fill in the most frequent value (False to the null records)\nspaceship_train_data[COLUMN_NAME] = fill_null_values(data=spaceship_train_data[[COLUMN_NAME]],\n                                                     strategy='constant',\n                                                     fill_value=False)\n\nspaceship_test_data[COLUMN_NAME] = fill_null_values(data=spaceship_test_data[[COLUMN_NAME]],\n                                                     strategy='constant',\n                                                     fill_value=False)\n\n# Ensure that there are no values\nprint(f\"\\nNumber of missing values after imputing: {spaceship_train_data[COLUMN_NAME].isna().sum()}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.376339Z","iopub.execute_input":"2022-07-31T09:43:09.376822Z","iopub.status.idle":"2022-07-31T09:43:09.408165Z","shell.execute_reply.started":"2022-07-31T09:43:09.376789Z","shell.execute_reply":"2022-07-31T09:43:09.406977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"COLUMN_NAME = 'VIP'\n\n# Check for null values\nprint(f\"Missing Values: {spaceship_train_data[COLUMN_NAME].isna().sum()}\")\n\n# Check for most frequent values \nprint(f\"\\nValue Counts:\\n{spaceship_train_data[COLUMN_NAME].value_counts()}\")\n\n# Fill in the most frequent value (False to the null records)\nspaceship_train_data[COLUMN_NAME] = fill_null_values(data=spaceship_train_data[[COLUMN_NAME]],\n                                                     strategy='constant',\n                                                     fill_value=False)\n\nspaceship_test_data[COLUMN_NAME] = fill_null_values(data=spaceship_test_data[[COLUMN_NAME]],\n                                                     strategy='constant',\n                                                     fill_value=False)\n\n# Ensure that there are no values\nprint(f\"\\nNumber of missing values after imputing: {spaceship_train_data[COLUMN_NAME].isna().sum()}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.409779Z","iopub.execute_input":"2022-07-31T09:43:09.410606Z","iopub.status.idle":"2022-07-31T09:43:09.433635Z","shell.execute_reply.started":"2022-07-31T09:43:09.410573Z","shell.execute_reply":"2022-07-31T09:43:09.432040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Cabin Column\n","metadata":{}},{"cell_type":"code","source":"COLUMN_NAME = 'Cabin'\n\n# Find the most frequent deck\ndeck = spaceship_train_data[spaceship_train_data[COLUMN_NAME].isna() == False][COLUMN_NAME].apply(lambda x: str(x).split('/')[0])\nprint(f'Most frequent deck letter: {deck.value_counts().index[0]}')\n\n# Find the median of the cabin numbers\ncabin_nums = spaceship_train_data[spaceship_train_data[COLUMN_NAME].isna() == False][COLUMN_NAME].apply(lambda x: str(x).split('/')[1])\nprint(f\"Median Cabin number: {cabin_nums.astype('float').median()}\")\n\n# Find the most frequent side\nside = spaceship_train_data[spaceship_train_data[COLUMN_NAME].isna() == False][COLUMN_NAME].apply(lambda x: str(x).split('/')[2])\nprint(f'Most frequent side letter: {side.value_counts().index[0]}')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.436181Z","iopub.execute_input":"2022-07-31T09:43:09.436820Z","iopub.status.idle":"2022-07-31T09:43:09.477354Z","shell.execute_reply.started":"2022-07-31T09:43:09.436785Z","shell.execute_reply":"2022-07-31T09:43:09.476180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for null values\nprint(f\"Missing Values: {spaceship_train_data[COLUMN_NAME].isna().sum()}\")\n\n# Fill in the value 'F/427/S' to null records\nspaceship_train_data[COLUMN_NAME] = fill_null_values(data=spaceship_train_data[[COLUMN_NAME]],\n                                                     strategy='constant',\n                                                     fill_value='F/427/S')\n\nspaceship_test_data[COLUMN_NAME] = fill_null_values(data=spaceship_test_data[[COLUMN_NAME]],\n                                                     strategy='constant',\n                                                     fill_value='F/427/S')\n\n# Ensure that there are no values\nprint(f\"\\nNumber of missing values after imputing: {spaceship_train_data[COLUMN_NAME].isna().sum()}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.478888Z","iopub.execute_input":"2022-07-31T09:43:09.479243Z","iopub.status.idle":"2022-07-31T09:43:09.499748Z","shell.execute_reply.started":"2022-07-31T09:43:09.479210Z","shell.execute_reply":"2022-07-31T09:43:09.498886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Destination and HomePlanet Column","metadata":{}},{"cell_type":"code","source":"COLUMN_NAME = 'Destination'\n\n# Check for null values\nprint(f\"Missing Values: {spaceship_train_data[COLUMN_NAME].isna().sum()}\")\n\n# Check for most frequent values \nprint(f\"\\nValue Counts:\\n{spaceship_train_data[COLUMN_NAME].value_counts()}\")\n\n# Fill in the most frequent value (TRAPPIST-1e to the null records)\nspaceship_train_data[COLUMN_NAME] = fill_null_values(data=spaceship_train_data[[COLUMN_NAME]],\n                                                     strategy='constant',\n                                                     fill_value='TRAPPIST-1e')\n\nspaceship_test_data[COLUMN_NAME] = fill_null_values(data=spaceship_test_data[[COLUMN_NAME]],\n                                                     strategy='constant',\n                                                     fill_value='TRAPPIST-1e')\n\n# Ensure that there are no values\nprint(f\"\\nNumber of missing values after imputing: {spaceship_train_data[COLUMN_NAME].isna().sum()}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.501174Z","iopub.execute_input":"2022-07-31T09:43:09.501842Z","iopub.status.idle":"2022-07-31T09:43:09.531823Z","shell.execute_reply.started":"2022-07-31T09:43:09.501794Z","shell.execute_reply":"2022-07-31T09:43:09.530036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"COLUMN_NAME = 'HomePlanet'\n\n# Check for null values\nprint(f\"Missing Values: {spaceship_train_data[COLUMN_NAME].isna().sum()}\")\n\n# Check for most frequent values \nprint(f\"\\nValue Counts:\\n{spaceship_train_data[COLUMN_NAME].value_counts()}\")\n\n# Fill in the most frequent value (Earth to the null records)\nspaceship_train_data[COLUMN_NAME] = fill_null_values(data=spaceship_train_data[[COLUMN_NAME]],\n                                                     strategy='constant',\n                                                     fill_value='Earth')\n\nspaceship_test_data[COLUMN_NAME] = fill_null_values(data=spaceship_test_data[[COLUMN_NAME]],\n                                                     strategy='constant',\n                                                     fill_value='Earth')\n\n# Ensure that there are no values\nprint(f\"\\nNumber of missing values after imputing: {spaceship_train_data[COLUMN_NAME].isna().sum()}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.534241Z","iopub.execute_input":"2022-07-31T09:43:09.535087Z","iopub.status.idle":"2022-07-31T09:43:09.560790Z","shell.execute_reply.started":"2022-07-31T09:43:09.535018Z","shell.execute_reply":"2022-07-31T09:43:09.559480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Numerical Columns","metadata":{}},{"cell_type":"code","source":"spaceship_train_data['Age'] = fill_null_values(data=spaceship_train_data[['Age']], strategy='mean')\nspaceship_train_data['RoomService'] = fill_null_values(data=spaceship_train_data[['RoomService']], strategy='mean')\nspaceship_train_data['ShoppingMall'] = fill_null_values(data=spaceship_train_data[['ShoppingMall']], strategy='mean')\nspaceship_train_data['Spa'] = fill_null_values(data=spaceship_train_data[['Spa']], strategy='mean')\nspaceship_train_data['VRDeck'] = fill_null_values(data=spaceship_train_data[['VRDeck']], strategy='mean')\nspaceship_train_data['FoodCourt'] = fill_null_values(data=spaceship_train_data[['FoodCourt']], strategy='mean')\n\nspaceship_test_data['Age'] = fill_null_values(data=spaceship_test_data[['Age']], strategy='mean')\nspaceship_test_data['RoomService'] = fill_null_values(data=spaceship_test_data[['RoomService']], strategy='mean')\nspaceship_test_data['ShoppingMall'] = fill_null_values(data=spaceship_test_data[['ShoppingMall']], strategy='mean')\nspaceship_test_data['Spa'] = fill_null_values(data=spaceship_test_data[['Spa']], strategy='mean')\nspaceship_test_data['VRDeck'] = fill_null_values(data=spaceship_test_data[['VRDeck']], strategy='mean')\nspaceship_test_data['FoodCourt'] = fill_null_values(data=spaceship_test_data[['FoodCourt']], strategy='mean')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.562740Z","iopub.execute_input":"2022-07-31T09:43:09.563352Z","iopub.status.idle":"2022-07-31T09:43:09.620956Z","shell.execute_reply.started":"2022-07-31T09:43:09.563315Z","shell.execute_reply":"2022-07-31T09:43:09.620058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spaceship_test_data.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.622431Z","iopub.execute_input":"2022-07-31T09:43:09.623011Z","iopub.status.idle":"2022-07-31T09:43:09.634941Z","shell.execute_reply.started":"2022-07-31T09:43:09.622978Z","shell.execute_reply":"2022-07-31T09:43:09.634093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Drop Unnecessary Columns\n","metadata":{}},{"cell_type":"code","source":"# Drop Name and PassengerId\nspaceship_train_data.drop(['Name', 'PassengerId'], axis=1, inplace=True)\nspaceship_test_data.drop(['Name', 'PassengerId'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.636337Z","iopub.execute_input":"2022-07-31T09:43:09.636697Z","iopub.status.idle":"2022-07-31T09:43:09.649789Z","shell.execute_reply.started":"2022-07-31T09:43:09.636667Z","shell.execute_reply":"2022-07-31T09:43:09.648716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spaceship_train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.655592Z","iopub.execute_input":"2022-07-31T09:43:09.656036Z","iopub.status.idle":"2022-07-31T09:43:09.678505Z","shell.execute_reply.started":"2022-07-31T09:43:09.655997Z","shell.execute_reply":"2022-07-31T09:43:09.677527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Converting all categories into numbers and Normalize the data","metadata":{}},{"cell_type":"code","source":"# Import sklearn libraries for category -> number conversio\n\ndef preprocessing(data, training=False):\n    \n    if training:\n        data['Transported'] = data['Transported'].apply(lambda x: int(x))\n        \n    data['VIP'] = data['VIP'].apply(lambda x: int(x))\n    data['CryoSleep'] = data['CryoSleep'].apply(lambda x: int(x))\n\n    data['Deck'] = data['Cabin'].apply(lambda x: str(x).split('/')[0])\n    data['Side'] = data['Cabin'].apply(lambda x: str(x).split('/')[2])\n    data['Cabin'] = data['Cabin'].apply(lambda x: int(str(x).split('/')[1]))\n    \n    data = pd.get_dummies(data, columns=['HomePlanet', 'Destination', 'Deck', 'Side'], drop_first=True)\n\n    return data","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.679911Z","iopub.execute_input":"2022-07-31T09:43:09.680879Z","iopub.status.idle":"2022-07-31T09:43:09.692223Z","shell.execute_reply.started":"2022-07-31T09:43:09.680841Z","shell.execute_reply":"2022-07-31T09:43:09.691284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply the preprocessing function\nspaceship_train_data = preprocessing(spaceship_train_data, True)\nspaceship_test_data = preprocessing(spaceship_test_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.693584Z","iopub.execute_input":"2022-07-31T09:43:09.694779Z","iopub.status.idle":"2022-07-31T09:43:09.792326Z","shell.execute_reply.started":"2022-07-31T09:43:09.694736Z","shell.execute_reply":"2022-07-31T09:43:09.791037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratry Data Analysis","metadata":{}},{"cell_type":"code","source":"# Display the train dataset\ndisplay(HTML(spaceship_train_data.head().to_html()))","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.796895Z","iopub.execute_input":"2022-07-31T09:43:09.797399Z","iopub.status.idle":"2022-07-31T09:43:09.819128Z","shell.execute_reply.started":"2022-07-31T09:43:09.797346Z","shell.execute_reply":"2022-07-31T09:43:09.818177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# View Dataset Statistics\ndisplay(HTML(spaceship_train_data.describe().to_html()))","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.820542Z","iopub.execute_input":"2022-07-31T09:43:09.821062Z","iopub.status.idle":"2022-07-31T09:43:09.896580Z","shell.execute_reply.started":"2022-07-31T09:43:09.821030Z","shell.execute_reply":"2022-07-31T09:43:09.895401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert target values (Transported) as 1:True and 0:False\nspaceship_train_data['Transported'].apply(lambda x: 1 if True else 0)\n\ntransported_true = [i for i,val in enumerate(spaceship_train_data['Transported']) if val==1] #indices of true cases\nn_vertical = 5 #vertical resolution of the contour data\nX = spaceship_train_data.drop('Transported', axis=1)\n    \nplot_countor_map(spaceship_train_data, transported_true, n_vertical, X)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:09.898253Z","iopub.execute_input":"2022-07-31T09:43:09.898819Z","iopub.status.idle":"2022-07-31T09:43:26.133695Z","shell.execute_reply.started":"2022-07-31T09:43:09.898783Z","shell.execute_reply":"2022-07-31T09:43:26.132549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(style = 'whitegrid', rc = {'figure.figsize': (20,15)})\nplot_heatmap(\n    height=15,\n    data = spaceship_train_data.corr(),\n    title = 'Spaceship Train Dataset Correlation Overview',\n    subtitle = 'Method of correlation: Pearson Correlation Coefficient'\n);","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:26.135395Z","iopub.execute_input":"2022-07-31T09:43:26.136961Z","iopub.status.idle":"2022-07-31T09:43:28.504285Z","shell.execute_reply.started":"2022-07-31T09:43:26.136911Z","shell.execute_reply":"2022-07-31T09:43:28.503046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.seterr(divide='ignore', invalid='ignore')\nsns.set(style = 'whitegrid',\n            rc = {'figure.figsize': (20,10)})\n\nanv = anova(spaceship_train_data, 'Transported', title='Feature Disparity')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:28.505870Z","iopub.execute_input":"2022-07-31T09:43:28.506950Z","iopub.status.idle":"2022-07-31T09:43:32.971696Z","shell.execute_reply.started":"2022-07-31T09:43:28.506905Z","shell.execute_reply":"2022-07-31T09:43:32.970457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(style = 'whitegrid',\n            rc = {'figure.figsize': (20, 10)})\nsns.despine(left=True, bottom=True)\n    \nqualitative = spaceship_train_data.drop('Transported', axis=1).columns.to_list()\ntarget = 'Transported'\nspearman(spaceship_train_data, \n         qualitative, target, \n         'Spearman Correlation',\n         'Correlation analysis for spaceship titanic features with target variable (Transported)')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:32.973235Z","iopub.execute_input":"2022-07-31T09:43:32.974268Z","iopub.status.idle":"2022-07-31T09:43:33.588202Z","shell.execute_reply.started":"2022-07-31T09:43:32.974227Z","shell.execute_reply":"2022-07-31T09:43:33.587090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training and Evaluation using Different Estimators","metadata":{}},{"cell_type":"markdown","source":"## Splitting Spaceship Train Data into Training and Validation sets","metadata":{}},{"cell_type":"code","source":"X = spaceship_train_data.drop('Transported', axis=1)\ny = spaceship_train_data['Transported']\nX_train, X_val, y_train, y_val = train_test_split(X,\n                                                  y,\n                                                  test_size=0.2, \n                                                  stratify=spaceship_train_data['Transported'])","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:33.590156Z","iopub.execute_input":"2022-07-31T09:43:33.591042Z","iopub.status.idle":"2022-07-31T09:43:33.606536Z","shell.execute_reply.started":"2022-07-31T09:43:33.590992Z","shell.execute_reply":"2022-07-31T09:43:33.605274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(style = 'whitegrid',\n            rc = {'figure.figsize': (20,5)})\nplot_countplot(y = y_train.astype(str).replace({\"0\": \"Not Transported [0]\", \"1\": \"Transported [1]\"}),\n               title = 'Countplot of Target variable (Transported) from the spaceship train dataset',\n               height = 3);","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:33.608032Z","iopub.execute_input":"2022-07-31T09:43:33.608507Z","iopub.status.idle":"2022-07-31T09:43:33.845134Z","shell.execute_reply.started":"2022-07-31T09:43:33.608474Z","shell.execute_reply":"2022-07-31T09:43:33.844201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(style = 'whitegrid',\n            rc = {'figure.figsize': (20,5)})\nplot_countplot(y = y_val.astype(str).replace({\"0\": \"Not Transported [0]\", \"1\": \"Transported [1]\"}), \n               title = 'Countplot of Target variable (Transported) from the spaceship validation dataset',\n               height = 3);","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:33.846218Z","iopub.execute_input":"2022-07-31T09:43:33.847330Z","iopub.status.idle":"2022-07-31T09:43:34.055565Z","shell.execute_reply.started":"2022-07-31T09:43:33.847284Z","shell.execute_reply":"2022-07-31T09:43:34.054408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## K Nearest Neighbors (KNN)\nThe k-nearest neighbors algorithm, also known as KNN or k-NN, is a non-parametric, supervised learning classifier, which uses proximity to make classifications or predictions about the grouping of an individual data point.","metadata":{}},{"cell_type":"code","source":"n_neighbors = [4, 5, 6]\nfig,axes = plt.subplots(1,3,figsize = (15,5)) #create figure\nperfs = []\npredictions = []\n\nfor i,(n,ax) in enumerate(zip(n_neighbors,axes)):\n\n    #create and train the model\n    model = KNeighborsClassifier(n_neighbors=n)\n    model.fit(X_train,y_train)\n\n    #create predictions of testing data and store these predictions\n    y_pred = model.predict(X_val)\n    predictions.append(y_pred)\n\n    #get training testing accuracy scores and store in a list\n    train_score = model.score(X_train,y_train)\n    val_score =  model.score(X_val,y_val) \n\n    perfs.append((train_score,val_score))\n    plot_confusion_matrix(y_val, y_pred, train_score, val_score, hp_name=f'Neighbors = {n}', ax=ax, i=i)\n\nplt.suptitle(f\"Confusion Matrices (K-Nearest Neighbors)\")\nplt.tight_layout();","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:34.056710Z","iopub.execute_input":"2022-07-31T09:43:34.057012Z","iopub.status.idle":"2022-07-31T09:43:40.331607Z","shell.execute_reply.started":"2022-07-31T09:43:34.056984Z","shell.execute_reply":"2022-07-31T09:43:40.330455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict using KNN Classifier\nknn_model = KNeighborsClassifier(n_neighbors=5) # Initialize KNN with the hyperparameters from the best performing model\ny_pred = knn_model.fit(X_train, y_train).predict(X_val)\n\nsns.set(style = 'whitegrid', rc = {'figure.figsize': (20,15)})\nsns.set_palette('hls')\n    \nknn_results = graph_classification_reports(y_pred, y_val, title='KNN Model Classification Reports')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:40.333022Z","iopub.execute_input":"2022-07-31T09:43:40.333345Z","iopub.status.idle":"2022-07-31T09:43:40.943468Z","shell.execute_reply.started":"2022-07-31T09:43:40.333315Z","shell.execute_reply":"2022-07-31T09:43:40.942128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_model_performance(parameters=n_neighbors,\n                       title='Training and Testing Model Performances',\n                       perfs=perfs,\n                       x_labels='KNN Model')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:40.944864Z","iopub.execute_input":"2022-07-31T09:43:40.945217Z","iopub.status.idle":"2022-07-31T09:43:41.309254Z","shell.execute_reply.started":"2022-07-31T09:43:40.945185Z","shell.execute_reply":"2022-07-31T09:43:41.308245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Logistic Regression\nIn statistics, the logistic model is a statistical model that models the probability of one event taking place by having the log-odds for the event be a linear combination of one or more independent variables. In regression analysis, logistic regression is estimating the parameters of a logistic model.","metadata":{}},{"cell_type":"code","source":"C = [0.1, 1, 10]\nfig,axes = plt.subplots(1,3,figsize = (15,5)) #create figure\nperfs = []\npredictions = []\n\nfor i,(c,ax) in enumerate(zip(C,axes)):\n\n    #create and train the model\n    model = LogisticRegression(C=c, max_iter=10000)\n    model.fit(X_train,y_train)\n\n    #create predictions of testing data and store these predictions\n    y_pred = model.predict(X_val)\n    predictions.append(y_pred)\n\n    #get training testing accuracy scores and store in a list\n    train_score = model.score(X_train,y_train)\n    val_score =  model.score(X_val,y_val) \n\n    perfs.append((train_score,val_score))\n    plot_confusion_matrix(y_val, y_pred, train_score, val_score, hp_name=f'C = {c}', ax=ax, i=i)\n\nplt.suptitle(f\"Confusion Matrices (Logistic Regression)\")\nplt.tight_layout();","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:41.310545Z","iopub.execute_input":"2022-07-31T09:43:41.311302Z","iopub.status.idle":"2022-07-31T09:43:45.430060Z","shell.execute_reply.started":"2022-07-31T09:43:41.311269Z","shell.execute_reply":"2022-07-31T09:43:45.429212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict using Logistic Regression\nlog_reg = LogisticRegression(C=0.1, max_iter=10000) # Initialize Logistic Regression with the hyperparameters from the best performing model\ny_pred = log_reg.fit(X_train, y_train).predict(X_val)\n\nlog_reg_results = graph_classification_reports(y_pred, y_val, title='Logistic Regression Model Classification Reports')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:45.431340Z","iopub.execute_input":"2022-07-31T09:43:45.431914Z","iopub.status.idle":"2022-07-31T09:43:46.146179Z","shell.execute_reply.started":"2022-07-31T09:43:45.431882Z","shell.execute_reply":"2022-07-31T09:43:46.144983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_model_performance(parameters=C,\n                       title='Training and Testing Model Performances',\n                       perfs=perfs,\n                       x_labels='Logistic Regression Model')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:46.147746Z","iopub.execute_input":"2022-07-31T09:43:46.148089Z","iopub.status.idle":"2022-07-31T09:43:46.510981Z","shell.execute_reply.started":"2022-07-31T09:43:46.148060Z","shell.execute_reply":"2022-07-31T09:43:46.509940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Random Forest Classifier\nRandom forests or random decision forests is an ensemble learning method for classification, regression and other tasks that operates by constructing a multitude of decision trees at training time. For classification tasks, the output of the random forest is the class selected by most trees.","metadata":{}},{"cell_type":"code","source":"n_estimators = [100, 500, 1000]\nfig,axes = plt.subplots(1,3,figsize = (15,5)) #create figure\nperfs = []\npredictions = []\n\nfor i,(n,ax) in enumerate(zip(n_estimators,axes)):\n\n    #create and train the model\n    model = RandomForestClassifier(n_estimators=n)\n    model.fit(X_train,y_train)\n\n    #create predictions of testing data and store these predictions\n    y_pred = model.predict(X_val)\n    predictions.append(y_pred)\n\n    #get training testing accuracy scores and store in a list\n    train_score = model.score(X_train,y_train)\n    val_score =  model.score(X_val,y_val) \n\n    perfs.append((train_score,val_score))\n    plot_confusion_matrix(y_val, y_pred, train_score, val_score, hp_name=f'n_estimatrs = {n}', ax=ax, i=i)\n\nplt.suptitle(f\"Confusion Matrices (Random Forest Classifier)\")\nplt.tight_layout();","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:43:46.512763Z","iopub.execute_input":"2022-07-31T09:43:46.513323Z","iopub.status.idle":"2022-07-31T09:44:05.603501Z","shell.execute_reply.started":"2022-07-31T09:43:46.513276Z","shell.execute_reply":"2022-07-31T09:44:05.602328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict using Random Forest Classifier\nrf = RandomForestClassifier(n_estimators=500) # Initialize Random Forest with the hyperparameters from the best performing model\ny_pred = rf.fit(X_train, y_train).predict(X_val)\n\nrf_results = graph_classification_reports(y_pred, y_val, title='Logistic Regression Model Classification Reports')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:44:05.605319Z","iopub.execute_input":"2022-07-31T09:44:05.606487Z","iopub.status.idle":"2022-07-31T09:44:10.630113Z","shell.execute_reply.started":"2022-07-31T09:44:05.606436Z","shell.execute_reply":"2022-07-31T09:44:10.629018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_model_performance(parameters=n_estimators,\n                       title='Training and Testing Model Performances',\n                       perfs=perfs,\n                       x_labels='Random Forest Model')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:44:10.631707Z","iopub.execute_input":"2022-07-31T09:44:10.632038Z","iopub.status.idle":"2022-07-31T09:44:10.995006Z","shell.execute_reply.started":"2022-07-31T09:44:10.632007Z","shell.execute_reply":"2022-07-31T09:44:10.993906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## XGBoost Classifier\nXGBoost, which stands for Extreme Gradient Boosting, is a scalable, distributed gradient-boosted decision tree (GBDT) machine learning library. It provides parallel tree boosting and is the leading machine learning library for regression, classification, and ranking problems.","metadata":{}},{"cell_type":"code","source":"n_estimators_xgb = [50, 100, 150]\nfig,axes = plt.subplots(1,3,figsize = (15,5)) #create figure\nperfs = []\npredictions = []\n\nfor i,(n,ax) in enumerate(zip(n_estimators_xgb,axes)):\n\n    #create and train the model\n    model = xgb.XGBClassifier(n_estimators=n)\n    model.fit(X_train,y_train)\n\n    #create predictions of testing data and store these predictions\n    y_pred = model.predict(X_val)\n    predictions.append(y_pred)\n\n    #get training testing accuracy scores and store in a list\n    train_score = model.score(X_train,y_train)\n    val_score =  model.score(X_val,y_val) \n\n    perfs.append((train_score,val_score))\n    plot_confusion_matrix(y_val, y_pred, train_score, val_score, hp_name=f'n_estimators = {n}', ax=ax, i=i)\n\nplt.suptitle(f\"Confusion Matrices (XGBoost Classifier)\")\nplt.tight_layout();","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:44:11.000618Z","iopub.execute_input":"2022-07-31T09:44:11.001243Z","iopub.status.idle":"2022-07-31T09:44:15.076595Z","shell.execute_reply.started":"2022-07-31T09:44:11.001203Z","shell.execute_reply":"2022-07-31T09:44:15.075410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict using XGBoost Classifier\nxgboost = xgb.XGBClassifier() # Initialize Random Forest with the hyperparameters from the best performing model\ny_pred = xgboost.fit(X_train, y_train).predict(X_val)\n\nxgb_results = graph_classification_reports(y_pred, y_val, title='Logistic Regression Model Classification Reports')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:44:15.078290Z","iopub.execute_input":"2022-07-31T09:44:15.078652Z","iopub.status.idle":"2022-07-31T09:44:16.112142Z","shell.execute_reply.started":"2022-07-31T09:44:15.078620Z","shell.execute_reply":"2022-07-31T09:44:16.110553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_model_performance(parameters=n_estimators_xgb,\n                       title='Training and Testing Model Performances',\n                       perfs=perfs,\n                       x_labels='XGBoost Classifier Model')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:44:16.113537Z","iopub.execute_input":"2022-07-31T09:44:16.114115Z","iopub.status.idle":"2022-07-31T09:44:16.478410Z","shell.execute_reply.started":"2022-07-31T09:44:16.114076Z","shell.execute_reply":"2022-07-31T09:44:16.477587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine model results into a DataFrame\nall_model_results = pd.DataFrame({\"K-Nearest Neighbors\": knn_results,\n                                  \"Logistic Regression\": log_reg_results,\n                                  \"Random Forest Classifier\": rf_results,\n                                  \"XGBoost Classifier\": xgb_results})\nall_model_results = all_model_results.T\nall_model_results","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:44:16.479489Z","iopub.execute_input":"2022-07-31T09:44:16.480205Z","iopub.status.idle":"2022-07-31T09:44:16.495452Z","shell.execute_reply.started":"2022-07-31T09:44:16.480170Z","shell.execute_reply":"2022-07-31T09:44:16.494111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reduce the accuracy to the same scale as the other metrics \nall_model_results[\"accuracy\"] = all_model_results[\"accuracy\"]/100\nall_model_results","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:44:16.497164Z","iopub.execute_input":"2022-07-31T09:44:16.497508Z","iopub.status.idle":"2022-07-31T09:44:16.514889Z","shell.execute_reply.started":"2022-07-31T09:44:16.497478Z","shell.execute_reply":"2022-07-31T09:44:16.513846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot and compare all of the model results\nsns.set(style = 'whitegrid',\n            rc = {'figure.figsize': (20, 10)})\nsns.set_palette('Reds_r')\nall_model_results.plot(kind=\"bar\", figsize=(10, 7)).legend(bbox_to_anchor=(1.0, 1.0))\n\nplt.xlabel('\\nModels', fontsize=15, fontweight='bold')\nplt.ylabel('Metric Values', fontsize=15, fontweight='bold');","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:44:16.516419Z","iopub.execute_input":"2022-07-31T09:44:16.516820Z","iopub.status.idle":"2022-07-31T09:44:16.826872Z","shell.execute_reply.started":"2022-07-31T09:44:16.516787Z","shell.execute_reply":"2022-07-31T09:44:16.825574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}