{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Part 1.0: Modelling/Training for our Feature Extract Data [(From Other Notebook)](https://www.kaggle.com/danielbozinovski/feature-extraction-for-pneumonia)","metadata":{}},{"cell_type":"code","source":"# Imports\nimport os\nimport cv2\nimport glob\nimport time\nimport pydicom\nimport skimage\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nfrom skimage import feature, filters\n%matplotlib inline\n\nfrom functools import partial\nfrom collections import defaultdict\nfrom joblib import Parallel, delayed\nfrom lightgbm import LGBMClassifier\nfrom tqdm import tqdm\n\n# sklearn\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import cross_val_score\n\n# Models\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.ensemble import GradientBoostingClassifier\n\nsns.set_style('whitegrid')\nnp.warnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:30:46.991197Z","iopub.execute_input":"2021-07-31T11:30:46.991630Z","iopub.status.idle":"2021-07-31T11:30:50.877745Z","shell.execute_reply.started":"2021-07-31T11:30:46.991542Z","shell.execute_reply":"2021-07-31T11:30:50.876699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imageFeaturesPath = \"../input/extractedfeatures/dicomImageFeatures.csv\"\ntestImageFeaturesPath = \"../input/extractedfeatures/testImageFeatures.csv\"\nlabelsPath = \"../input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv\"\n\nimageFeatures = pd.read_csv(imageFeaturesPath)\ntestImageFeatures = pd.read_csv(testImageFeaturesPath)\nlabels = pd.read_csv(labelsPath)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:31:01.799036Z","iopub.execute_input":"2021-07-31T11:31:01.799552Z","iopub.status.idle":"2021-07-31T11:31:02.173194Z","shell.execute_reply.started":"2021-07-31T11:31:01.799507Z","shell.execute_reply":"2021-07-31T11:31:02.172178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Imports\nfrom sklearn.model_selection import train_test_split\nimport sklearn.metrics as skm","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:31:04.278977Z","iopub.execute_input":"2021-07-31T11:31:04.279380Z","iopub.status.idle":"2021-07-31T11:31:04.283236Z","shell.execute_reply.started":"2021-07-31T11:31:04.279342Z","shell.execute_reply":"2021-07-31T11:31:04.282444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imageFeatures.head(2)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:31:06.057265Z","iopub.execute_input":"2021-07-31T11:31:06.057619Z","iopub.status.idle":"2021-07-31T11:31:06.084434Z","shell.execute_reply.started":"2021-07-31T11:31:06.057592Z","shell.execute_reply":"2021-07-31T11:31:06.083471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Part 1.1: Get Features into their own Dataframe","metadata":{}},{"cell_type":"code","source":"def getFeaturesDF(imgFeatures):\n    \n    features = imgFeatures.features.apply(lambda x: list(eval(x)))\n\n    df = pd.DataFrame(features.values.tolist(), \n                            columns=['mean', 'stddev', 'area', 'perimeter', 'irregularity', 'equiv_diam', 'hu1', 'hu2', 'hu4', 'hu5', 'hu6'],\n                            index=imgFeatures.index)\n\n    df['hasPneumonia'] = labels['Target']\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:31:08.633074Z","iopub.execute_input":"2021-07-31T11:31:08.633704Z","iopub.status.idle":"2021-07-31T11:31:08.642542Z","shell.execute_reply.started":"2021-07-31T11:31:08.633648Z","shell.execute_reply":"2021-07-31T11:31:08.641028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get train and test features dataframes\ntrainData = getFeaturesDF(imageFeatures)\ntestData = getFeaturesDF(testImageFeatures)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:31:10.311835Z","iopub.execute_input":"2021-07-31T11:31:10.312185Z","iopub.status.idle":"2021-07-31T11:31:11.257991Z","shell.execute_reply.started":"2021-07-31T11:31:10.312155Z","shell.execute_reply":"2021-07-31T11:31:11.257163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Part 1.2: Get Class Weights","metadata":{}},{"cell_type":"code","source":"COUNT_NORMAL = len(trainData.loc[trainData['hasPneumonia'] == 0])\nCOUNT_PNE = len(trainData.loc[trainData['hasPneumonia'] == 1])\nTRAIN_IMG_COUNT = len(trainData)\n\nweight_for_0 = (1 / COUNT_NORMAL)*(TRAIN_IMG_COUNT)/2.0 \nweight_for_1 = (1 / COUNT_PNE)*(TRAIN_IMG_COUNT)/2.0\n\nclassWeight = {0: weight_for_0, 1: weight_for_1}\n\nprint('Weight for class 0: {:.2f}'.format(weight_for_0))\nprint('Weight for class 1: {:.2f}'.format(weight_for_1))","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:31:42.824761Z","iopub.execute_input":"2021-07-31T11:31:42.825414Z","iopub.status.idle":"2021-07-31T11:31:42.840611Z","shell.execute_reply.started":"2021-07-31T11:31:42.825378Z","shell.execute_reply":"2021-07-31T11:31:42.839560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Part 1.3: Normalise the Data","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\n# For Training Data\ntrainData.dropna()\n\n# Split data into x and y\nx = trainData.drop(columns=['hasPneumonia'])\ny = trainData['hasPneumonia']\n\n# Scale the features\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(x)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:32:42.293825Z","iopub.execute_input":"2021-07-31T11:32:42.294468Z","iopub.status.idle":"2021-07-31T11:32:42.325500Z","shell.execute_reply.started":"2021-07-31T11:32:42.294434Z","shell.execute_reply":"2021-07-31T11:32:42.324707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For test data\ntestData.dropna()\n\n# Split data into x and y\nx_test_unseen = testData.drop(columns=['hasPneumonia'])\ny_test_unseen = testData['hasPneumonia']\n\n# Scale the features\nscaler = StandardScaler()\nX_scaled_test_unseen = scaler.fit_transform(x_test_unseen)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:32:43.963195Z","iopub.execute_input":"2021-07-31T11:32:43.963683Z","iopub.status.idle":"2021-07-31T11:32:43.978471Z","shell.execute_reply.started":"2021-07-31T11:32:43.963651Z","shell.execute_reply":"2021-07-31T11:32:43.977628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Part 1.4: Split into Training and Testing Data","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(\n                                        X_scaled, \n                                        y,\n                                        stratify=y,\n                                        shuffle=True, \n                                        test_size = 0.3)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:32:46.751245Z","iopub.execute_input":"2021-07-31T11:32:46.751777Z","iopub.status.idle":"2021-07-31T11:32:46.782973Z","shell.execute_reply.started":"2021-07-31T11:32:46.751745Z","shell.execute_reply":"2021-07-31T11:32:46.782142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create function to print our scoring metrics\ndef printScores(pred_y_test, y_test, pred_y_train, y_train):\n    \n    print(\"===== Training Metrics =====\")\n    print(f\"Accuracy: {round(skm.accuracy_score(y_train, pred_y_train)*100, 3)}%\")\n    print(f\"Precision: {round(skm.precision_score(y_train, pred_y_train)*100, 3)}%\")\n    print(f\"Recall: {round(skm.recall_score(y_train, pred_y_train)*100, 3)}%\")\n    print(f\"MSE: {round(skm.mean_squared_error(y_train, pred_y_train)*100, 3)}%\")\n    print(f\"Area Under Curve: {round(skm.roc_auc_score(y_train, pred_y_train)*100, 3)}%\")\n    \n    print(\"\\n===== Testing Metrics =====\")\n    print(f\"Accuracy: {round(skm.accuracy_score(y_test, pred_y_test)*100, 3)}%\")\n    print(f\"Precision: {round(skm.precision_score(y_test, pred_y_test)*100, 3)}%\")\n    print(f\"Recall: {round(skm.recall_score(y_test, pred_y_test)*100, 3)}%\")\n    print(f\"MSE: {round(skm.mean_squared_error(y_test, pred_y_test)*100, 3)}%\")\n    print(f\"Area Under Curve: {round(skm.roc_auc_score(y_test, pred_y_test)*100, 3)}%\")","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:32:50.851495Z","iopub.execute_input":"2021-07-31T11:32:50.852019Z","iopub.status.idle":"2021-07-31T11:32:50.857981Z","shell.execute_reply.started":"2021-07-31T11:32:50.851987Z","shell.execute_reply":"2021-07-31T11:32:50.857129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import RandomizedSearchCV","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:32:53.233150Z","iopub.execute_input":"2021-07-31T11:32:53.233695Z","iopub.status.idle":"2021-07-31T11:32:53.238135Z","shell.execute_reply.started":"2021-07-31T11:32:53.233659Z","shell.execute_reply":"2021-07-31T11:32:53.237051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Part 1.5: Perform RandomisedGridSearch For Optimal Hyper-paramters","metadata":{}},{"cell_type":"markdown","source":"### Model 1: Logistic Regression","metadata":{}},{"cell_type":"code","source":"# Performing randomised Grid Search\ncVals = list(range(1, 6))\ncWeight = [None, 'Balanced']\n\nparams = dict(C = cVals, class_weight=cWeight)\n\nlogReg = LogisticRegression()\nclf = RandomizedSearchCV(logReg, params, random_state=0)\n\nsearch = clf.fit(X_train, y_train)\nsearch.best_params_ # Return the best hyper-parameters","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:33:00.843842Z","iopub.execute_input":"2021-07-31T11:33:00.844385Z","iopub.status.idle":"2021-07-31T11:33:07.796549Z","shell.execute_reply.started":"2021-07-31T11:33:00.844352Z","shell.execute_reply":"2021-07-31T11:33:07.795572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model 2: kNN","metadata":{}},{"cell_type":"code","source":"# Performing randomised Grid Search\nkValues = list(range(10, 210, 10))\nweight_options = [\"uniform\", \"distance\"]\n\nparams = dict(n_neighbors = kValues, weights = weight_options)\n\nkNN = KNeighborsClassifier()\nclf = RandomizedSearchCV(kNN, params, random_state=0)\n\nsearch = clf.fit(X_train, y_train)\nsearch.best_params_ # Return the best hyper-parameters","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:33:07.798327Z","iopub.execute_input":"2021-07-31T11:33:07.798921Z","iopub.status.idle":"2021-07-31T11:33:55.525047Z","shell.execute_reply.started":"2021-07-31T11:33:07.798882Z","shell.execute_reply":"2021-07-31T11:33:55.523848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model 3: Naive Bayes","metadata":{}},{"cell_type":"code","source":"# Did not run randomised grid search since lack of hyper-parameters\ngnb = GaussianNB()\ngnb.fit(X_train, y_train)\n\npred_y_test = gnb.predict(X_test)\npred_y_train = gnb.predict(X_train)\n\nprintScores(pred_y_test, y_test, pred_y_train, y_train) # Scoring function","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:35:11.090148Z","iopub.execute_input":"2021-07-31T11:35:11.090912Z","iopub.status.idle":"2021-07-31T11:35:11.160312Z","shell.execute_reply.started":"2021-07-31T11:35:11.090869Z","shell.execute_reply":"2021-07-31T11:35:11.159182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model 4: Random Forest","metadata":{}},{"cell_type":"code","source":"# Perform Randomised Grid Search\nclassWeights = [None, 'Balanced']\nnEstimatorValues = list(range(300, 800, 100))\nmaxDepthValues = list(range(6, 10))\nminSamplesSplitValues = list(range(2, 5))\n\nparams = dict(n_estimators = nEstimatorValues, \n              max_depth = maxDepthValues, \n              class_weight = classWeights,\n              min_samples_split = minSamplesSplitValues)\n\nrfc = RandomForestClassifier(n_jobs=-1)\n\nclf = RandomizedSearchCV(rfc, params, random_state=0)\n\nsearch = clf.fit(X_train, y_train)\nsearch.best_params_ # Display best hyper-parameters","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:35:20.437632Z","iopub.execute_input":"2021-07-31T11:35:20.438100Z","iopub.status.idle":"2021-07-31T11:38:48.412314Z","shell.execute_reply.started":"2021-07-31T11:35:20.438052Z","shell.execute_reply":"2021-07-31T11:38:48.411303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model 5: Support Vector Machine","metadata":{}},{"cell_type":"code","source":"# Perform Randomised Grid Seach\ncVals = np.arange(0.5, 1.6, 0.1)\nclassWeights = [None, 'Balanced']\n\nparams = dict(C = cVals, class_weight = classWeights)\n\nsvm = SVC()\n\nclf = RandomizedSearchCV(svm, params, random_state=0)\nsearch = clf.fit(X_train, y_train)\nsearch.best_params_ # Return the best hyper-parametrs","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:38:48.414003Z","iopub.execute_input":"2021-07-31T11:38:48.414308Z","iopub.status.idle":"2021-07-31T11:44:04.781215Z","shell.execute_reply.started":"2021-07-31T11:38:48.414278Z","shell.execute_reply":"2021-07-31T11:44:04.780072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model 6: Gradient Boosted Classifier","metadata":{}},{"cell_type":"code","source":"# Perform Randomised Grid Search to Find Optimal Hyper-parameters\nnEstimatorValues = list(range(300, 800, 100))\nmaxDepthValues = list(range(6, 10))\nminSamplesSplitValues = list(range(2, 5))\nlrs = [0.0001, 0.001, 0.01, 0.1, 0.2, 0.3]\n\nparams = dict(n_estimators = nEstimatorValues, \n              max_depth = maxDepthValues, \n              learning_rate = lrs,\n              min_samples_split = minSamplesSplitValues)\n\n\ngbc = GradientBoostingClassifier()\n\nclf = RandomizedSearchCV(gbc, params, random_state=0)\nsearch = clf.fit(X_train, y_train)\nsearch.best_params_","metadata":{"execution":{"iopub.status.busy":"2021-07-29T05:32:36.096328Z","iopub.execute_input":"2021-07-29T05:32:36.096735Z","iopub.status.idle":"2021-07-29T05:32:36.10236Z","shell.execute_reply.started":"2021-07-29T05:32:36.096699Z","shell.execute_reply":"2021-07-29T05:32:36.101056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Part 1.6: Create Function to Perform K-Fold Cross Validation","metadata":{}},{"cell_type":"code","source":"# Function to perform K-fold cross val\ndef performCV(model, name, K):\n    \n    print(f\"===== Performing CV for {name} =====\")\n    kfold = KFold(n_splits = K, shuffle = True)\n    \n    accuracy_per_fold = []\n    precision_per_fold = []\n    recall_per_fold = []\n    mse_per_fold = []\n    auc_per_fold = []\n\n    for train_index, test_index in kfold.split(X_scaled):\n\n        X_train, X_test = X_scaled[train_index], X_scaled[test_index] # Split data\n        y_train, y_test = y[train_index], y[test_index]\n\n        model.fit(X_train, y_train)# Fit data\n        pred_y_test = model.predict(X_test) # Make a prediction\n        \n        accuracy = skm.accuracy_score(y_test, pred_y_test)\n        precision = skm.precision_score(y_test, pred_y_test)\n        recall = skm.recall_score(y_test, pred_y_test)\n        mse = skm.mean_squared_error(y_test, pred_y_test)\n        auc = skm.roc_auc_score(y_test, pred_y_test)\n\n        accuracy_per_fold.append(accuracy)\n        precision_per_fold.append(precision)\n        recall_per_fold.append(recall)\n        mse_per_fold.append(mse)\n        auc_per_fold.append(auc)\n        \n    \n    return {\n        'mean_accuracy': np.mean(accuracy_per_fold),\n        'mean_precision': np.mean(precision_per_fold),\n        'mean_recall': np.mean(recall_per_fold),\n        'mean_mse': np.mean(mse_per_fold),\n        'mean_auc': np.mean(auc_per_fold)\n    }\n    ","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:44:04.783491Z","iopub.execute_input":"2021-07-31T11:44:04.783895Z","iopub.status.idle":"2021-07-31T11:44:04.794683Z","shell.execute_reply.started":"2021-07-31T11:44:04.783852Z","shell.execute_reply":"2021-07-31T11:44:04.793696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Take models with most optimal hyper parameters\nlogReg = LogisticRegression(C = 1)\nkNN = KNeighborsClassifier(150, weights = \"distance\")\ngnb = GaussianNB()\nrfc = RandomForestClassifier(n_estimators = 700, max_depth = 9, min_samples_split = 4, n_jobs = -1)\nsvm = SVC(C = 1.5)\ngbc = GradientBoostingClassifier(n_estimators = 700, max_depth = 9, min_samples_split = 4, learning_rate=0.005)\n\nmodelsList = [(logReg, \"Logistic Regression\"), \n             (kNN, \"K-Nearest Neighbour\"),\n             (gnb, \"Naive Bayes\"),\n             (rfc, \"Random Forest\"),\n             (svm, \"Support Vector Machine\"),\n             (gbc, \"Gradient Boosting Classifier\")]\n\nCVResults = {}\n\nfor m in modelsList:\n    CVResults[m[1]] = performCV(m[0], m[1], 5)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T11:44:04.796292Z","iopub.execute_input":"2021-07-31T11:44:04.796885Z","iopub.status.idle":"2021-07-31T12:05:08.367964Z","shell.execute_reply.started":"2021-07-31T11:44:04.796831Z","shell.execute_reply":"2021-07-31T12:05:08.366378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Note if these aren't the same in the report, they were re-run\nCVResults # Display the Cross Validation results for each model","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:05:08.370408Z","iopub.execute_input":"2021-07-31T12:05:08.371002Z","iopub.status.idle":"2021-07-31T12:05:08.380837Z","shell.execute_reply.started":"2021-07-31T12:05:08.370958Z","shell.execute_reply":"2021-07-31T12:05:08.379522Z"},"trusted":true},"execution_count":null,"outputs":[]}]}