{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\nThis notebook holds the best model I've devised.\nCurrent metrics include:\n\n**Cross Validation Scores:**\n\nmean_roc_auc_CV = 0.885 \\\nmean_error_rate_CV = 0.201 \\\nmean_precision_CV = 0.793 \\\nmean_recall_CV = 0.812 \\\n\n\n**Private Validation on Unseen Test Data:** \n\nValidation Set Precision: 0.83 \\\nValidation Set Recall: 0.83 \\\nValidation Set Accuracy: 0.83 \\\nValidation Set Average Precision Score: 0.9 \\\nValidation Set Roc Auc Score: 0.9\n\n**Kaggle Competition ROCAUC Leaderboard Scores:**\n\nPrivate: 0.8531\nPublic: 0.8702\n","metadata":{}},{"cell_type":"markdown","source":"Notebook Contents:\n\n1. Model scores\n2. Model file names\n3. Setting up environment\n4. Preparing data for training, testing and validation\n5. Modeling and predictions\n6. Model Evaluation","metadata":{}},{"cell_type":"markdown","source":"## Import Libraries","metadata":{}},{"cell_type":"code","source":"#hide output\n!pip install -Uqq fastbook --q --q\n!pip install ipywidgets","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","scrolled":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-10-17T21:25:20.190284Z","iopub.execute_input":"2022-10-17T21:25:20.190778Z","iopub.status.idle":"2022-10-17T21:25:51.675280Z","shell.execute_reply.started":"2022-10-17T21:25:20.190740Z","shell.execute_reply":"2022-10-17T21:25:51.673904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport os\nimport imageio.v2 as imageio\n\nsns.set_style('darkgrid')\nplt.style.use('seaborn-notebook')\n\nimport fastbook\nfastbook.setup_book()\nimport fastai\n\nimport random\nrandom.seed(42)\n\nfrom fastbook import *\nfrom fastai.vision.all import *\nimport torch\nfrom pathlib import Path\nfrom PIL import Image\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import *\nimport statistics","metadata":{"execution":{"iopub.status.busy":"2022-10-17T21:25:51.681574Z","iopub.execute_input":"2022-10-17T21:25:51.684045Z","iopub.status.idle":"2022-10-17T21:25:51.703825Z","shell.execute_reply.started":"2022-10-17T21:25:51.683996Z","shell.execute_reply":"2022-10-17T21:25:51.702711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#set device and ensure GPU is running\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\ndevice","metadata":{"execution":{"iopub.status.busy":"2022-10-17T21:25:51.709232Z","iopub.execute_input":"2022-10-17T21:25:51.711629Z","iopub.status.idle":"2022-10-17T21:25:51.723332Z","shell.execute_reply.started":"2022-10-17T21:25:51.711589Z","shell.execute_reply":"2022-10-17T21:25:51.722052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Preparation\n1. Load metadata csv files, \n2. select all the malignant images, then downsample the benign category in remaining training data so n_malignant = n_benign for training, \n3. split off 20% of training data as a validation set, \n4. then create a 5-fold split of what remains for cross-validation in model training","metadata":{}},{"cell_type":"code","source":"#1. Load metadata .csv files\n\n#read in train csv\ntrain = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\n\n#read in test csv\ntest = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-10-17T21:25:51.729642Z","iopub.execute_input":"2022-10-17T21:25:51.731938Z","iopub.status.idle":"2022-10-17T21:25:51.846603Z","shell.execute_reply.started":"2022-10-17T21:25:51.731887Z","shell.execute_reply":"2022-10-17T21:25:51.845512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Step 2: select all the malignant images, then downsample the benign category in remaining training data so n_malignant = n_benign for training\n\n#split train_ims into 5 sets using StratifiedKFold from sklearn\n#create a df with ALL of the malignant images\nmal_ims = train[train['target']==1]\nn_mal_ims = len(mal_ims)\nprint(\"Number of malignant images: {}\".format(n_mal_ims))\nmal_ims.head()\n\n#create a df of a subset of benevolent images\nben_ims_subset = train[train['target']==0].sample(n=n_mal_ims, random_state=42)\nn_ben_ims = len(ben_ims_subset)\nprint(\"Number of benign images in subset: {}\".format(n_ben_ims))\n\n#concatenate the two together and check the result\ntrain_ims = pd.concat([mal_ims, ben_ims_subset])\nn_training_items = len(train_ims)\nprint(\"Number of training items: {}\".format(n_training_items))\n\n#add /train prefix and /test prefix to the respective dfs\ntrain_ims['image_name'] = 'train/' + train_ims['image_name'].astype(str)\ntest['image_name'] = 'test/' + test['image_name'].astype(str)\n\ntrain_ims.reset_index(drop=True, inplace=True)\ntrain_ims.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-17T21:25:51.851598Z","iopub.execute_input":"2022-10-17T21:25:51.853938Z","iopub.status.idle":"2022-10-17T21:25:51.901544Z","shell.execute_reply.started":"2022-10-17T21:25:51.853898Z","shell.execute_reply":"2022-10-17T21:25:51.900569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Step 3: split off 20% of training data as a validation set,\n\nskf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\nX = train_ims['image_name'].copy()\ny = train_ims['target'].copy()\nfold = 0\nfor train_index, test_index in skf.split(X, y):\n    fold+= 1\n    print('In fold',fold)\n    print(\"TRAIN LENGTH:\", len(train_index), \"VALIDATION LENGTH:\", len(test_index))\n    train_ims[f'fold_{fold}_valid']=False\n    train_ims.loc[test_index,f'fold_{fold}_valid']=True\n    \n\n#Pull out 20% of entries to be held out as unseen test data for validation.\nvalid_ims = train_ims[train_ims['fold_5_valid'] == True]\n\nvalid_ims.reset_index(drop=True, inplace=True)\nvalid_ims.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-10-17T21:25:51.905921Z","iopub.execute_input":"2022-10-17T21:25:51.908448Z","iopub.status.idle":"2022-10-17T21:25:51.952318Z","shell.execute_reply.started":"2022-10-17T21:25:51.908398Z","shell.execute_reply":"2022-10-17T21:25:51.951265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Step 4: create a 5-fold split of what remains for cross-validation in model training\n\n#drop the test set from train_ims, then split the remaining images into 5 folds for cross-validation during model training\ntrain_ims = train_ims[train_ims['fold_5_valid'] == False]\ntrain_ims = train_ims[['image_name','patient_id','sex','age_approx','anatom_site_general_challenge','diagnosis','benign_malignant','target']]\ntrain_ims.reset_index(drop=True, inplace=True)\n\nskf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\nX = train_ims['image_name'].copy()\ny = train_ims['target'].copy()\nfold = 0\nfor train_index, test_index in skf.split(X, y):\n    fold+= 1\n    print('In fold',fold)\n    print(\"TRAIN LENGTH:\", len(train_index), \"VALIDATION LENGTH:\", len(test_index))\n    train_ims[f'fold_{fold}_valid']=False\n    train_ims.loc[test_index,f'fold_{fold}_valid']=True\n    \ntrain_ims.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-17T21:25:51.956826Z","iopub.execute_input":"2022-10-17T21:25:51.959248Z","iopub.status.idle":"2022-10-17T21:25:52.005884Z","shell.execute_reply.started":"2022-10-17T21:25:51.959208Z","shell.execute_reply":"2022-10-17T21:25:52.004935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check for any overlap between the train_ims df and valid_ims df\n\nidx1 = pd.Index(train_ims['image_name'])\nidx2 = pd.Index(valid_ims['image_name'])\nidx1.intersection(idx2)","metadata":{"execution":{"iopub.status.busy":"2022-10-17T21:25:52.010271Z","iopub.execute_input":"2022-10-17T21:25:52.012588Z","iopub.status.idle":"2022-10-17T21:25:52.026453Z","shell.execute_reply.started":"2022-10-17T21:25:52.012546Z","shell.execute_reply":"2022-10-17T21:25:52.025287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Set up dataloader, CV training loop, and store results","metadata":{}},{"cell_type":"code","source":"def dataloader(fold, bs=8, b_tfms=Normalize.from_stats(*imagenet_stats)):\n    dls = ImageDataLoaders.from_df(df = train_ims, #specify df holding image names\n                                   path = '../input/siic-isic-224x224-images',    #set path for where to find images\n                                   suff = '.png',      #add the .png suffix to file names from df\n                                   label_col = 'target',      \n                                   bs = bs,            #set batch size\n                                   device=device,      #set device\n                                   batch_tfms = b_tfms,\n                                   valid_col = f'fold_{fold}_valid')\n    return dls\n    ","metadata":{"execution":{"iopub.status.busy":"2022-10-17T21:25:52.030663Z","iopub.execute_input":"2022-10-17T21:25:52.032923Z","iopub.status.idle":"2022-10-17T21:25:52.043174Z","shell.execute_reply.started":"2022-10-17T21:25:52.032885Z","shell.execute_reply":"2022-10-17T21:25:52.042153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#instantiate metrics\nrocAucBinary = RocAucBinary()\nrecall = Recall()\nprecision = Precision()\n\n#instantiate arrays to hold probability results\nvalid_preds = np.zeros((valid_ims.shape[0],2))\nkaggle_preds = np.zeros((test.shape[0],2))\n\n#choose batch transforms, batch size, and the name of \nb_tfms = [#*aug_transforms(do_flip=True, flip_vert=True, max_rotate=45.0, max_zoom=1.1, size=224, max_lighting=0.2, max_warp=0.4, p_affine=0.75, p_lighting=0.75, xtra_tfms=None, mode='bilinear'),\n          Normalize.from_stats(*imagenet_stats)]\nbatch_size=8\n\nfor fold in range(1,6):\n    dls=dataloader(fold, batch_size, b_tfms)\n    print(f'Fold {fold}:')\n    learn = vision_learner(dls,                  #specify dataloader object\n                       models.resnet34,             #specify a pre-trained model we want to build off of\n                       metrics=[rocAucBinary, error_rate, precision, recall], #specify metrics we want to see\n                       model_dir = '/kaggle/working')   #specify output location to store model\n    \n    learn.fine_tune(15,                               #set number of epochs \n                 #base_lr=valley,                  #set the initial learning rate\n                 cbs=[SaveModelCallback(            #use the SaveModelCallback to save the best model\n                     monitor='roc_auc_score',      #set the roc_auc_score as the montitored metric\n                     fname = f'resnet34_rocauc_fold{fold}',       #choose the name the best model will be saved under\n                     comp = np.greater,            #specify that when the roc_auc_score increases, that's considered better\n                     with_opt=True),               #saves optimizer state, if available, when saving model\n                    ReduceLROnPlateau(monitor = 'roc_auc_score', comp = np.greater, patience=2)])\n                   # EarlyStoppingCallback(monitor = 'roc_auc_score', comp=np.greater, patience=6)])\n    \n    learn.load(f'resnet34_rocauc_fold{fold}', \n               device=device,     #ensure the loaded model uses the active cuda:0 device\n               with_opt=True,     #load optimizer state\n               strict=True)\n    \n    test_dl = learn.dls.test_dl(valid_ims)\n    preds, _ = learn.tta(dl=test_dl) \n    valid_preds += preds.numpy()\n    \n    kag_dl = learn.dls.test_dl(test)\n    preds, _ = learn.get_preds(dl=kag_dl) \n    kaggle_preds += preds.numpy()\n    \n    print(f'Prediction completed in fold: {fold}')\n\n    fold+=1\n\nkaggle_pred = kaggle_preds/5\nvalid_preds = valid_preds/5","metadata":{"execution":{"iopub.status.busy":"2022-10-17T21:25:52.050280Z","iopub.execute_input":"2022-10-17T21:25:52.052639Z","iopub.status.idle":"2022-10-17T21:45:34.479725Z","shell.execute_reply.started":"2022-10-17T21:25:52.052600Z","shell.execute_reply":"2022-10-17T21:45:34.478566Z"},"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fold_1_metrics=[0.893274,0.203209,0.785714,0.819149]\nfold_2_metrics=[0.887440,0.213904,0.813953,0.744681]\nfold_3_metrics=[0.879661,0.187166,0.775701,0.882979]\nfold_4_metrics=[0.882864,0.208556,0.787234,0.795699]\nfold_5_metrics=[0.881720,0.192513,0.800000,0.817204]\n\nmean_roc_auc = statistics.mean([fold_1_metrics[0],fold_2_metrics[0],fold_3_metrics[0],fold_4_metrics[0],fold_5_metrics[0]])\nmean_error = statistics.mean([fold_1_metrics[1],fold_2_metrics[1],fold_3_metrics[1],fold_4_metrics[1],fold_5_metrics[1]])\nmean_precision = statistics.mean([fold_1_metrics[2],fold_2_metrics[2],fold_3_metrics[2],fold_4_metrics[2],fold_5_metrics[2]])\nmean_recall = statistics.mean([fold_1_metrics[3],fold_2_metrics[3],fold_3_metrics[3],fold_4_metrics[3],fold_5_metrics[3]])\n\nprint(f'mean_roc_auc_CV = {np.round(mean_roc_auc,3)}')\nprint(f'mean_error_rate_CV = {np.round(mean_error,3)}')\nprint(f'mean_precision_CV = {np.round(mean_precision,3)}')\nprint(f'mean_recall_CV = {np.round(mean_recall,3)}')\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-10-17T22:14:20.945933Z","iopub.execute_input":"2022-10-17T22:14:20.946303Z","iopub.status.idle":"2022-10-17T22:14:20.957376Z","shell.execute_reply.started":"2022-10-17T22:14:20.946272Z","shell.execute_reply":"2022-10-17T22:14:20.956391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Interpret Results","metadata":{}},{"cell_type":"code","source":"ss = pd.read_csv('../input/siim-isic-melanoma-classification/sample_submission.csv')\n\n#Save kaggle predictions to csv for submission\nsub_file_name=\"submission_resnet34_Oct172022_rocauc_opt.csv\"\nsubmission = pd.DataFrame({'image_name':ss['image_name'], 'target':list(kaggle_preds[:,1])})\nsubmission.to_csv(sub_file_name, index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-17T21:45:34.507647Z","iopub.execute_input":"2022-10-17T21:45:34.509990Z","iopub.status.idle":"2022-10-17T21:45:34.576441Z","shell.execute_reply.started":"2022-10-17T21:45:34.509953Z","shell.execute_reply":"2022-10-17T21:45:34.575496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Kaggle Submission ROC_AUC_scores:\n**Private:** 0.8531\n\n**Public:** 0.8702\n\n\n\n\n","metadata":{}},{"cell_type":"code","source":"#transfer averaged CV results into a DataFrame for visualization and to calculate the relevent metrics\ntest_res = pd.DataFrame({'image_name':valid_ims['image_name'], 'true_label':valid_ims['target'] ,'target_preds':list(valid_preds[:,1])})\n\n#assign class predictions based on a prediction threshold of 0.5\ntest_res['pred_label'] = np.where(test_res['target_preds'] >= .5, 1, 0)\n","metadata":{"execution":{"iopub.status.busy":"2022-10-17T21:45:34.580628Z","iopub.execute_input":"2022-10-17T21:45:34.582886Z","iopub.status.idle":"2022-10-17T21:45:34.592740Z","shell.execute_reply.started":"2022-10-17T21:45:34.582849Z","shell.execute_reply":"2022-10-17T21:45:34.591686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#visualize ROC for validation data\nfpr = dict()\ntpr = dict()\nroc_auc = dict()\nfpr, tpr, _ = roc_curve(test_res['true_label'], test_res['target_preds'])\nroc_auc = auc(fpr, tpr)\n\n#plot Reciever Operating Characteristic\nplt.figure()\nlw = 2\nplt.plot(\n    fpr,\n    tpr,\n    color=\"darkorange\",\n    lw=lw,\n    label=\"ROC curve (area = %0.2f)\" % roc_auc,\n)\nplt.plot([0, 1], [0, 1], color=\"navy\", lw=lw, linestyle=\"--\")\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel(\"False Positive Rate\")\nplt.ylabel(\"True Positive Rate\")\nplt.title(\"Receiver operating characteristic\")\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-17T22:16:15.809968Z","iopub.execute_input":"2022-10-17T22:16:15.810823Z","iopub.status.idle":"2022-10-17T22:16:16.049270Z","shell.execute_reply.started":"2022-10-17T22:16:15.810786Z","shell.execute_reply":"2022-10-17T22:16:16.048330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Generate and Plot Precision Recall Curve\nfrom sklearn.metrics import PrecisionRecallDisplay\nax1 = PrecisionRecallDisplay.from_predictions(test_res['true_label'], test_res['target_preds'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-17T22:16:17.630093Z","iopub.execute_input":"2022-10-17T22:16:17.630815Z","iopub.status.idle":"2022-10-17T22:16:17.864959Z","shell.execute_reply.started":"2022-10-17T22:16:17.630778Z","shell.execute_reply":"2022-10-17T22:16:17.864024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_set_precision = precision_score(test_res['true_label'], test_res['pred_label'])\ntest_set_recall = recall_score(test_res['true_label'], test_res['pred_label'])\ntest_set_accuracy = accuracy_score(test_res['true_label'], test_res['pred_label'])\ntest_set_average_precision = average_precision_score(test_res['true_label'], test_res['target_preds'])\ntest_set_auroc = roc_auc_score(test_res['true_label'], test_res['target_preds'])\n\nprint(f'Validation Set Precision: {np.round(test_set_precision,2)}')\nprint(f'Validation Set Recall: {np.round(test_set_recall,2)}')\nprint(f'Validation Set Accuracy: {np.round(test_set_accuracy,2)}')\n\nprint(f'Validation Set Average Precision Score: {np.round(test_set_average_precision,2)}')\nprint(f'Validation Set Roc Auc Score: {np.round(test_set_auroc,2)}')\n\nprint(classification_report(test_res['true_label'], test_res['pred_label']))\n    \nConfusionMatrixDisplay.from_predictions(test_res['true_label'], test_res['pred_label'], normalize='true')","metadata":{"execution":{"iopub.status.busy":"2022-10-17T22:16:24.632507Z","iopub.execute_input":"2022-10-17T22:16:24.632863Z","iopub.status.idle":"2022-10-17T22:16:24.900542Z","shell.execute_reply.started":"2022-10-17T22:16:24.632831Z","shell.execute_reply":"2022-10-17T22:16:24.899679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Export best models for future deployment","metadata":{}},{"cell_type":"code","source":"for fold in range(1,6):\n    learn.load(f'resnet34_rocauc_fold{fold}', \n               device=device,     #ensure the loaded model uses the active cuda:0 device\n               with_opt=True,     #load optimizer state\n               strict=True)\n    learn.export(f'/kaggle/working/melanoma_detector_fold{fold}.pkl')","metadata":{"execution":{"iopub.status.busy":"2022-10-17T22:26:34.072045Z","iopub.execute_input":"2022-10-17T22:26:34.072833Z","iopub.status.idle":"2022-10-17T22:26:36.180488Z","shell.execute_reply.started":"2022-10-17T22:26:34.072794Z","shell.execute_reply":"2022-10-17T22:26:36.179482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}