{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Histopathological Cancer Detection ","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"## RoadMap\n- Import Libraries\n- Check GPU\n- EDA\n- Model Building\n- Model Improvement\n- Model Validation\n- Submission","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nfrom fastai.vision import *\nimport fastai\nfrom fastai.metrics import *\nfrom fastai import *\nfrom os import *\nimport seaborn as sns\nfrom sklearn.metrics import auc,roc_curve,accuracy_score, roc_auc_score\nimport matplotlib.patches as patches\nimport matplotlib.pyplot as plt\nimport random\nnp.random.seed(42)\nfrom glob import glob \n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"model_path='.'\npath='/kaggle/input/histopathologic-cancer-detection/'\ntrain_folder=f'{path}train'\ntest_folder=f'{path}test'\ntrain_lbl=f'{path}train_labels.csv'\n\nbs=64\nnum_workers=None \nsz=96","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Check GPU ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Programming framework behind the scenes of NVIDIA GPU is CUDA\nprint(torch.cuda.is_available())\n# Check if gpu is enabled\nprint(torch.backends.cudnn.enabled)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# EDA","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train = pd.read_csv(train_lbl)\nprint(f'Number of labels {len(df_train)}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Proportion of classes \ndf_train['label'].value_counts(normalize=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(x='label',data=df_train)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Analyze cancer and non-cancer cell","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"cancer_cell = df_train[df_train['label']==1].head()\ncancer_cell","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"non_cancer_cell = df_train[df_train['label']==0].head()\nnon_cancer_cell","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.subplot(1 , 2 , 1)\nimg = np.asarray(plt.imread(train_folder+'/'+cancer_cell.iloc[1][0]+'.tif'))\nplt.title('METASTATIC CELL TISSUE')\nplt.imshow(img)\n\nplt.subplot(1 , 2 , 2)\nimg = np.asarray(plt.imread(train_folder+'/'+ non_cancer_cell.iloc[1][0]+'.tif'))\nplt.title('NON-METASTATIC CELL TISSUE')\nplt.imshow(img)\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"list = os.listdir(test_folder) # dir is your directory path\nlen(list)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"list = os.listdir(train_folder) # dir is your directory path\nlen(list)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Model Building\n- Model Training requires the objects of DataBunch and Learner","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"### Data Augmentation","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"tfms = get_transforms(do_flip=True, flip_vert=True, max_rotate=.0, max_zoom=1.1,max_lighting=0.05, max_warp=0.)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Create DataBunch object","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"data = ImageDataBunch.from_csv(path,folder='train',valid_pct=0.3,csv_labels=train_lbl,ds_tfms=tfms, size=90, suffix='.tif',test=test_folder,bs=64)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.classes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(data.c, len(data.train_ds), len(data.valid_ds))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Data Preprocesing (Normalization)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"stats=data.batch_stats()        \ndata.normalize(stats)\n#data.normalize(imagenet_stats)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#See the classes and labels\ndata.show_batch(rows=3, figsize=(8,5))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Model Training\n- Model Training requires DataBunch object and learner","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"model_dir = \"/kaggle/working/tmp/models/\"\nos.makedirs('/kaggle/working/tmp/models/')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"#### Create Learning Classifier","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"#fastai comes with various models\ndir(fastai.vision.models)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#create learner object by passing data bunch, specifying model architecture and metrics to use to evaluate training stats\nlearner_resnet50 = cnn_learner(data=data, base_arch=models.resnet50,model_dir=model_dir, metrics=[accuracy,error_rate], ps=0.5) #densenet201","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Classifier Training","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"lr_find(learner_resnet50)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learner_resnet50.recorder.plot()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"defaults.device = torch.device('cuda') # makes sure the gpu is used","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Transfer learning\n- Allows you to train nets with 1/100th less time using 1/100 less data.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"learner_resnet50.fit_one_cycle(1, 1e-02)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learner_resnet50.recorder.plot(return_fig=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#See how the learning rate and momentum varies with the training and losses\nlearner_resnet50.recorder.plot_lr(show_moms=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learner_resnet50.recorder.plot_losses(show_grid=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learner_resnet50.show_results(alpha=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#save weights in a file\nlearner_resnet50.save('stage-1',return_path=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Model Improvement\n- Generally, when you call fit_one_cycle it only trains the last or last few layers. To improve this better, you need to call learn.unfreeze() to unfreeze the model and train it again.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"#Unfreeze the encoder resnet\nlearner_resnet50.unfreeze()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lr_find(learner_resnet50)\nlearner_resnet50.recorder.plot()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#slice suggests is, train the initial layers at start value specified and last layer at the end value specified and interpolate for the rest of the layers\nlearner_resnet50.fit_one_cycle(1,slice(1e-06,1e-05),pct_start=0.8)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learner_resnet50.recorder.plot_losses()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learner_resnet50.save('stage-2',return_path=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Model Validation\n    - Plot top losses images\n    - Confusion Matrix\n    - Validate across validation set by auc_score and accuracy\n    - Plot roc_curve","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"#create interpreter object\ninterp = ClassificationInterpretation.from_learner(learner_resnet50)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Plot the biggest losses of the model\ninterp.plot_top_losses(9,figsize=(12,12),heatmap=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"losses,idxs = interp.top_losses()\nlen(data.valid_ds)==len(losses)==len(idxs)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"interp.plot_confusion_matrix(figsize=(15,5))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#To view the list of classes most misclassified as a list\n#Sorted descending list of largest non-diagonal entries of confusion matrix, presented as actual, predicted, number of occurrences\ninterp.most_confused(min_val=2)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### AUC SCORE","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_val ,y_val = learner_resnet50.get_preds()\n\ndef auc_score(y_pred,y_true,tens=True):\n    score=roc_auc_score(y_true,torch.sigmoid(y_pred)[:,1])\n    if tens:\n        score=tensor(score)\n    else:\n        score=score\n    return score\n\npred_score=auc_score(pred_val ,y_val)\npred_score","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### ACCURACY","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_score_acc=accuracy(pred_val ,y_val)\npred_score_acc","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### ROC CURVE","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fpr, tpr, thresholds = roc_curve(y_val.numpy(), pred_val.numpy()[:,1], pos_label=1)\npred_score_auc = auc(fpr, tpr)\nprint(f'ROC area: {pred_score_auc}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure()\nplt.plot(fpr, tpr, color='orange', label='ROC curve (area = %0.2f)' % pred_score_auc)\nplt.plot([0, 1], [0, 1], color='navy', linestyle='--')\nplt.xlim([-0.01, 1.0])\nplt.ylim([0.0, 1.01])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic')\nplt.legend(loc=\"lower right\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Predictions\n    - Export the final learner object for productionizing\n    - Prediction using above saved model on:\n        - Train and validation images\n        - Uploaded single image\n        - Test images        ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"learner_resnet50.export('/kaggle/working/tmp/models/export.pkl')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Prediction using saved model on training and validation images","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"loaded_learner = load_learner(Path(model_dir))\nloaded_learner.data.classes","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Predict on train data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"img, cat = data.train_ds[0]\nimg.show()\nprint(cat)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_class,pred_idx,pred_probs = loaded_learner.predict(img)\nprint(pred_class, pred_idx,pred_probs)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Predict on validation data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"img, cat = data.valid_ds[1]\nimg.show()\nprint(cat)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_class,pred_idx,pred_probs = loaded_learner.predict(img)\nprint(pred_class, pred_idx,pred_probs)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Prediction using saved model on testing single image","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"img = open_image(Path('../input/test-image/test_img.tif'))\npred_class,pred_idx,pred_probs = loaded_learner.predict(img)\nimg.show()\ntargets = ['Non-Cancerous','Cancerous'] #since sequence of classes in data is as 0,1\nprint(\"Tissue cell is identified as\" , targets[pred_idx] , \"with probability of\", float(pred_probs[pred_idx]*100))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Prediction using saved model on testing data for submission purpose","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"loaded_learner_val = load_learner(Path(model_dir),test=ImageList.from_folder(Path(test_folder)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_test ,y_test = loaded_learner_val.get_preds(ds_type=DatasetType.Test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"## Submission","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"sub=pd.read_csv('../input/histopathologic-cancer-detection/sample_submission.csv').set_index('id')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"clean_names = np.vectorize(lambda imgname: str(imgname).split('/')[-1][:-4])\ncleaned_names = clean_names(data.test_ds.items).astype(str)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.loc[cleaned_names,'label']=pred_test.numpy()[:,1]\nsub.to_csv(f'/kaggle/working/submission_{int(pred_score_auc*100)}auc.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predicted_prob_test = pd.read_csv('./submission_98auc.csv')\npredicted_prob_test.head(10)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Feel free to share doubts, feedbacks or concerns. Also, fuel some motivation by upvoting if notebook has enhanced your learning. \n\n<b> Happy Learning! \n","execution_count":null}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}