{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"# Goal: Cancer image classification\n# no_cancer: 0, cancer: 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%reload_ext autoreload\n%autoreload 2\n%matplotlib inline ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Import libraries \n\nfrom fastai import *\nfrom fastai.vision import *\nimport pandas as pd\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# Creating  paths and display a list\ntrain_dir = ('../input/train')\ntest_dir = ('../input/test')\npath = Path('../input')\npath.ls()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Reading the labels  \n\ndf = pd.read_csv(path/'../input/train_labels.csv')\ndf.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# View the top five train files \nfnames = get_image_files(train_dir)\nfnames[:5]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Create an image data bunch\n# ImageDataBunch: contains all the data to build a model(Train, Validation, and Test(optional))\n# from_csv: Extract labels \n# get_transforms(): Will return tuples Train, Validation and test \n# size: size of image 96x96\n# normalize:changes the range of pixels (0 - 255) and removes the noise\n'''Note: if pixels are below 0 becomes 0, if they are above 255 they\nbecome 255. Making data the same size, same mean and\nsame standard deviation. RGB channels becomes mean of 0\nand standard deviation of 1.  If data is not normalize,\ntraining the model will be difficult'''\n\nnp.random.seed(42) #making sure we get the same dataset \ndata = ImageDataBunch.from_csv(path, folder = 'train', csv_labels = \"train_labels.csv\",\n                               suffix=\".tif\", test = test_dir, size = 96, ds_tfms = get_transforms())\ndata.path = pathlib.Path('.')\ndata.normalize(imagenet_stats)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''AUC, or Area Under Curve, is a metric for binary classification. \nIt’s probably the second most popular one, after accuracy.  However,\nI will be using AUC over accuracy since it is often preferred as \nit provides a “broader” view of the performance '''\n# y_pred: calculates output\n# y_true: target data\n# tens = True: values will all be multiples of 1/10\n\nfrom sklearn.metrics import roc_auc_score\n\ndef auc_score(y_pred,y_true,tens=True):\n    score=roc_auc_score(y_true,torch.sigmoid(y_pred)[:,1])\n    if tens:\n        score=tensor(score)\n    else:\n        score=score\n    return score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Display some images 3x3\n\ndata.show_batch(rows = 3, figsize = (7,6))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Display all the labels(2) and categories(2) \n\nprint(data.classes)\nlen(data.classes), data.c","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Model Training\n'''The model will take images as input and predicted\nprobability will the output for each of the categories.\nCNN will be use as a backbone and a fully connected\nhead with a single hidden layer as a classifier'''\n# models.resnet34: is the architecture 34(34 layers) which will train fast\n'''Note: Basically, is downloading resnet34 pre-trained\nweights. This means that this model has already been trained\nfor a particular task.  This model knows something.'''\n# other option: resnet50 (50 layers)\n# auc_score: for binary classification \n\nlearn = cnn_learner(data, models.resnet34, metrics = auc_score)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n# Let's plot learn to find the best learning rate \n# Select where is decending before is going up\n# This is a key hyperparameter to train a neural network \n'''Determines how quickly (or slowly) we want to update the weights \nafter each iteration''' \n'''Note: If the learning rate is too low: loss function does not improve.\nIf decending: optimal learning rate range (a quick drop in the loss function).\nIf learning rate too high: begins to diverge. '''\n\nlearn.lr_find()\nlearn.recorder.plot()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n# run it for four cycles \n# error_rate 2% and 98% accuracy\nlearn.fit_one_cycle(4, max_lr = slice(3e-4,3e-2))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Results\n# the learn is pass throught and here we have a classification object\n\ninterp = ClassificationInterpretation.from_learner(learn)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Display the top losses of the prediction\n# Note: This will tell us how good is the prediction\n\ninterp.plot_top_losses(9, figsize = (15,11))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# confusion matrix \n# This plot is showing the right number of predictions in blue\n\ninterp.plot_confusion_matrix()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Validation Prediction\n\npreds,y=learn.get_preds()\npred_score= auc_score(preds,y)\npred_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# We have 98% accuracy, but could we make it better?\n# Let's try TTA\n# Test Time Augmentation\n\n'''TTA takes data augmentations at random as \nwell as the un-augmented original.  Then calculate predictions \nfor all these images, take the average, \nand make that our final prediction. This is only \nfor validation set and/or test set.'''\n\npreds,y=learn.TTA()\npred_score_tta=auc_score(preds,y)\npred_score_tta","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Target output\ny[:5]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# predicted output\npreds[:5]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Test Prediction \n\npreds_test,y_test=learn.get_preds(ds_type=DatasetType.Test)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# target output\ny_test[:5]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# predicted output\npreds_test[:5]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\npreds_test_tta,y_test_tta=learn.TTA(ds_type=DatasetType.Test)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# target output\ny_test_tta[:5]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# predicted output\npreds_test_tta[:5]","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}