{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Brief description of the problem and data\nIn this project I will make use of a convolutional nueral network (CNN) in order to detect metatstatic cancer in cells. The data are RGB images, with corresponding labels supplied for the training data. I will run my model on the test data to generate models and validate the test accuracy using Kaggle's submission feature. The data are split into two binary exlusive classes. The images are of cells that either do or do not have cancer. In total, there are 220,025 training cases. I chose to use 80% for training, and 20% for validation in accordance with the rule of thumb. The According to the Kaggle listing, the training data, test data, and labels combined consist of a ~7.75 GB file. There will be some duplication of data in the effort of unzipping the file during download, so if your machine experiences lag after loading, a good intermediate step would be to delete the ZIP file after unzipping.\n\nThe individual image files are all .tif files of shape (96, 96, 3). According to the Kaggle listing, the training label is assigned as positive if at least one pixel in the center (32, 32) of an image contains tumorous tissue. The rest of the image is provided to enable CNN with padding schemes / convolution stride that eats away at the exterior of the image.\n\nThe test dataset consists of unlabeled .tif images. The goal of this project is to predict the labels of those test images with the CNN I will build.","metadata":{}},{"cell_type":"markdown","source":"# Loading the Data\nFirst we'll load the necessary libraries. Please ensure that you have these installed on your environment before running this notebook.","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nimport time\nimport cv2 # For EDA\nfrom PIL import Image","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-30T17:39:53.889434Z","iopub.execute_input":"2024-01-30T17:39:53.889896Z","iopub.status.idle":"2024-01-30T17:40:12.669263Z","shell.execute_reply.started":"2024-01-30T17:39:53.88986Z","shell.execute_reply":"2024-01-30T17:40:12.667699Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The image files provided in the Kaggle competition have the .tif extension, which Keras dislikes. We will need to convert these to .png in order for Keras to succeed in creating our Data object.","metadata":{}},{"cell_type":"code","source":"def convertToPNG(newDir, oldDir, fileName):\n  \"\"\"\n  This function converts a .tif image file from oldDir to a .png and saves it to\n  newDir. \n  Input: string filePath\n  Output: None\n  \"\"\"\n  targetPath = newDir + fileName[:-4] + \".png\"\n  this_img = Image.open(oldDir + fileName)\n  this_img.save(targetPath)\n  this_img.close()","metadata":{"execution":{"iopub.status.busy":"2024-01-30T17:40:12.67153Z","iopub.execute_input":"2024-01-30T17:40:12.672325Z","iopub.status.idle":"2024-01-30T17:40:12.68111Z","shell.execute_reply.started":"2024-01-30T17:40:12.67229Z","shell.execute_reply":"2024-01-30T17:40:12.679894Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make necessary containers for converted training images\n## with subdirectories for binary classification\ndir0train = \"./pngContainer/pngTrain/0/\"\ndir1train = \"./pngContainer/pngTrain/1/\"\nos.makedirs(dir0train)\nos.makedirs(dir1train)\n\n#Read in and sort training labels alphanumerically. The sorting is identical\n## to the sorting we will perform on the images (they have the same names).\ntrain_labs_df = pd.read_csv(\"../input/histopathologic-cancer-detection/train_labels.csv\")\ntraining_labels = list(train_labs_df.sort_values(by = \"id\", axis = 0)[\"label\"])\n\n# Convert all .tif files in \"./train/\" to .png 's . This will take about 10\n## minutes total.\ntrain_dir = \"../input/histopathologic-cancer-detection/train/\"\ntrain_imgs = os.listdir(train_dir)\ntrain_imgs.sort()\n\n#Quick sanity check\nassert len(train_imgs) == len(training_labels)\n\n#Populate subdirectories. This will take a while (for me ~ 10 minutes). Grab a\n## snack. If there is a better way to do this I'm all ears.\nfor i in range(len(train_imgs)):\n  if training_labels[i] == 0:\n    convertToPNG(dir0train, train_dir, train_imgs[i])\n  else:\n    assert training_labels[i] == 1\n    convertToPNG(dir1train, train_dir, train_imgs[i])","metadata":{"execution":{"iopub.status.busy":"2024-01-30T17:41:00.3787Z","iopub.execute_input":"2024-01-30T17:41:00.379207Z","iopub.status.idle":"2024-01-30T18:20:18.219809Z","shell.execute_reply.started":"2024-01-30T17:41:00.379172Z","shell.execute_reply":"2024-01-30T18:20:18.217526Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Call image_dataset_from_directory to build keras Data object\ntrain_data = tf.keras.utils.image_dataset_from_directory(\n    directory = \"../working/pngContainer/pngTrain\",\n    labels = \"inferred\",\n    label_mode = \"binary\",\n    color_mode = \"rgb\",\n    image_size = (96, 96),\n    seed = 42,\n    batch_size = 32\n)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T18:21:15.328427Z","iopub.execute_input":"2024-01-30T18:21:15.328841Z","iopub.status.idle":"2024-01-30T18:21:35.505954Z","shell.execute_reply.started":"2024-01-30T18:21:15.328811Z","shell.execute_reply":"2024-01-30T18:21:35.504516Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Confirm that training data was propertly loaded\ntrain_data","metadata":{"execution":{"iopub.status.busy":"2024-01-30T18:21:35.508214Z","iopub.execute_input":"2024-01-30T18:21:35.508588Z","iopub.status.idle":"2024-01-30T18:21:35.518684Z","shell.execute_reply.started":"2024-01-30T18:21:35.508557Z","shell.execute_reply":"2024-01-30T18:21:35.517147Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA\nI will first show an example photo from the training data, along with its type and shape.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 10))\nfor images, labels in train_data.take(1):\n    for i in range(9):\n        ax = plt.subplot(3, 3, i + 1)\n        plt.imshow(images[i].numpy().astype(\"uint8\"))\n        plt.title(int(labels[i]))\n        plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2024-01-30T18:21:39.244261Z","iopub.execute_input":"2024-01-30T18:21:39.244674Z","iopub.status.idle":"2024-01-30T18:21:41.011147Z","shell.execute_reply.started":"2024-01-30T18:21:39.244644Z","shell.execute_reply":"2024-01-30T18:21:41.009658Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make Histogram of Training Labels\nplt.hist(training_labels, bins = 2)\nplt.xlabel(\"Category\")\nplt.ylabel(\"Count\")\nplt.title(\"Histogram of Training Labels\")\nplt.show()\n     ","metadata":{"execution":{"iopub.status.busy":"2024-01-30T18:21:45.700716Z","iopub.execute_input":"2024-01-30T18:21:45.701247Z","iopub.status.idle":"2024-01-30T18:21:47.001966Z","shell.execute_reply.started":"2024-01-30T18:21:45.701205Z","shell.execute_reply":"2024-01-30T18:21:47.000466Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Architecture and Construction\nI will try a couple different architectures, because CNN architecture itself is a hyperparameter that needs optimization. Keras provides a wonderful API to build simple CNN's in. In particular, the Sequential API is built for single input single output models. We are inputting single images, and want to output a single label, so the Sequential API works perfect.\n\nWe learned that either relu or PreLu is the best choice of activation function for hidden layers. The Keras Sequential API doesn't have a built in PreLu, and I fear that any that I implement will be algorthmically inefficient in Python, so I won't worry about optimizing the interem activation functions. We will use (3x3) filters as suggested. I will try both the Adam and RMSProp optimizers, and various values of learning rate between (0.0001, 0.001). I will also use a learning_rate scheduler to minimize variance towards the end of training. I will try Dropout layers in various locations. I will use binary cross entropy as my loss function, because it is the most appropriate for logit outputs from a sigmoid output activation.\n\nAll of my hyperparameter tuning will be done manually. The fitting will take a while, but will return the optimal choice of architecture and hyperparameters by validation accuracy.","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.layers import Conv2D, BatchNormalization, MaxPooling2D, Dropout, Dense, Flatten\n\ndef buildaModel(eta, dropout1, dropout2, optimizer):\n  \"\"\"\n  This function builds a model according to the hyperparamaters passed. The documentation for keras-tuner is quite poor, and there are many bugs.\n  I will just make my own grid searcher.\n\n  Inputs:\n    float eta, learning rate\n    Boolean dropout1, whether to include dropout layer after first c-c-m\n    Boolean dropout2, whether to include dropout layer after second c-c-m\n    string optimizer, either \"Adam\" or \"RMSprop\"\n  \n  Output: Keras sequential model object\n  \"\"\"\n  batch_input_shape = (32, 96, 96, 3) # Mini-batches are of size 32. Images are of shape (96, 96, 3)\n  model = tf.keras.Sequential()\n  model.add(BatchNormalization(axis = -1))\n  #CNN [c, c, m] Layers\n  model.add(Conv2D(32,\n                   kernel_size = 3,\n                   padding=\"same\",\n                   activation = \"relu\",\n                   batch_input_shape = batch_input_shape))\n  model.add(Conv2D(32, kernel_size = 3, activation = \"relu\"))\n  model.add(MaxPooling2D(pool_size = 2))\n  model.add(BatchNormalization(axis = -1)) #Normalize to prevent overfitting\n\n  model.add(Conv2D(64, kernel_size = 3, activation = \"relu\"))\n  model.add(Conv2D(64, kernel_size = 3, activation = \"relu\"))\n  model.add(MaxPooling2D(pool_size = 2))\n  #Decide whether or not to use a Dropout between [c, c, m] instances\n  model.add(BatchNormalization(axis = -1)) #Normalize to prevent overfitting\n\n  model.add(Conv2D(128, kernel_size = 3, activation = \"relu\"))\n  model.add(Conv2D(128, kernel_size = 3, activation = \"relu\"))\n  model.add(MaxPooling2D(pool_size = 2))\n  #Decide whether or not to use a Dropout between [c, c, m] instances\n  model.add(BatchNormalization(axis = -1)) #Normalize to prevent overfitting\n\n  #Now send the model into an ANN for binary classification\n  model.add(Flatten())\n  if dropout1:\n    model.add(Dropout(0.25))\n  model.add(Dense(256, activation = \"relu\"))\n  model.add(Dense(128, activation = \"relu\"))\n  if dropout2:\n    model.add(Dropout(0.25))\n  model.add(Dense(32, activation = \"relu\"))\n  model.add(Dense(1, activation = \"sigmoid\")) #Output layer\n\n  #Make a learning rate schedule to prevent variance at the end of training\n  lr_schedule = tf.keras.optimizers.schedules.ExponentialDecay(\n      initial_learning_rate = eta,\n      decay_steps = 10000,\n      decay_rate=0.9\n      )\n  \n  #Compile the model\n  if optimizer == \"Adam\":\n    model.compile(\n      optimizer=tf.keras.optimizers.Adam(learning_rate = lr_schedule),\n      loss = \"binary_crossentropy\",\n      metrics=[\"accuracy\"]\n    )\n  else:\n    assert optimizer == \"RMSprop\"\n    model.compile(\n      optimizer = tf.keras.optimizers.RMSprop(learning_rate = lr_schedule),\n      loss = \"binary_crossentropy\",\n      metrics = [\"accuracy\"]\n    )\n  return(model)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T18:21:50.967647Z","iopub.execute_input":"2024-01-30T18:21:50.968039Z","iopub.status.idle":"2024-01-30T18:21:50.989335Z","shell.execute_reply.started":"2024-01-30T18:21:50.968009Z","shell.execute_reply":"2024-01-30T18:21:50.987569Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Trial model before gridsearch\nmyModel = buildaModel(eta = 0.001, dropout1 = True, dropout2 = True, optimizer = \"Adam\")\ncallbacks_list =  [tf.keras.callbacks.EarlyStopping(monitor=\"loss\", patience = 10, mode = \"min\")]\nhistory = myModel.fit(train_data,\n                      epochs = 99,\n                      steps_per_epoch = len(train_data)/100,\n                      callbacks = callbacks_list)\n\nprint(myModel.summary())","metadata":{"execution":{"iopub.status.busy":"2024-01-30T18:21:55.214551Z","iopub.execute_input":"2024-01-30T18:21:55.214971Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make a nice visualization of our test model: Training Accuracy Vs. Epoch\ntestAccs = history.history[\"accuracy\"]\nepochs = [i for i in range(len(testAccs))]\nplt.plot(epochs, testAccs)\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Training Accuracy\")\nplt.title(\"Training Accuracy Vs. Epoch\")\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make another nice visualization: Loss Vs. Epoch\ntestLosses = history.history[\"loss\"]\nplt.plot(epochs, testLosses)\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Training Loss\")\nplt.title(\"Training Loss Vs. Epoch\")\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"#Perform HP optimization\nimport itertools\nstartTime = time.time()\ncallbacks_list =  [tf.keras.callbacks.EarlyStopping(monitor=\"loss\", patience = 10, mode = \"min\")]\neta_to_try = [0.0001, 0.001, 0.005, 0.01]\ndropouts_to_try = [(True, True)]\noptimizers_to_try = [\"Adam\", \"RMSprop\"]\n\nhpPermutations = list(itertools.product(*[eta_to_try, dropouts_to_try, optimizers_to_try]))\ncurrent_best_model = (None, 0)\ncurrent_best_history = None\n\nmodel_accuracies = list()\nfor hpList in hpPermutations:\n  print(\"Now trying hyperparams:\", hpList)\n  this_model = buildaModel(eta = hpList[0],\n                           dropout1 = True,\n                           dropout2 = True,\n                           optimizer = hpList[2]\n                           )\n  this_history = this_model.fit(train_data,\n                                epochs = 99,\n                                steps_per_epoch = len(train_data)/100,\n                                callbacks = callbacks_list\n                                )\n  if this_history.history[\"accuracy\"][-1] > current_best_model[1]:\n    current_best_model = (this_model, this_history.history[\"accuracy\"][-1])\n    current_best_history = this_history.history\n\nendTime = time.time()\nprint(\"Total HP optimization time:\", round((endTime - startTime)/60, 3), \"minutes\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Try best settings with a smaller value for eta and compare\nmod = buildaModel(0.0001, True, True, \"Adam\")\nmod.fit(train_data,\n        epochs = 99,\n        steps_per_epoch = len(train_data)/100,\n        callbacks = callbacks_list)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"print(current_best_model[0].optimizer)\ndir(current_best_model[0])\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make Epochs Vs. Training Accuracy for beest model\naccs = current_best_history[\"accuracy\"]\nepochs = [i for i in range(len(accs))]\n\nplt.plot(epochs, accs)\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Training Accuracy\")\nplt.title(\"Training Accuracy Vs. Epoch for Best Model: HPs (Fill in later)\")\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make Epochs Vs. Training Loss for best model\nloss = current_best_history[\"loss\"]\n\nplt.plot(epochs, loss)\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Training Loss\")\nplt.title(\"Training Loss Vs. Epoch for Best Model: HPs (Fill in Later)\")\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make necessary containers for converted test images. This section will be\n## basically a mirror of the .tif -> .png conversion done on the training data.\ndirTestTarget = \"./pngContainer/pngTest/imgs/\"\n#Note we require .../imgs/ so that there exists a subdirectory for keras to find\nos.makedirs(dirTestTarget)\ntest_dir = \"./test/\"\ntest_imgs = os.listdir(test_dir)\ntest_imgs.sort()\n\nfor img in test_imgs:\n  convertToPNG(dirTestTarget, test_dir, img)\n\n# Make test_data as BatchDataset\ntest_data = tf.keras.utils.image_dataset_from_directory(\n    directory = \"/content/pngContainer/pngTest\",\n    labels = None,\n    label_mode = \"binary\",\n    color_mode = \"rgb\",\n    image_size = (96, 96),\n    batch_size = 32\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make Predictions using test_data\nyhat = current_best_model[0].predict(test_data)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Finalize predictions and make csv to submit to Kaggle\nyhat = yhat.reshape((yhat.shape[0], ))\nlabel = [round(yhat[i]) for i in range(len(yhat))]\ntestids = test_data.file_paths\nid = [testids[i][35:-4] for i in range(len(testids))]\n\npredDic = {\n    \"id\" : id,\n    \"label\" : label\n}\npreddf = pd.DataFrame(predDic)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preddf.to_csv(\"./BestModelPredictions.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}