{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<a id = \"intro\"></a>\n# Problem at Hand\n<br>\nAn algorithm is to be executed to identify metastatic cancer using image classification. The data comes from a PatchCamelyon dataset with all unique images. Aside from the image data, there is text-related training data which consists of a CSV file containing the identification codes of the images and the associated label that it is issued. Any images labeled as non-cancerous are under the value 1, and any images labeled as cancerous are under the value 0. \n\nThis is a submission for the Kaggle competition in Histopathologic Cancer Detection. \n\nFirst, import all the necessary libraries and packages needed.","metadata":{}},{"cell_type":"code","source":"from kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\nuser_credential = user_secrets.get_gcloud_credential()\nuser_secrets.set_tensorflow_credential(user_credential)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:25.745168Z","iopub.execute_input":"2025-06-17T23:15:25.745459Z","iopub.status.idle":"2025-06-17T23:15:25.891566Z","shell.execute_reply.started":"2025-06-17T23:15:25.745438Z","shell.execute_reply":"2025-06-17T23:15:25.891066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport cv2\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# libraries for EDA and Data Visualization\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Importing Tensorflow Libraries\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D\nfrom tensorflow.keras.layers import Dense, Dropout, Flatten, Activation\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, ModelCheckpoint\nfrom tensorflow.keras.optimizers import Adam\n\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import shuffle","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:25.920331Z","iopub.execute_input":"2025-06-17T23:15:25.920530Z","iopub.status.idle":"2025-06-17T23:15:25.925501Z","shell.execute_reply.started":"2025-06-17T23:15:25.920515Z","shell.execute_reply":"2025-06-17T23:15:25.924632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.getcwd()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:26.104872Z","iopub.execute_input":"2025-06-17T23:15:26.105526Z","iopub.status.idle":"2025-06-17T23:15:26.109428Z","shell.execute_reply.started":"2025-06-17T23:15:26.105503Z","shell.execute_reply":"2025-06-17T23:15:26.108904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.listdir(os.getcwd())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:26.999866Z","iopub.execute_input":"2025-06-17T23:15:27.000126Z","iopub.status.idle":"2025-06-17T23:15:27.005340Z","shell.execute_reply.started":"2025-06-17T23:15:27.000106Z","shell.execute_reply":"2025-06-17T23:15:27.004636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the directories and data\nos.listdir('../input/')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:27.199969Z","iopub.execute_input":"2025-06-17T23:15:27.200551Z","iopub.status.idle":"2025-06-17T23:15:27.204459Z","shell.execute_reply.started":"2025-06-17T23:15:27.200527Z","shell.execute_reply":"2025-06-17T23:15:27.203812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.listdir('../input/histopathologic-cancer-detection')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:29.843752Z","iopub.execute_input":"2025-06-17T23:15:29.844202Z","iopub.status.idle":"2025-06-17T23:15:29.849218Z","shell.execute_reply.started":"2025-06-17T23:15:29.844179Z","shell.execute_reply":"2025-06-17T23:15:29.848427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv')\ntrain_path = '/kaggle/input/train/'\ntest_path = '/kaggle/input/test/'\n# quick look at the label stats - there are two: 0 (not cancerous) and 1 (cancerous)\nprint(dataset['label'].value_counts())","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:30.013965Z","iopub.execute_input":"2025-06-17T23:15:30.014502Z","iopub.status.idle":"2025-06-17T23:15:30.209384Z","shell.execute_reply.started":"2025-06-17T23:15:30.014484Z","shell.execute_reply":"2025-06-17T23:15:30.208654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nGet dataset columns' names\nThere are two: the images' ids and the labels (0 for non-canceerous images, 1 otherwise)\n'''\ndataset.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:30.210437Z","iopub.execute_input":"2025-06-17T23:15:30.210715Z","iopub.status.idle":"2025-06-17T23:15:30.215212Z","shell.execute_reply.started":"2025-06-17T23:15:30.210688Z","shell.execute_reply":"2025-06-17T23:15:30.214616Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Get the shape of dataset containing all of the labels. There are two columns and 220,025 rows of image classification data. ","metadata":{}},{"cell_type":"code","source":"print(dataset.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:31.432493Z","iopub.execute_input":"2025-06-17T23:15:31.432780Z","iopub.status.idle":"2025-06-17T23:15:31.436645Z","shell.execute_reply.started":"2025-06-17T23:15:31.432758Z","shell.execute_reply":"2025-06-17T23:15:31.435752Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:31.603742Z","iopub.execute_input":"2025-06-17T23:15:31.604388Z","iopub.status.idle":"2025-06-17T23:15:31.612850Z","shell.execute_reply.started":"2025-06-17T23:15:31.604355Z","shell.execute_reply":"2025-06-17T23:15:31.611965Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset.tail()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:31.755603Z","iopub.execute_input":"2025-06-17T23:15:31.755892Z","iopub.status.idle":"2025-06-17T23:15:31.762963Z","shell.execute_reply.started":"2025-06-17T23:15:31.755873Z","shell.execute_reply":"2025-06-17T23:15:31.762233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Take a look at the labels in the training data; there should be all unique id values\ndataset['label'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:31.881614Z","iopub.execute_input":"2025-06-17T23:15:31.881880Z","iopub.status.idle":"2025-06-17T23:15:31.888388Z","shell.execute_reply.started":"2025-06-17T23:15:31.881858Z","shell.execute_reply":"2025-06-17T23:15:31.887729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get all value counts\ndataset.value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:32.035127Z","iopub.execute_input":"2025-06-17T23:15:32.035406Z","iopub.status.idle":"2025-06-17T23:15:32.410478Z","shell.execute_reply.started":"2025-06-17T23:15:32.035384Z","shell.execute_reply":"2025-06-17T23:15:32.409723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check if there are any null values in the dataset; there should be none\ndataset.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:32.411674Z","iopub.execute_input":"2025-06-17T23:15:32.411900Z","iopub.status.idle":"2025-06-17T23:15:32.427966Z","shell.execute_reply.started":"2025-06-17T23:15:32.411877Z","shell.execute_reply":"2025-06-17T23:15:32.427130Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get percentages of which labels are cancerous or not\nzeros = len(dataset[dataset['label'] == 0])\nones = len(dataset[dataset['label'] == 1])\ntotal = dataset.shape[0]\n\nprint(\"Value is 0:\", zeros, '÷', total, '=', zeros / total)\nprint(\"Value is 1:\", ones, '÷', total, '=', ones / total)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:32.888476Z","iopub.execute_input":"2025-06-17T23:15:32.889012Z","iopub.status.idle":"2025-06-17T23:15:32.905974Z","shell.execute_reply.started":"2025-06-17T23:15:32.888989Z","shell.execute_reply":"2025-06-17T23:15:32.905039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot the distributions\n# Credit can be attributed to the following notebook: \n# https://www.kaggle.com/code/alexandermaitken/cnn-cancer-detection?scriptVersionId=240762737&cellId=10\n# \n# These should show the values from the previous code block\n\nplt.bar('0', len(dataset[dataset['label'] == 0]))\nplt.bar('1', len(dataset[dataset['label'] == 1]))\nplt.xlabel(\"Label Value\")\nplt.ylabel(\"Number of Occurrences\")\nplt.title(\"Training Data Labels Count\")\nplt.legend(['Not Cancerous', 'Cancerous'])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:33.081109Z","iopub.execute_input":"2025-06-17T23:15:33.081357Z","iopub.status.idle":"2025-06-17T23:15:33.224387Z","shell.execute_reply.started":"2025-06-17T23:15:33.081322Z","shell.execute_reply":"2025-06-17T23:15:33.223619Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"About 59% of the data is listed as 0 or non-cancerous, with the other 41% being labeled as 1 or cancerous. \n<br>\nNow, let's take a look at all the descriptive statistics.","metadata":{}},{"cell_type":"code","source":"dataset.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:33.958069Z","iopub.execute_input":"2025-06-17T23:15:33.958335Z","iopub.status.idle":"2025-06-17T23:15:33.972285Z","shell.execute_reply.started":"2025-06-17T23:15:33.958315Z","shell.execute_reply":"2025-06-17T23:15:33.971639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Credit can be attributed to the following cell in this notebook to show sample images from the test and train folders\n# https://www.kaggle.com/code/alexandermaitken/cnn-cancer-detection?scriptVersionId=240762737&cellId=12\n\n# Will need to alter base_path for training data since it is being used online via Kaggle\n# Also shows the image of any one shape - or the number of pixels on any one image\nbase_path='/kaggle/input/histopathologic-cancer-detection/train'\ndef show_samples(df, label, n=10):\n    samples = df[df['label'] == label].sample(n)\n    fig, axes = plt.subplots(1, n, figsize=(15, 5))\n    for img_id, ax in zip(samples['id'], axes):\n        # img = get_image(img_id)\n        path = os.path.join(base_path, f\"{img_id}.tif\")\n        img = cv2.imread(path)\n        print(img.shape)\n\n        ax.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n        ax.axis('off')\n    plt.suptitle(f\"Label: {label}\")\n    plt.show()\n\n# Show example images of non-cancerous images along with their respective shapes\n# in pixels and colors type\nshow_samples(dataset, label=0)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:34.124080Z","iopub.execute_input":"2025-06-17T23:15:34.124264Z","iopub.status.idle":"2025-06-17T23:15:34.418561Z","shell.execute_reply.started":"2025-06-17T23:15:34.124251Z","shell.execute_reply":"2025-06-17T23:15:34.417909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Show example images of cancerous images along with their respective shapes\n# in pixels and colors type\nshow_samples(dataset, label=1)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:34.419693Z","iopub.execute_input":"2025-06-17T23:15:34.419918Z","iopub.status.idle":"2025-06-17T23:15:34.722705Z","shell.execute_reply.started":"2025-06-17T23:15:34.419901Z","shell.execute_reply":"2025-06-17T23:15:34.721991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Total image count in TRAIN dataset: \", len(os.listdir(\"/kaggle/input/histopathologic-cancer-detection/train\")))\nprint(\"Total image count in TEST dataset: \", len(os.listdir(\"/kaggle/input/histopathologic-cancer-detection/test\")))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:34.723715Z","iopub.execute_input":"2025-06-17T23:15:34.723925Z","iopub.status.idle":"2025-06-17T23:15:37.719500Z","shell.execute_reply.started":"2025-06-17T23:15:34.723909Z","shell.execute_reply":"2025-06-17T23:15:37.718839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for duplicates - there should be none - again check for uniques\nprint(\"\\nDuplicate rows in training labels:\")\nprint(dataset[dataset.duplicated(keep=False)])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:37.720784Z","iopub.execute_input":"2025-06-17T23:15:37.721311Z","iopub.status.idle":"2025-06-17T23:15:37.776250Z","shell.execute_reply.started":"2025-06-17T23:15:37.721291Z","shell.execute_reply":"2025-06-17T23:15:37.775527Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id = \"data prep\"></a>\n# Data Preparation\n\nLet's take a sample of a fraction of these images that are featured in the entire dataset. For simplicity and demonstration purposes, work with a sample size half of the actual dataset, which would be 50,000. <br>\nI will be working with 50,000 sample images, as running through all of the samples will take lots more time running any models. This is less than one-fourth of all the images that are available.<br>\n<br>\nLet's take a sample of a fraction of these images that are featured in the entire dataset. For simplicity and demonstration purposes, work with a sample size less than half of the actual dataset, which would be 110,000.<br>\nI will be working with 50,000 sample images, as running through all of the samples will take lots more time running any models. This is less than one-fourth of all the images that are available.\n","metadata":{}},{"cell_type":"code","source":"# Set sample size to 50,000\nSAMPLE_SIZE=50000\n# Set random states to 72\nRANDOM_STATE=72","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:37.777116Z","iopub.execute_input":"2025-06-17T23:15:37.777394Z","iopub.status.idle":"2025-06-17T23:15:37.780661Z","shell.execute_reply.started":"2025-06-17T23:15:37.777372Z","shell.execute_reply":"2025-06-17T23:15:37.780132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create the train and test sets with the sample size\ndset0=dataset[dataset['label']==0].sample(SAMPLE_SIZE,random_state=RANDOM_STATE)\ndset1=dataset[dataset['label']==1].sample(SAMPLE_SIZE,random_state=RANDOM_STATE)\n\n# Put the dataset dataframes together\n# update the dataset variable appropriately\ndataset = pd.concat([dset0, dset1], axis=0).reset_index(drop=True)\n# shuffle\ndataset = shuffle(dataset)\n\n# Check that both the label counts are 50,000 each\ndataset['label'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:37.781741Z","iopub.execute_input":"2025-06-17T23:15:37.782222Z","iopub.status.idle":"2025-06-17T23:15:37.820038Z","shell.execute_reply.started":"2025-06-17T23:15:37.782205Z","shell.execute_reply":"2025-06-17T23:15:37.819302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:37.820863Z","iopub.execute_input":"2025-06-17T23:15:37.821628Z","iopub.status.idle":"2025-06-17T23:15:37.828096Z","shell.execute_reply.started":"2025-06-17T23:15:37.821602Z","shell.execute_reply":"2025-06-17T23:15:37.827407Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Do an 80%-to-20% train-test data split on the image data labels. A total of 110,000 labels in dataset, 80,000 of the labels should be in the training data and 20,000 of the labels should be in testing data.","metadata":{}},{"cell_type":"code","source":"# Create the 80-20 split for the image data labels\n\n# The validation/actual labels - get the labels of either 0 or 1\ny = dataset['label']\n\n# Split the train data and test labels\ntrain_data, train_labels = train_test_split(dataset, test_size=0.2, \n                                            random_state=RANDOM_STATE, stratify=y)\n\n# Print out the shapes to verify sizes of the train and test datasets\nprint(train_data.shape, train_labels.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:50.649458Z","iopub.execute_input":"2025-06-17T23:15:50.650145Z","iopub.status.idle":"2025-06-17T23:15:50.687284Z","shell.execute_reply.started":"2025-06-17T23:15:50.650121Z","shell.execute_reply":"2025-06-17T23:15:50.686517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get a sample of train_data and train_labels\ntrain_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:51.995500Z","iopub.execute_input":"2025-06-17T23:15:51.996250Z","iopub.status.idle":"2025-06-17T23:15:52.002783Z","shell.execute_reply.started":"2025-06-17T23:15:51.996227Z","shell.execute_reply":"2025-06-17T23:15:52.001981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:15:52.167531Z","iopub.execute_input":"2025-06-17T23:15:52.167774Z","iopub.status.idle":"2025-06-17T23:15:52.174696Z","shell.execute_reply.started":"2025-06-17T23:15:52.167758Z","shell.execute_reply":"2025-06-17T23:15:52.173882Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.columns, train_labels.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:16:00.376966Z","iopub.execute_input":"2025-06-17T23:16:00.377440Z","iopub.status.idle":"2025-06-17T23:16:00.381883Z","shell.execute_reply.started":"2025-06-17T23:16:00.377413Z","shell.execute_reply":"2025-06-17T23:16:00.381092Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id = \"data modeling\"></a>\n# Building the Data Model\n<br>\nSet up a base directory as well as training and validation directories inside the base directory. Join them with the base directory and in each of the train and validate directories create two subdirectories which separate the positive cancerous images and data from the negative and non-cancerous images and data. ","metadata":{}},{"cell_type":"code","source":"base_dir = '/kaggle/working/base_dir'\ntrain_dir = os.path.join(base_dir, 'train_dir')\nval_dir = os.path.join(base_dir, 'val_dir')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:16:01.554039Z","iopub.execute_input":"2025-06-17T23:16:01.554626Z","iopub.status.idle":"2025-06-17T23:16:01.557864Z","shell.execute_reply.started":"2025-06-17T23:16:01.554604Z","shell.execute_reply":"2025-06-17T23:16:01.557040Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Use ImageDataGenerator and create a base folder called 'base' to test out model\n# and determine what images are to be sorted as cancerous or not cancerous\n\n# First, create the 'base' directory\n# Should be removed and recreated if already there\n\nimport shutil\n\nbase_dir = '/kaggle/working/base_dir'\nos.makedirs(base_dir, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:16:01.711290Z","iopub.execute_input":"2025-06-17T23:16:01.711475Z","iopub.status.idle":"2025-06-17T23:16:01.715245Z","shell.execute_reply.started":"2025-06-17T23:16:01.711461Z","shell.execute_reply":"2025-06-17T23:16:01.714395Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Create 2 folders inside 'base', with 2 subfolders in each folder titled 'negative' and 'positive' which will contain associated data of cancerous or non-cancerous images or related information.\n<br><br>\ntrain_imgs (train_dir) - the trained images/values<br>\n    negative - no cancerous tissue/no tumors<br>\n    positive - images contain cancerous tissue/tumors<br><br>\n\nval_imgs (val_dir) - the validated values tied to images<br>\n    negative - no cancerous tissue/no tumors<br>\n    positive - images contain cancerous tissue/tumors<br>\n","metadata":{}},{"cell_type":"code","source":"# Create a path to 'base' to make the two directories inside of base\n# Make the directory train_dir\ntrain_dir = os.path.join(base_dir, 'train_imgs')\nos.makedirs(train_dir, exist_ok=True)\n\n# Make the directory val_dir\nval_dir = os.path.join(base_dir, 'val_imgs')\nos.makedirs(val_dir, exist_ok=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:16:02.007253Z","iopub.execute_input":"2025-06-17T23:16:02.007459Z","iopub.status.idle":"2025-06-17T23:16:02.011304Z","shell.execute_reply.started":"2025-06-17T23:16:02.007443Z","shell.execute_reply":"2025-06-17T23:16:02.010745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create the subdirectories inside the train directories that were just created\nnegative = os.path.join(train_dir, 'negative')\nos.makedirs(negative, exist_ok=True)\npositive = os.path.join(train_dir, 'positive')\nos.makedirs(positive, exist_ok=True)\n\n\n# create new folders inside value directories that were just created\nnegative = os.path.join(val_dir, 'negative')\nos.makedirs(negative, exist_ok=True)\npositive = os.path.join(val_dir, 'positive')\nos.makedirs(positive, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:16:02.188976Z","iopub.execute_input":"2025-06-17T23:16:02.189166Z","iopub.status.idle":"2025-06-17T23:16:02.193886Z","shell.execute_reply.started":"2025-06-17T23:16:02.189152Z","shell.execute_reply":"2025-06-17T23:16:02.193351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check that the directories and subdirectories have been created\nprint(os.listdir('base_dir/train_imgs'))\nprint(os.listdir('base_dir/val_imgs'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:16:03.218041Z","iopub.execute_input":"2025-06-17T23:16:03.218599Z","iopub.status.idle":"2025-06-17T23:16:03.222564Z","shell.execute_reply.started":"2025-06-17T23:16:03.218557Z","shell.execute_reply":"2025-06-17T23:16:03.221870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(os.listdir('/kaggle/working/base_dir/train_imgs'))\nprint(os.listdir('/kaggle/working/base_dir/val_imgs'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:16:03.344974Z","iopub.execute_input":"2025-06-17T23:16:03.345466Z","iopub.status.idle":"2025-06-17T23:16:03.349319Z","shell.execute_reply.started":"2025-06-17T23:16:03.345449Z","shell.execute_reply":"2025-06-17T23:16:03.348513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# dataset id set as index\ndataset.set_index('id', inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:16:03.621367Z","iopub.execute_input":"2025-06-17T23:16:03.621574Z","iopub.status.idle":"2025-06-17T23:16:03.625335Z","shell.execute_reply.started":"2025-06-17T23:16:03.621558Z","shell.execute_reply":"2025-06-17T23:16:03.624572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the train data and train values in list form\n\nlist_train_data = list(train_data['id'])\nlist_train_values = list(train_labels['id'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:16:03.825336Z","iopub.execute_input":"2025-06-17T23:16:03.825526Z","iopub.status.idle":"2025-06-17T23:16:03.848463Z","shell.execute_reply.started":"2025-06-17T23:16:03.825511Z","shell.execute_reply":"2025-06-17T23:16:03.847934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_path_pull = \"/kaggle/input/histopathologic-cancer-detection\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:16:05.158633Z","iopub.execute_input":"2025-06-17T23:16:05.159319Z","iopub.status.idle":"2025-06-17T23:16:05.162756Z","shell.execute_reply.started":"2025-06-17T23:16:05.159278Z","shell.execute_reply":"2025-06-17T23:16:05.161999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Transfer the train images\n\nfor image in list_train_data:\n    \n    # the id in the csv file does not have the .tif extension therefore we add it here\n    fname = image + '.tif'\n    # get the label for a certain image\n    target = dataset.loc[image,'label']\n    \n    # these must match the folder names\n    if target == 0:\n        label = 'negative'\n    else:\n        label = 'positive'\n    \n    # source path to image data path\n    src = os.path.join(data_path_pull, 'train', fname)\n    # destination path to image\n    dst = os.path.join(train_dir, label, fname)\n    # copy the image from the source to the destination\n    shutil.copyfile(src, dst)\n\n\n# Transfer the val images\nfor image in list_train_values:\n    \n    # the id in the csv file does not have the .tif extension therefore we add it here\n    fname = image + '.tif'\n    # get the label for a certain image\n    target = dataset.loc[image,'label']\n    \n    # these must match the folder names\n    if target == 0:\n        label = 'negative'\n    else:\n        label = 'positive'\n    \n\n    # source path to image\n    src = os.path.join(data_path_pull, 'train', fname)\n    # destination path to image\n    dst = os.path.join(val_dir, label, fname)\n    # copy the image from the source to the destination\n    shutil.copyfile(src, dst)\n\n\n# check how many train images there are in each directory\nprint(len(os.listdir('base_dir/train_imgs/negative')))\nprint(len(os.listdir('base_dir/train_imgs/positive')))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:16:05.414202Z","iopub.execute_input":"2025-06-17T23:16:05.414857Z","iopub.status.idle":"2025-06-17T23:20:39.255146Z","shell.execute_reply.started":"2025-06-17T23:16:05.414823Z","shell.execute_reply":"2025-06-17T23:20:39.254298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set up the generators and the appropriate directories/paths\ntrain_path = 'base_dir/train_imgs'\nvalid_path = 'base_dir/val_imgs'\ntest_path = '/kaggle/input/histopathologic-cancer-detection/test'\n\n# Splits were titled train_data (80), train_labels (20)\n\nnum_train_samples = len(train_data)\nnum_val_samples = len(train_labels)\ntrain_batch_size = 10\nval_batch_size = 10\n\n\ntrain_steps = np.ceil(num_train_samples / train_batch_size)\nval_steps = np.ceil(num_val_samples / val_batch_size)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:21:03.267518Z","iopub.execute_input":"2025-06-17T23:21:03.268376Z","iopub.status.idle":"2025-06-17T23:21:03.273222Z","shell.execute_reply.started":"2025-06-17T23:21:03.268343Z","shell.execute_reply":"2025-06-17T23:21:03.272441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check the types on the train_steps and val_steps\ntype(train_steps), type(val_steps)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:21:04.712959Z","iopub.execute_input":"2025-06-17T23:21:04.713432Z","iopub.status.idle":"2025-06-17T23:21:04.717477Z","shell.execute_reply.started":"2025-06-17T23:21:04.713410Z","shell.execute_reply":"2025-06-17T23:21:04.716781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Since the model only takes integers, set the train_steps and val_steps to type int and then check it. \ntrain_steps = train_steps.astype(int)\nval_steps= val_steps.astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:21:05.989763Z","iopub.execute_input":"2025-06-17T23:21:05.990307Z","iopub.status.idle":"2025-06-17T23:21:05.993392Z","shell.execute_reply.started":"2025-06-17T23:21:05.990281Z","shell.execute_reply":"2025-06-17T23:21:05.992618Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check once again\nprint(train_steps, train_steps.dtype)\nprint(val_steps, val_steps.dtype)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:21:06.299271Z","iopub.execute_input":"2025-06-17T23:21:06.299512Z","iopub.status.idle":"2025-06-17T23:21:06.303476Z","shell.execute_reply.started":"2025-06-17T23:21:06.299497Z","shell.execute_reply":"2025-06-17T23:21:06.302699Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The size of one image is 96 pixels by 96 pixels, or 96^2. Take one length and make the default image size 96. ","metadata":{}},{"cell_type":"code","source":"IMAGE_SIZE = 96\n\ndatagen = ImageDataGenerator(rescale=1.0/255)\n\ntrain_gen = datagen.flow_from_directory(train_path,\n                                        target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                        batch_size=train_batch_size,\n                                        class_mode='categorical')\n\nval_gen = datagen.flow_from_directory(valid_path,\n                                        target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                        batch_size=val_batch_size,\n                                        class_mode='categorical')\n\n# Note: shuffle=False causes the test dataset to not be shuffled\ntest_gen = datagen.flow_from_directory(valid_path,\n                                        target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                        batch_size=1,\n                                        class_mode='categorical',\n                                        shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:21:07.512992Z","iopub.execute_input":"2025-06-17T23:21:07.513571Z","iopub.status.idle":"2025-06-17T23:21:11.975869Z","shell.execute_reply.started":"2025-06-17T23:21:07.513550Z","shell.execute_reply":"2025-06-17T23:21:11.975320Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kernel_size = (3,3)\npool_size= (2,2)\nfirst_filters = 32\nsecond_filters = 64\nthird_filters = 128\n\ndropout_conv = 0.3\ndropout_dense = 0.3\n\n\nmodel = Sequential()\nmodel.add(Conv2D(first_filters, kernel_size, activation = 'relu', input_shape = (96, 96, 3)))\nmodel.add(Conv2D(first_filters, kernel_size, activation = 'relu'))\nmodel.add(Conv2D(first_filters, kernel_size, activation = 'relu'))\nmodel.add(MaxPooling2D(pool_size = pool_size)) \nmodel.add(Dropout(dropout_conv))\n\nmodel.add(Conv2D(second_filters, kernel_size, activation ='relu'))\nmodel.add(Conv2D(second_filters, kernel_size, activation ='relu'))\nmodel.add(Conv2D(second_filters, kernel_size, activation ='relu'))\nmodel.add(MaxPooling2D(pool_size = pool_size))\nmodel.add(Dropout(dropout_conv))\n\nmodel.add(Conv2D(third_filters, kernel_size, activation ='relu'))\nmodel.add(Conv2D(third_filters, kernel_size, activation ='relu'))\nmodel.add(Conv2D(third_filters, kernel_size, activation ='relu'))\nmodel.add(MaxPooling2D(pool_size = pool_size))\nmodel.add(Dropout(dropout_conv))\n\nmodel.add(Flatten())\nmodel.add(Dense(256, activation = \"relu\"))\nmodel.add(Dropout(dropout_dense))\nmodel.add(Dense(2, activation = \"softmax\"))\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:23:25.458219Z","iopub.execute_input":"2025-06-17T23:23:25.458505Z","iopub.status.idle":"2025-06-17T23:23:25.665708Z","shell.execute_reply.started":"2025-06-17T23:23:25.458485Z","shell.execute_reply":"2025-06-17T23:23:25.665069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(Adam(learning_rate=0.0001), loss='binary_crossentropy', \n              metrics=['accuracy'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:23:27.635218Z","iopub.execute_input":"2025-06-17T23:23:27.635923Z","iopub.status.idle":"2025-06-17T23:23:27.643526Z","shell.execute_reply.started":"2025-06-17T23:23:27.635900Z","shell.execute_reply":"2025-06-17T23:23:27.643049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(val_gen.class_indices)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:23:28.340636Z","iopub.execute_input":"2025-06-17T23:23:28.340859Z","iopub.status.idle":"2025-06-17T23:23:28.344833Z","shell.execute_reply.started":"2025-06-17T23:23:28.340844Z","shell.execute_reply":"2025-06-17T23:23:28.344029Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id = \"model\"></a>\n# Creating the Model\n<br>\nCreating model checkpoints and reduces learning rates whenever metrics stop improving. Building the model and fitting with the data.","metadata":{}},{"cell_type":"code","source":"# Creating the model\nfilepath = \"model.h5\"\ncheckpoint = ModelCheckpoint(filepath, monitor='val_acc', verbose=1, \n                             save_best_only=True, mode='max')\n\nreduce_lr = ReduceLROnPlateau(monitor='val_acc', factor=0.5, patience=2, \n                                   verbose=1, mode='max', min_lr=0.00001)\n                              \n                              \ncallbacks_list = [checkpoint, reduce_lr]\n\n# Set the model to 15 epochs\nhistory = model.fit(train_gen, steps_per_epoch=train_steps, \n                    validation_data=val_gen,\n                    validation_steps=val_steps,\n                    epochs=15, verbose=1,\n                   callbacks=callbacks_list)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:23:29.373141Z","iopub.execute_input":"2025-06-17T23:23:29.373843Z","iopub.status.idle":"2025-06-17T23:42:43.532171Z","shell.execute_reply.started":"2025-06-17T23:23:29.373818Z","shell.execute_reply":"2025-06-17T23:42:43.531631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.metrics_names","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:43:15.985414Z","iopub.execute_input":"2025-06-17T23:43:15.986013Z","iopub.status.idle":"2025-06-17T23:43:15.990023Z","shell.execute_reply.started":"2025-06-17T23:43:15.985991Z","shell.execute_reply":"2025-06-17T23:43:15.989458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_loss= model.evaluate(test_gen, steps=len(train_labels))\n\nprint('val_loss:', val_loss)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:43:16.123907Z","iopub.execute_input":"2025-06-17T23:43:16.124082Z","iopub.status.idle":"2025-06-17T23:44:09.831975Z","shell.execute_reply.started":"2025-06-17T23:43:16.124066Z","shell.execute_reply":"2025-06-17T23:44:09.831369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get loss and accuracy rates\nprint('val_loss:', val_loss[0])\nprint('val_accuracy', val_loss[1])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:44:41.943128Z","iopub.execute_input":"2025-06-17T23:44:41.943388Z","iopub.status.idle":"2025-06-17T23:44:41.947547Z","shell.execute_reply.started":"2025-06-17T23:44:41.943368Z","shell.execute_reply":"2025-06-17T23:44:41.946854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the loss and accuracy visuals\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs = range(1, len(acc) + 1)\n\nplt.plot(epochs, loss, 'bo', label='Training loss')\nplt.plot(epochs, val_loss, 'b', label='Validation loss')\nplt.title('Training and validation loss')\nplt.legend()\nplt.figure()\n\nplt.plot(epochs, acc, 'bo', label='Training acc')\nplt.plot(epochs, val_acc, 'b', label='Validation acc')\nplt.title('Training and validation accuracy')\nplt.legend()\nplt.figure()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:44:43.912295Z","iopub.execute_input":"2025-06-17T23:44:43.912846Z","iopub.status.idle":"2025-06-17T23:44:44.222178Z","shell.execute_reply.started":"2025-06-17T23:44:43.912824Z","shell.execute_reply":"2025-06-17T23:44:44.221624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# make a prediction\npredictions = model.predict(test_gen)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:44:51.658346Z","iopub.execute_input":"2025-06-17T23:44:51.659014Z","iopub.status.idle":"2025-06-17T23:46:36.699400Z","shell.execute_reply.started":"2025-06-17T23:44:51.658993Z","shell.execute_reply":"2025-06-17T23:46:36.698633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions.shape\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:46:46.233336Z","iopub.execute_input":"2025-06-17T23:46:46.233637Z","iopub.status.idle":"2025-06-17T23:46:46.238133Z","shell.execute_reply.started":"2025-06-17T23:46:46.233614Z","shell.execute_reply":"2025-06-17T23:46:46.237446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_preds = pd.DataFrame(predictions, columns=['negative', 'positive'])\n\ndata_preds.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:46:47.305283Z","iopub.execute_input":"2025-06-17T23:46:47.305995Z","iopub.status.idle":"2025-06-17T23:46:47.313481Z","shell.execute_reply.started":"2025-06-17T23:46:47.305971Z","shell.execute_reply":"2025-06-17T23:46:47.312735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_preds.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:46:48.476506Z","iopub.execute_input":"2025-06-17T23:46:48.477193Z","iopub.status.idle":"2025-06-17T23:46:48.481162Z","shell.execute_reply.started":"2025-06-17T23:46:48.477169Z","shell.execute_reply":"2025-06-17T23:46:48.480438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(test_gen.filenames)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:46:48.697710Z","iopub.execute_input":"2025-06-17T23:46:48.698305Z","iopub.status.idle":"2025-06-17T23:46:48.702324Z","shell.execute_reply.started":"2025-06-17T23:46:48.698285Z","shell.execute_reply":"2025-06-17T23:46:48.701683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the true labels\ny_true = test_gen.classes\n\n# Get the predicted labels as probabilities\ny_pred = data_preds['positive']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:46:49.761399Z","iopub.execute_input":"2025-06-17T23:46:49.761940Z","iopub.status.idle":"2025-06-17T23:46:49.765472Z","shell.execute_reply.started":"2025-06-17T23:46:49.761916Z","shell.execute_reply":"2025-06-17T23:46:49.764652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\n# Get ROC Score\nroc_auc_score(y_true, y_pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:46:50.771320Z","iopub.execute_input":"2025-06-17T23:46:50.771608Z","iopub.status.idle":"2025-06-17T23:46:50.784041Z","shell.execute_reply.started":"2025-06-17T23:46:50.771567Z","shell.execute_reply":"2025-06-17T23:46:50.783283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_labels = test_gen.classes\ntest_labels.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:46:52.095320Z","iopub.execute_input":"2025-06-17T23:46:52.095914Z","iopub.status.idle":"2025-06-17T23:46:52.100836Z","shell.execute_reply.started":"2025-06-17T23:46:52.095881Z","shell.execute_reply":"2025-06-17T23:46:52.100120Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\n\n\n# Get max value in row using argmax\ncm = confusion_matrix(test_labels, predictions.argmax(axis=1))\n# Print the label associated with each class\ntest_gen.class_indices","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:46:53.932875Z","iopub.execute_input":"2025-06-17T23:46:53.933539Z","iopub.status.idle":"2025-06-17T23:46:53.940461Z","shell.execute_reply.started":"2025-06-17T23:46:53.933515Z","shell.execute_reply":"2025-06-17T23:46:53.939911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cm_display = ConfusionMatrixDisplay(confusion_matrix=cm)\ncm_display.plot()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:46:55.511325Z","iopub.execute_input":"2025-06-17T23:46:55.512070Z","iopub.status.idle":"2025-06-17T23:46:55.656571Z","shell.execute_reply.started":"2025-06-17T23:46:55.512043Z","shell.execute_reply":"2025-06-17T23:46:55.655931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remove base_dir and create test_dir\n\nshutil.rmtree('base_dir')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:47:01.698528Z","iopub.execute_input":"2025-06-17T23:47:01.699046Z","iopub.status.idle":"2025-06-17T23:47:05.959172Z","shell.execute_reply.started":"2025-06-17T23:47:01.699024Z","shell.execute_reply":"2025-06-17T23:47:05.958376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# create test_dir\ntest_dir = 'test_dir'\nos.makedirs(test_dir, exist_ok=True)\n\ntest_imgs = os.path.join(test_dir, 'test_imgs')\nos.makedirs(test_imgs, exist_ok=True)\n\nos.listdir('test_dir')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:47:05.960167Z","iopub.execute_input":"2025-06-17T23:47:05.960835Z","iopub.status.idle":"2025-06-17T23:47:05.970108Z","shell.execute_reply.started":"2025-06-17T23:47:05.960805Z","shell.execute_reply":"2025-06-17T23:47:05.969471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Transfer the test images into image_dir\n\ntest_list = os.listdir('../input/histopathologic-cancer-detection/test')\n\nfor image in test_list:\n    \n    fname = image\n    \n    # source path to image\n    src = os.path.join('../input/histopathologic-cancer-detection/test', fname)\n    # destination path to image\n    dst = os.path.join(test_imgs, fname)\n    # copy the image from the source to the destination\n    shutil.copyfile(src, dst)\n# check that the images are now in the test_images\n# Total is 57458 images in the test_imgs folder\nlen(os.listdir('test_dir/test_imgs'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:47:07.270972Z","iopub.execute_input":"2025-06-17T23:47:07.271770Z","iopub.status.idle":"2025-06-17T23:49:30.949963Z","shell.execute_reply.started":"2025-06-17T23:47:07.271741Z","shell.execute_reply":"2025-06-17T23:49:30.949203Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id = \"validation\"></a>\n# Testing and Validating the Model\n<br>\nGet the results from the test predictions and prepare to submit the file with the cancerous images and respective IDs.","metadata":{}},{"cell_type":"code","source":"test_path ='test_dir'\n# Adjust the path to put images into the test_imgs directory.\ntest_gen = datagen.flow_from_directory(test_path,\n                                        target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                        batch_size=1,\n                                        class_mode='categorical',\n                                        shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:50:48.764138Z","iopub.execute_input":"2025-06-17T23:50:48.764650Z","iopub.status.idle":"2025-06-17T23:50:49.507936Z","shell.execute_reply.started":"2025-06-17T23:50:48.764627Z","shell.execute_reply":"2025-06-17T23:50:49.507332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_imgs_ct = 57458\ntest_predictions = model.predict(test_gen, steps=test_imgs_ct, verbose=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:50:52.012432Z","iopub.execute_input":"2025-06-17T23:50:52.012728Z","iopub.status.idle":"2025-06-17T23:53:24.451708Z","shell.execute_reply.started":"2025-06-17T23:50:52.012707Z","shell.execute_reply":"2025-06-17T23:53:24.451056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(test_predictions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:54:31.978991Z","iopub.execute_input":"2025-06-17T23:54:31.979270Z","iopub.status.idle":"2025-06-17T23:54:31.983356Z","shell.execute_reply.started":"2025-06-17T23:54:31.979246Z","shell.execute_reply":"2025-06-17T23:54:31.982816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create dataframe with all test's predictions\ntest_preds= pd.DataFrame(test_predictions, columns=['negative', 'positive'])\ntest_preds.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:54:32.154960Z","iopub.execute_input":"2025-06-17T23:54:32.155157Z","iopub.status.idle":"2025-06-17T23:54:32.163049Z","shell.execute_reply.started":"2025-06-17T23:54:32.155142Z","shell.execute_reply":"2025-06-17T23:54:32.162366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get all filenames and add to the test_preds dataframe\ntest_fnames = test_gen.filenames\n\ntest_preds['fnames'] = test_fnames\ntest_preds.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:54:35.641064Z","iopub.execute_input":"2025-06-17T23:54:35.641620Z","iopub.status.idle":"2025-06-17T23:54:35.653454Z","shell.execute_reply.started":"2025-06-17T23:54:35.641597Z","shell.execute_reply":"2025-06-17T23:54:35.652897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_preds.fnames","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:54:36.315953Z","iopub.execute_input":"2025-06-17T23:54:36.316740Z","iopub.status.idle":"2025-06-17T23:54:36.322249Z","shell.execute_reply.started":"2025-06-17T23:54:36.316716Z","shell.execute_reply":"2025-06-17T23:54:36.321482Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"''' Similar to the sample submission, create an id column\nGet rid of the 'test_imgs/' portion of each of the file names\nto get just the file name as well as rid the '.tif' at the end of \neach cell in fname column\n'''\n\n\ndef get_id(x):\n    \n    # split into a list\n    a = x.split('/')\n    # split into a list\n    b = a[1].split('.')\n    extracted_id = b[0]\n    \n    return extracted_id\n\ntest_preds['id'] = test_preds['fnames'].apply(get_id)\n\ntest_preds.head()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:54:37.000427Z","iopub.execute_input":"2025-06-17T23:54:37.001081Z","iopub.status.idle":"2025-06-17T23:54:37.033301Z","shell.execute_reply.started":"2025-06-17T23:54:37.001061Z","shell.execute_reply":"2025-06-17T23:54:37.032780Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sample of a fnames cell\ntest_preds['fnames'][0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:54:38.331218Z","iopub.execute_input":"2025-06-17T23:54:38.331685Z","iopub.status.idle":"2025-06-17T23:54:38.335821Z","shell.execute_reply.started":"2025-06-17T23:54:38.331662Z","shell.execute_reply":"2025-06-17T23:54:38.335195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# fnames column without the 'test_imgs/' at beginning or the trailing '.tif' at end\ntest_preds['id']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:54:39.321777Z","iopub.execute_input":"2025-06-17T23:54:39.322034Z","iopub.status.idle":"2025-06-17T23:54:39.327549Z","shell.execute_reply.started":"2025-06-17T23:54:39.322015Z","shell.execute_reply":"2025-06-17T23:54:39.326954Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id = \"Submit the file\"></a>\n# Submission","metadata":{}},{"cell_type":"code","source":"# Get predicted labels with cancerous/positive images and their respective ids\ny_pred = test_preds['positive']\nimg_ids = test_preds['id']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:54:40.942356Z","iopub.execute_input":"2025-06-17T23:54:40.943042Z","iopub.status.idle":"2025-06-17T23:54:40.946265Z","shell.execute_reply.started":"2025-06-17T23:54:40.943019Z","shell.execute_reply":"2025-06-17T23:54:40.945458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create submission file dataframe\nsubmission = pd.DataFrame({'id':img_ids, \n                           'label':y_pred, \n                          }).set_index('id')\n\n# Create CSV file to be submitted\nsubmission.to_csv('patch_preds.csv', columns=['label']) \nsubmission.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:54:41.449487Z","iopub.execute_input":"2025-06-17T23:54:41.449785Z","iopub.status.idle":"2025-06-17T23:54:41.593926Z","shell.execute_reply.started":"2025-06-17T23:54:41.449764Z","shell.execute_reply":"2025-06-17T23:54:41.593317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remove all contents in test_dir\nshutil.rmtree('test_dir')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:54:42.407704Z","iopub.execute_input":"2025-06-17T23:54:42.408325Z","iopub.status.idle":"2025-06-17T23:54:44.190688Z","shell.execute_reply.started":"2025-06-17T23:54:42.408304Z","shell.execute_reply":"2025-06-17T23:54:44.189974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cm = confusion_matrix(test_labels, predictions.argmax(axis=1))\n# Print the label associated with each class\ntest_gen.class_indices\ncm_display = ConfusionMatrixDisplay(confusion_matrix=cm)\ncm_display.plot()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:54:44.192044Z","iopub.execute_input":"2025-06-17T23:54:44.192327Z","iopub.status.idle":"2025-06-17T23:54:44.343975Z","shell.execute_reply.started":"2025-06-17T23:54:44.192305Z","shell.execute_reply":"2025-06-17T23:54:44.343285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compare subplots from previous confusion matrix. \nax= plt.subplot()\nsns.heatmap(cm, annot=True, ax = ax); #annot=True to annotate cells\n\n# labels, title and ticks\nax.set_xlabel('Predicted labels')\nax.set_ylabel('True labels'); \nax.set_title('Confusion Matrix')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-17T23:54:44.493619Z","iopub.execute_input":"2025-06-17T23:54:44.494205Z","iopub.status.idle":"2025-06-17T23:54:44.652712Z","shell.execute_reply.started":"2025-06-17T23:54:44.494186Z","shell.execute_reply":"2025-06-17T23:54:44.652079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}