{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"},{"sourceId":9247326,"sourceType":"datasetVersion","datasetId":5594143}],"dockerImageVersionId":30747,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Import Packages","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport os\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import shuffle\nfrom sklearn.metrics import confusion_matrix\nimport pickle\nimport cv2\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.models import Sequential, load_model\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping,ReduceLROnPlateau\nfrom tensorflow.keras import backend as K","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:41:29.859003Z","iopub.execute_input":"2024-08-26T04:41:29.859647Z","iopub.status.idle":"2024-08-26T04:41:33.972567Z","shell.execute_reply.started":"2024-08-26T04:41:29.859616Z","shell.execute_reply":"2024-08-26T04:41:33.971782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Error Prevention","metadata":{}},{"cell_type":"code","source":"# Clearing sessions prevents errors\nK.clear_session()","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:41:33.974494Z","iopub.execute_input":"2024-08-26T04:41:33.975227Z","iopub.status.idle":"2024-08-26T04:41:34.137779Z","shell.execute_reply.started":"2024-08-26T04:41:33.975191Z","shell.execute_reply":"2024-08-26T04:41:34.136755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Helper Functions","metadata":{}},{"cell_type":"code","source":"# Define functions to merge history and plot training and validation after epoch runs\n\ndef merge_history(hlist):\n    history = {}\n    for k in hlist[0].history.keys():\n        history[k] = sum([h.history[k] for h in hlist], [])\n    return history\n\ndef vis_training(h, start=1):\n    epoch_range = range(start, len(h['loss'])+1)\n    s = slice(start-1, None)\n\n    plt.figure(figsize=[14,4])\n\n    n = int(len(h.keys()) / 2)\n\n    for i in range(n):\n        k = list(h.keys())[i]\n        plt.subplot(1,n,i+1)\n        plt.plot(epoch_range, h[k][s], label='Training')\n        plt.plot(epoch_range, h['val_' + k][s], label='Validation')\n        plt.xlabel('Epoch'); plt.ylabel(k); plt.title(k)\n        plt.grid()\n        plt.legend()\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:41:34.139273Z","iopub.execute_input":"2024-08-26T04:41:34.140371Z","iopub.status.idle":"2024-08-26T04:41:34.153740Z","shell.execute_reply.started":"2024-08-26T04:41:34.140333Z","shell.execute_reply":"2024-08-26T04:41:34.152881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load DataFrames ","metadata":{}},{"cell_type":"code","source":"# Create df and identify train_path\n\ntrain = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv', dtype=str)\n# 0 = non-cancerous\n# 1 = cancerous\n\ntrain_path = ('/kaggle/input/histopathologic-cancer-detection/train/')","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:41:34.154732Z","iopub.execute_input":"2024-08-26T04:41:34.155035Z","iopub.status.idle":"2024-08-26T04:41:34.384017Z","shell.execute_reply.started":"2024-08-26T04:41:34.155010Z","shell.execute_reply":"2024-08-26T04:41:34.383202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Exploratory Data Analysis showed an imbalanced dataset. The sample size has been set to a value that provides equal representation for both labels prior to initiating a model. ","metadata":{}},{"cell_type":"code","source":"# Create sample size to balance data in both classes\n# Create new df with balanced data\n\nsample_size = 85000\n\nclass_0 = train[train['label'] == '0'].sample(sample_size, random_state=1)\nclass_1 = train[train['label'] == '1'].sample(sample_size, random_state=1)\n\n\nbalanced_train = pd.concat([class_0, class_1], axis=0).reset_index(drop=True)\n\nbalanced_train = shuffle(balanced_train)\n\n# Verify data is balanced with bar chart\n\nplt.figure(figsize=(6, 4))\nclass_dist = balanced_train.groupby('label').size().sort_values(ascending=True)\ncolors = plt.cm.viridis(np.linspace(0, len(class_dist)))\nclass_dist.plot(kind='bar', color=colors)  \nplt.title('Label Distribution', fontsize=20)\nplt.xlabel('Label', fontsize=16)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:41:34.386388Z","iopub.execute_input":"2024-08-26T04:41:34.386682Z","iopub.status.idle":"2024-08-26T04:41:34.749311Z","shell.execute_reply.started":"2024-08-26T04:41:34.386658Z","shell.execute_reply":"2024-08-26T04:41:34.747972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data split\ntrain_df, valid_df = train_test_split(balanced_train, test_size=0.2, stratify=balanced_train['label'], random_state=1)\nprint(train_df.shape)\nprint(valid_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:41:34.751627Z","iopub.execute_input":"2024-08-26T04:41:34.752506Z","iopub.status.idle":"2024-08-26T04:41:34.990658Z","shell.execute_reply.started":"2024-08-26T04:41:34.752461Z","shell.execute_reply":"2024-08-26T04:41:34.989717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data augmentation including rotations, horizontal and vertical shifts, shearing, zoom, flipping and filling new pixels with the value of the nearest pixel\ntrain_datagen = tf.keras.preprocessing.image.ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=40,   \n    width_shift_range=0.2,   \n    height_shift_range=0.2,   \n    shear_range=0.2,   \n    zoom_range=0.2,   \n    horizontal_flip=True,   \n    fill_mode='nearest'   \n)\nvalid_datagen = tf.keras.preprocessing.image.ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=40,   \n    width_shift_range=0.2,   \n    height_shift_range=0.2,   \n    shear_range=0.2,   \n    zoom_range=0.2,   \n    horizontal_flip=True,   \n    fill_mode='nearest'   \n)","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:41:34.991671Z","iopub.execute_input":"2024-08-26T04:41:34.991952Z","iopub.status.idle":"2024-08-26T04:41:34.998054Z","shell.execute_reply.started":"2024-08-26T04:41:34.991911Z","shell.execute_reply":"2024-08-26T04:41:34.997133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lowered batch size repeatedly down to 32, adding the .tif extension to filename\n\nBATCH_SIZE = 32\n\ntrain_df['id'] += '.tif'\nvalid_df['id'] += '.tif'\n\ntrain_loader = train_datagen.flow_from_dataframe(\n    dataframe = train_df,\n    directory = train_path,\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 42,\n    shuffle = True,\n    class_mode = 'binary',\n    target_size = (96, 96)\n)\n\nvalid_loader = train_datagen.flow_from_dataframe(\n    dataframe = valid_df,\n    directory = train_path,\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 42,\n    shuffle = True,\n    class_mode = 'binary',\n    target_size = (96, 96)\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:41:34.999150Z","iopub.execute_input":"2024-08-26T04:41:34.999439Z","iopub.status.idle":"2024-08-26T04:43:44.857957Z","shell.execute_reply.started":"2024-08-26T04:41:34.999417Z","shell.execute_reply":"2024-08-26T04:43:44.857181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TR_STEPS = len(train_loader)//BATCH_SIZE\nVA_STEPS = len(valid_loader)//BATCH_SIZE\n\nprint(TR_STEPS)\nprint(VA_STEPS)","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:43:44.859091Z","iopub.execute_input":"2024-08-26T04:43:44.859385Z","iopub.status.idle":"2024-08-26T04:43:44.864872Z","shell.execute_reply.started":"2024-08-26T04:43:44.859361Z","shell.execute_reply":"2024-08-26T04:43:44.864060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CNN Model","metadata":{}},{"cell_type":"code","source":"# Create model \ncnn = Sequential([\n    Input(shape=(96,96,3)),\n    \n    Conv2D(64, (3,3), padding='same', activation = 'relu'),\n    MaxPooling2D(2,2),\n    BatchNormalization(),\n    Conv2D(128, (3,3), padding='same',activation = 'relu'),\n    MaxPooling2D(2,2),\n    BatchNormalization(),\n    Conv2D(256, (3,3), padding='same',activation = 'relu'),\n    MaxPooling2D(2,2),\n    BatchNormalization(),\n    GlobalAveragePooling2D(),\n    Dense(512, activation = 'relu'),\n    Dropout(0.5),\n    Dense(1, activation = 'sigmoid')\n])\n\n    \ncnn.summary()","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:43:44.866051Z","iopub.execute_input":"2024-08-26T04:43:44.866345Z","iopub.status.idle":"2024-08-26T04:43:45.824059Z","shell.execute_reply.started":"2024-08-26T04:43:44.866322Z","shell.execute_reply":"2024-08-26T04:43:45.823108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Network","metadata":{}},{"cell_type":"code","source":"# Define Callbacks\nlr_scheduler =  tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', patience=5, factor=0.5)\nearly_stopping =  tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=10, mode='min')","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:43:45.825212Z","iopub.execute_input":"2024-08-26T04:43:45.825498Z","iopub.status.idle":"2024-08-26T04:43:45.830486Z","shell.execute_reply.started":"2024-08-26T04:43:45.825474Z","shell.execute_reply":"2024-08-26T04:43:45.829617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set optimizer, lr, loss function and metrics \nopt = tf.keras.optimizers.Adam(learning_rate=0.001)\ncnn.compile(optimizer=opt,\n              loss=tf.keras.losses.BinaryCrossentropy(),\n              metrics=[tf.keras.metrics.AUC(), 'accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:43:45.831781Z","iopub.execute_input":"2024-08-26T04:43:45.832215Z","iopub.status.idle":"2024-08-26T04:43:45.856830Z","shell.execute_reply.started":"2024-08-26T04:43:45.832183Z","shell.execute_reply":"2024-08-26T04:43:45.856174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set random seed for reproducibility and fit the model, adding callbacks\n\nnp.random.seed(1)\ntf.random.set_seed(1)\n\nh1 = cnn.fit(\nx = train_loader, \nsteps_per_epoch = TR_STEPS, \nepochs = 30, \nvalidation_data = valid_loader, \nvalidation_steps = VA_STEPS, \ncallbacks = [lr_scheduler, early_stopping],\nverbose = 1\n)","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:43:45.857794Z","iopub.execute_input":"2024-08-26T04:43:45.858062Z","iopub.status.idle":"2024-08-26T04:57:42.220209Z","shell.execute_reply.started":"2024-08-26T04:43:45.858039Z","shell.execute_reply":"2024-08-26T04:57:42.219312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize the results using helper functions defined above\nhistory = merge_history([h1])\nvis_training(history)","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:57:42.224580Z","iopub.execute_input":"2024-08-26T04:57:42.224990Z","iopub.status.idle":"2024-08-26T04:57:43.156718Z","shell.execute_reply.started":"2024-08-26T04:57:42.224949Z","shell.execute_reply":"2024-08-26T04:57:43.155862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnn.save('v28.h5')\npickle.dump(history, open(f'v28.h5.pk1','wb'))","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:57:43.157892Z","iopub.execute_input":"2024-08-26T04:57:43.158286Z","iopub.status.idle":"2024-08-26T04:57:43.232953Z","shell.execute_reply.started":"2024-08-26T04:57:43.158258Z","shell.execute_reply":"2024-08-26T04:57:43.232064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# Load Test Dataframe and Directory ","metadata":{}},{"cell_type":"code","source":"# Create test df, identify test directory path and add .tif to filename\n\ntest = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/sample_submission.csv')\ntest_directory = '/kaggle/input/histopathologic-cancer-detection/test'\ntest['id'] = test['id'] + '.tif'","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:57:43.234160Z","iopub.execute_input":"2024-08-26T04:57:43.234519Z","iopub.status.idle":"2024-08-26T04:57:43.320866Z","shell.execute_reply.started":"2024-08-26T04:57:43.234476Z","shell.execute_reply":"2024-08-26T04:57:43.319919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Generator","metadata":{}},{"cell_type":"code","source":"# Repeat from above, except shuffle is False and the data augmentation is not included here\n\nBATCH_SIZE = 32\n\ntest_datagen = ImageDataGenerator(rescale=1/255)\n\ntest_loader = test_datagen.flow_from_dataframe(\n    dataframe = test,\n    directory = test_directory,\n    x_col = 'id', \n    batch_size = BATCH_SIZE, \n    shuffle = False,\n    class_mode = None,\n    target_size = (96,96)\n)","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:57:43.322128Z","iopub.execute_input":"2024-08-26T04:57:43.322439Z","iopub.status.idle":"2024-08-26T04:59:44.063785Z","shell.execute_reply.started":"2024-08-26T04:57:43.322412Z","shell.execute_reply":"2024-08-26T04:59:44.062980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Model","metadata":{}},{"cell_type":"code","source":"# Load the model \ncnn = keras.models.load_model('/kaggle/input/week8v28/v28.h5')\ncnn.summary()","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:59:44.065125Z","iopub.execute_input":"2024-08-26T04:59:44.065962Z","iopub.status.idle":"2024-08-26T04:59:44.350159Z","shell.execute_reply.started":"2024-08-26T04:59:44.065903Z","shell.execute_reply":"2024-08-26T04:59:44.349250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test Predictions","metadata":{}},{"cell_type":"code","source":"# Call the predict function on test_loader and store in test_probs variable\n# Review shape of variable\ntest_probs = cnn.predict(test_loader)\nprint(test_probs.shape)","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:59:44.351549Z","iopub.execute_input":"2024-08-26T04:59:44.352237Z","iopub.status.idle":"2024-08-26T05:04:57.278224Z","shell.execute_reply.started":"2024-08-26T04:59:44.352201Z","shell.execute_reply":"2024-08-26T05:04:57.277178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print out 10 test probabilities rounded 2 decimal places\nprint(test_probs[:10,].round(2))","metadata":{"execution":{"iopub.status.busy":"2024-08-26T05:04:57.279611Z","iopub.execute_input":"2024-08-26T05:04:57.279909Z","iopub.status.idle":"2024-08-26T05:04:57.285857Z","shell.execute_reply.started":"2024-08-26T05:04:57.279883Z","shell.execute_reply":"2024-08-26T05:04:57.284871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred = np.argmax(test_probs, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-08-26T05:04:57.287351Z","iopub.execute_input":"2024-08-26T05:04:57.288031Z","iopub.status.idle":"2024-08-26T05:04:57.297409Z","shell.execute_reply.started":"2024-08-26T05:04:57.287996Z","shell.execute_reply":"2024-08-26T05:04:57.296566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare Submission","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/sample_submission.csv')\nsubmission.label = test_probs\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-26T05:04:57.298471Z","iopub.execute_input":"2024-08-26T05:04:57.298746Z","iopub.status.idle":"2024-08-26T05:04:57.366794Z","shell.execute_reply.started":"2024-08-26T05:04:57.298722Z","shell.execute_reply":"2024-08-26T05:04:57.365879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False, header=True)","metadata":{"execution":{"iopub.status.busy":"2024-08-26T05:04:57.368209Z","iopub.execute_input":"2024-08-26T05:04:57.368981Z","iopub.status.idle":"2024-08-26T05:04:57.578623Z","shell.execute_reply.started":"2024-08-26T05:04:57.368923Z","shell.execute_reply":"2024-08-26T05:04:57.577859Z"},"trusted":true},"execution_count":null,"outputs":[]}]}