{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"},{"sourceId":9242092,"sourceType":"datasetVersion","datasetId":5590636}],"dockerImageVersionId":30747,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"##### Import Packages","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport os\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import shuffle\nfrom sklearn.metrics import confusion_matrix\nimport pickle\nimport cv2\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential, load_model, save_model\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping,ReduceLROnPlateau, ModelCheckpoint\nfrom tensorflow.keras import backend as K\nfrom tensorflow import keras\nfrom tensorflow.keras.models import Sequential, load_model","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:18:04.146616Z","iopub.execute_input":"2024-08-25T05:18:04.146941Z","iopub.status.idle":"2024-08-25T05:18:08.210748Z","shell.execute_reply.started":"2024-08-25T05:18:04.146915Z","shell.execute_reply":"2024-08-25T05:18:08.209820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"K.clear_session()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:18:08.212433Z","iopub.execute_input":"2024-08-25T05:18:08.212938Z","iopub.status.idle":"2024-08-25T05:18:08.375351Z","shell.execute_reply.started":"2024-08-25T05:18:08.212912Z","shell.execute_reply":"2024-08-25T05:18:08.374283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_size = 96\nsample_size = 85000","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:18:08.376725Z","iopub.execute_input":"2024-08-25T05:18:08.377086Z","iopub.status.idle":"2024-08-25T05:18:08.385431Z","shell.execute_reply.started":"2024-08-25T05:18:08.377055Z","shell.execute_reply":"2024-08-25T05:18:08.384543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Helper Functions","metadata":{}},{"cell_type":"code","source":"def merge_history(hlist):\n    history = {}\n    for k in hlist[0].history.keys():\n        history[k] = sum([h.history[k] for h in hlist], [])\n    return history\n\ndef vis_training(h, start=1):\n    epoch_range = range(start, len(h['loss'])+1)\n    s = slice(start-1, None)\n\n    plt.figure(figsize=[14,4])\n\n    n = int(len(h.keys()) / 2)\n\n    for i in range(n):\n        k = list(h.keys())[i]\n        plt.subplot(1,n,i+1)\n        plt.plot(epoch_range, h[k][s], label='Training')\n        plt.plot(epoch_range, h['val_' + k][s], label='Validation')\n        plt.xlabel('Epoch'); plt.ylabel(k); plt.title(k)\n        plt.grid()\n        plt.legend()\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:18:08.386631Z","iopub.execute_input":"2024-08-25T05:18:08.386915Z","iopub.status.idle":"2024-08-25T05:18:08.396290Z","shell.execute_reply.started":"2024-08-25T05:18:08.386894Z","shell.execute_reply":"2024-08-25T05:18:08.395436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load DataFrames ","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv', dtype=str)\n# 0 = non-cancerous\n# 1 = cancerous\n\ntrain_path = ('/kaggle/input/histopathologic-cancer-detection/train/')","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:18:08.398946Z","iopub.execute_input":"2024-08-25T05:18:08.399230Z","iopub.status.idle":"2024-08-25T05:18:08.621536Z","shell.execute_reply.started":"2024-08-25T05:18:08.399203Z","shell.execute_reply":"2024-08-25T05:18:08.620733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.shape)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:18:08.622704Z","iopub.execute_input":"2024-08-25T05:18:08.622974Z","iopub.status.idle":"2024-08-25T05:18:08.628123Z","shell.execute_reply.started":"2024-08-25T05:18:08.622953Z","shell.execute_reply":"2024-08-25T05:18:08.627147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Review the distribution\n\ntrain['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:18:08.629475Z","iopub.execute_input":"2024-08-25T05:18:08.629802Z","iopub.status.idle":"2024-08-25T05:18:08.666804Z","shell.execute_reply.started":"2024-08-25T05:18:08.629778Z","shell.execute_reply":"2024-08-25T05:18:08.665798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Balance the distribution\n\nclass_0 = train[train['label'] == '0'].sample(sample_size, random_state=1)\nclass_1 = train[train['label'] == '1'].sample(sample_size, random_state=1)\n\n\nbalanced_train = pd.concat([class_0, class_1], axis=0).reset_index(drop=True)\n\nbalanced_train = shuffle(balanced_train)\n\nbalanced_train.label.value_counts().sort_values().plot(kind = 'bar')","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:18:08.668158Z","iopub.execute_input":"2024-08-25T05:18:08.668479Z","iopub.status.idle":"2024-08-25T05:18:09.051621Z","shell.execute_reply.started":"2024-08-25T05:18:08.668449Z","shell.execute_reply":"2024-08-25T05:18:09.050709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the data\n\ntrain_df, valid_df = train_test_split(balanced_train, test_size=0.2, stratify=balanced_train['label'], random_state=1)\nprint(train_df.shape)\nprint(valid_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:18:09.052971Z","iopub.execute_input":"2024-08-25T05:18:09.053346Z","iopub.status.idle":"2024-08-25T05:18:09.317925Z","shell.execute_reply.started":"2024-08-25T05:18:09.053305Z","shell.execute_reply":"2024-08-25T05:18:09.316784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = tf.keras.preprocessing.image.ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=40,   \n    width_shift_range=0.2,   \n    height_shift_range=0.2,   \n    shear_range=0.2,   \n    zoom_range=0.2,   \n    horizontal_flip=True,   \n    fill_mode='nearest'   \n)\nvalid_datagen = tf.keras.preprocessing.image.ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=40,   \n    width_shift_range=0.2,   \n    height_shift_range=0.2,   \n    shear_range=0.2,   \n    zoom_range=0.2,   \n    horizontal_flip=True,   \n    fill_mode='nearest'   \n)","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:18:09.319670Z","iopub.execute_input":"2024-08-25T05:18:09.320038Z","iopub.status.idle":"2024-08-25T05:18:09.327225Z","shell.execute_reply.started":"2024-08-25T05:18:09.320005Z","shell.execute_reply":"2024-08-25T05:18:09.326021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 64\n\ntrain_df['id'] += '.tif'\nvalid_df['id'] += '.tif'\n\ntrain_loader = train_datagen.flow_from_dataframe(\n    dataframe = train_df,\n    directory = train_path,\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 42,\n    shuffle = True,\n    class_mode = 'binary',\n    target_size = (img_size, img_size)\n)\n\nvalid_loader = train_datagen.flow_from_dataframe(\n    dataframe = valid_df,\n    directory = train_path,\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 42,\n    shuffle = True,\n    class_mode = 'binary',\n    target_size = (img_size, img_size)\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:18:09.328819Z","iopub.execute_input":"2024-08-25T05:18:09.330079Z","iopub.status.idle":"2024-08-25T05:21:26.448753Z","shell.execute_reply.started":"2024-08-25T05:18:09.330019Z","shell.execute_reply":"2024-08-25T05:21:26.447742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TR_STEPS = len(train_loader)//BATCH_SIZE\nVA_STEPS = len(valid_loader)//BATCH_SIZE\n\nprint(TR_STEPS)\nprint(VA_STEPS)","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:21:26.450106Z","iopub.execute_input":"2024-08-25T05:21:26.450996Z","iopub.status.idle":"2024-08-25T05:21:26.456215Z","shell.execute_reply.started":"2024-08-25T05:21:26.450961Z","shell.execute_reply":"2024-08-25T05:21:26.455325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CNN Model","metadata":{}},{"cell_type":"code","source":"cnn = Sequential([\n    Input(shape=(96,96,3)),\n    \n    Conv2D(64, (3,3), activation = 'relu'),\n    BatchNormalization(),\n    MaxPooling2D(2,2),\n    Conv2D(128, (3,3), activation = 'relu'),\n    BatchNormalization(),\n    MaxPooling2D(2,2),\n    Conv2D(256, (3,3), activation = 'relu'),\n    BatchNormalization(),\n    MaxPooling2D(2,2),\n    GlobalAveragePooling2D(),\n    Dense(512, activation = 'relu'),\n    Dropout(0.5),\n    Dense(1, activation = 'sigmoid')\n])\n\n    \ncnn.summary()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:21:57.045881Z","iopub.execute_input":"2024-08-25T05:21:57.046708Z","iopub.status.idle":"2024-08-25T05:21:57.734078Z","shell.execute_reply.started":"2024-08-25T05:21:57.046676Z","shell.execute_reply":"2024-08-25T05:21:57.733128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Network","metadata":{}},{"cell_type":"code","source":"opt = tf.keras.optimizers.Adam(learning_rate=0.001)\n\ncnn.compile(optimizer=opt,\n              loss=tf.keras.losses.BinaryCrossentropy(),\n              metrics=[tf.keras.metrics.AUC(), 'accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:22:05.285917Z","iopub.execute_input":"2024-08-25T05:22:05.286610Z","iopub.status.idle":"2024-08-25T05:22:05.309318Z","shell.execute_reply.started":"2024-08-25T05:22:05.286577Z","shell.execute_reply":"2024-08-25T05:22:05.308618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(1)\ntf.random.set_seed(1)\n\n\nh1 = cnn.fit(\nx = train_loader, \nsteps_per_epoch = TR_STEPS, \nepochs = 30, \nvalidation_data = valid_loader, \nvalidation_steps = VA_STEPS, \nverbose = 1\n)","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:22:08.093396Z","iopub.execute_input":"2024-08-25T05:22:08.093733Z","iopub.status.idle":"2024-08-25T05:26:52.051211Z","shell.execute_reply.started":"2024-08-25T05:22:08.093709Z","shell.execute_reply":"2024-08-25T05:26:52.050364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = merge_history([h1])\nvis_training(history)","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:26:57.945050Z","iopub.execute_input":"2024-08-25T05:26:57.945979Z","iopub.status.idle":"2024-08-25T05:26:58.849178Z","shell.execute_reply.started":"2024-08-25T05:26:57.945943Z","shell.execute_reply":"2024-08-25T05:26:58.848114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnn.save('v25v1.h5')\npickle.dump(history, open(f'v25v1.h5.pk1','wb'))","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:28:02.631551Z","iopub.execute_input":"2024-08-25T05:28:02.632622Z","iopub.status.idle":"2024-08-25T05:28:02.699949Z","shell.execute_reply.started":"2024-08-25T05:28:02.632584Z","shell.execute_reply":"2024-08-25T05:28:02.699189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# Load Test Dataframe and Directory ","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/sample_submission.csv')\ntest_directory = '/kaggle/input/histopathologic-cancer-detection/test'\ntest['id'] = test['id'] + '.tif'","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:38:26.548824Z","iopub.execute_input":"2024-08-25T05:38:26.549693Z","iopub.status.idle":"2024-08-25T05:38:26.624372Z","shell.execute_reply.started":"2024-08-25T05:38:26.549658Z","shell.execute_reply":"2024-08-25T05:38:26.623005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Generator","metadata":{}},{"cell_type":"code","source":"BATCH_SIZE = 64\n\ntest_datagen = ImageDataGenerator(rescale=1/255)\n\ntest_loader = test_datagen.flow_from_dataframe(\n    dataframe = test,\n    directory = test_directory,\n    x_col = 'id', \n    batch_size = BATCH_SIZE, \n    shuffle = False,\n    class_mode = None,\n    target_size = (96,96)\n)","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:38:35.212766Z","iopub.execute_input":"2024-08-25T05:38:35.213398Z","iopub.status.idle":"2024-08-25T05:39:50.469958Z","shell.execute_reply.started":"2024-08-25T05:38:35.213364Z","shell.execute_reply":"2024-08-25T05:39:50.468963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Model","metadata":{}},{"cell_type":"code","source":"cnn = keras.models.load_model('/kaggle/input/week825v1/v25v1.h5')\ncnn.summary()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:39:56.644193Z","iopub.execute_input":"2024-08-25T05:39:56.645239Z","iopub.status.idle":"2024-08-25T05:39:56.924787Z","shell.execute_reply.started":"2024-08-25T05:39:56.645175Z","shell.execute_reply":"2024-08-25T05:39:56.923862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test Predictions","metadata":{}},{"cell_type":"code","source":"test_probs = cnn.predict(test_loader)\nprint(test_probs.shape)","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:40:10.449215Z","iopub.execute_input":"2024-08-25T05:40:10.449583Z","iopub.status.idle":"2024-08-25T05:41:33.590670Z","shell.execute_reply.started":"2024-08-25T05:40:10.449547Z","shell.execute_reply":"2024-08-25T05:41:33.589680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_probs[:10,].round(2))","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:42:03.388611Z","iopub.execute_input":"2024-08-25T05:42:03.389259Z","iopub.status.idle":"2024-08-25T05:42:03.394618Z","shell.execute_reply.started":"2024-08-25T05:42:03.389228Z","shell.execute_reply":"2024-08-25T05:42:03.393538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred = np.argmax(test_probs, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:42:07.959109Z","iopub.execute_input":"2024-08-25T05:42:07.959522Z","iopub.status.idle":"2024-08-25T05:42:07.964737Z","shell.execute_reply.started":"2024-08-25T05:42:07.959492Z","shell.execute_reply":"2024-08-25T05:42:07.963669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare Submission","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/sample_submission.csv')\nsubmission.label = test_probs\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:42:16.751881Z","iopub.execute_input":"2024-08-25T05:42:16.752264Z","iopub.status.idle":"2024-08-25T05:42:16.812133Z","shell.execute_reply.started":"2024-08-25T05:42:16.752233Z","shell.execute_reply":"2024-08-25T05:42:16.811169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False, header=True)","metadata":{"execution":{"iopub.status.busy":"2024-08-25T05:42:20.865342Z","iopub.execute_input":"2024-08-25T05:42:20.865713Z","iopub.status.idle":"2024-08-25T05:42:21.074245Z","shell.execute_reply.started":"2024-08-25T05:42:20.865683Z","shell.execute_reply":"2024-08-25T05:42:21.073221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}