{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport keras, os\nimport glob\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, BatchNormalization, Activation, Conv2D, MaxPooling2D, LeakyReLU, SpatialDropout2D\nfrom keras import regularizers, optimizers\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom sklearn.model_selection import train_test_split\nimport pickle\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-26T23:57:44.458629Z","iopub.execute_input":"2025-06-26T23:57:44.458871Z","iopub.status.idle":"2025-06-26T23:58:02.159724Z","shell.execute_reply.started":"2025-06-26T23:57:44.458854Z","shell.execute_reply":"2025-06-26T23:58:02.158847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n#Reading in the training data, we'll merge this with the given labels\npdir = '/kaggle/input/histopathologic-cancer-detection/'\ntrain_dir = pdir + 'train/'\n\ndf_train = pd.DataFrame({'path': glob.glob(os.path.join(train_dir, '*.tif'))}) #Take all .tif files in training folder\ndf_train['id'] = df_train['path'].str.extract(r'([^//]+).tif$') #Use filenames as ID\nlabels = pd.read_csv(pdir + 'train_labels.csv')\ndf_train = df_train.merge(labels, on='id') #merge with labels\nprint(df_train.head(5))\nprint(df_train.info())\nprint(df_train['path'][0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T23:58:02.161119Z","iopub.execute_input":"2025-06-26T23:58:02.161876Z","iopub.status.idle":"2025-06-26T23:58:08.530097Z","shell.execute_reply.started":"2025-06-26T23:58:02.161845Z","shell.execute_reply":"2025-06-26T23:58:08.529416Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Creating dataframe with the test data in the same way\ntest_dir = pdir + 'test/'\ndf_test = pd.DataFrame({'path': glob.glob(os.path.join(test_dir, '*.tif'))})\ndf_test['id'] = df_test['path'].str.extract(r'([^//]+).tif$')\nprint(df_test.head(5))\nprint(df_test.info())\nprint(df_test['path'][0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T23:58:26.041758Z","iopub.execute_input":"2025-06-26T23:58:26.042563Z","iopub.status.idle":"2025-06-26T23:58:26.718330Z","shell.execute_reply.started":"2025-06-26T23:58:26.042519Z","shell.execute_reply":"2025-06-26T23:58:26.717495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Checking the distribution of classes\ncounts = df_train['label'].value_counts()\nprint(counts)\nsns.barplot(x=counts.index, y=counts.values)\nplt.xlabel('Label')\nplt.ylabel('Number of Observations')\nplt.title('Distribution of Labels in the Training Data')\nplt.xticks(ticks=[0, 1], labels=['0: No Cancer', '1: Cancer'])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T23:58:28.893079Z","iopub.execute_input":"2025-06-26T23:58:28.893707Z","iopub.status.idle":"2025-06-26T23:58:29.125200Z","shell.execute_reply.started":"2025-06-26T23:58:28.893679Z","shell.execute_reply":"2025-06-26T23:58:29.124492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Balancing the data by undersampling class 0\nc0 = df_train[df_train['label'] == 0]\nc1 = df_train[df_train['label'] == 1]\n\nc0_bal = c0.sample(n=len(c1), random_state=1)\n\ndf_trainbal = pd.concat([c0_bal, c1], axis=0)\n\ndf_trainbal = df_trainbal.sample(frac=1, random_state=1).reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T23:58:31.291327Z","iopub.execute_input":"2025-06-26T23:58:31.292172Z","iopub.status.idle":"2025-06-26T23:58:31.397529Z","shell.execute_reply.started":"2025-06-26T23:58:31.292142Z","shell.execute_reply":"2025-06-26T23:58:31.396688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Checking the distribution of classes\ncounts = df_trainbal['label'].value_counts()\nprint(counts)\nsns.barplot(x=counts.index, y=counts.values)\nplt.xlabel('Label')\nplt.ylabel('Number of Observations')\nplt.title('Distribution of Labels in the Balanced Training Data')\nplt.xticks(ticks=[0, 1], labels=['0: No Cancer', '1: Cancer'])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T23:58:34.586287Z","iopub.execute_input":"2025-06-26T23:58:34.586876Z","iopub.status.idle":"2025-06-26T23:58:34.718244Z","shell.execute_reply.started":"2025-06-26T23:58:34.586850Z","shell.execute_reply":"2025-06-26T23:58:34.717581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random_samples = np.random.randint(1, len(df_trainbal) + 1, size=15)\nfor i in random_samples:\n\n    image = mpimg.imread(df_trainbal['path'][i])\n    imageplot = plt.imshow(image)\n    plt.title('Label: ' + df_trainbal['label'][i].astype(str))\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T23:58:38.646136Z","iopub.execute_input":"2025-06-26T23:58:38.646862Z","iopub.status.idle":"2025-06-26T23:58:40.968380Z","shell.execute_reply.started":"2025-06-26T23:58:38.646840Z","shell.execute_reply":"2025-06-26T23:58:40.967524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Let's add a column with the full filenames for use with imagedatagenerator and labels must be strings\ndf_trainbal['filename'] = df_trainbal['id'] + '.tif'\ndf_test['filename'] = df_test['id'] + '.tif'\ndf_trainbal['label'] = df_trainbal['label'].astype(str)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T23:58:56.593903Z","iopub.execute_input":"2025-06-26T23:58:56.594604Z","iopub.status.idle":"2025-06-26T23:58:56.693611Z","shell.execute_reply.started":"2025-06-26T23:58:56.594579Z","shell.execute_reply":"2025-06-26T23:58:56.692698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Create training and validation sets with image datagenerator\ntrain_df, val_df = train_test_split(df_trainbal, test_size=0.2, stratify=df_trainbal['label'], random_state=1)\ntrain_datagen = ImageDataGenerator(rescale=1/255, rotation_range=20, width_shift_range=0.2, height_shift_range=0.2, horizontal_flip=True,\n                                      vertical_flip=True, zoom_range=0.2, shear_range=0.2, fill_mode='nearest')\nval_datagen = ImageDataGenerator(rescale=1/255)\ntrain_generator = train_datagen.flow_from_dataframe(dataframe=train_df,\n                                                    directory='/kaggle/input/histopathologic-cancer-detection/train',\n                                                    x_col='filename',\n                                                    y_col='label',\n                                                    target_size=(96, 96),\n                                                    batch_size=32,\n                                                    class_mode='binary')\nval_generator = val_datagen.flow_from_dataframe(dataframe=val_df,\n                                                    directory='/kaggle/input/histopathologic-cancer-detection/train',\n                                                    x_col='filename',\n                                                    y_col='label',\n                                                    target_size=(96, 96),\n                                                    batch_size=32,\n                                                    class_mode='binary',\n                                                    shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-27T00:05:10.274679Z","iopub.execute_input":"2025-06-27T00:05:10.274926Z","iopub.status.idle":"2025-06-27T00:09:08.205323Z","shell.execute_reply.started":"2025-06-27T00:05:10.274911Z","shell.execute_reply":"2025-06-27T00:09:08.204582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Create the test image datagenerator\ntest_datagen = ImageDataGenerator(rescale=1/255)\ntest_generator = test_datagen.flow_from_dataframe(dataframe=df_test,\n                                                    directory='/kaggle/input/histopathologic-cancer-detection/test',\n                                                    x_col='filename',\n                                                    y_col=None,\n                                                    target_size=(96, 96),\n                                                    batch_size=32,\n                                                    class_mode=None,\n                                                    shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-27T00:10:58.159064Z","iopub.execute_input":"2025-06-27T00:10:58.159924Z","iopub.status.idle":"2025-06-27T00:13:25.019026Z","shell.execute_reply.started":"2025-06-27T00:10:58.159898Z","shell.execute_reply":"2025-06-27T00:13:25.018257Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# #Let's start with a simple baseline model using 2 convolutional blocks and relu activation\nmodel = Sequential()\n    \n# Conv block 1\nmodel.add(Conv2D(32, (3, 3), padding='same', input_shape=(96, 96, 3)))\nmodel.add(BatchNormalization())\nmodel.add(LeakyReLU(alpha=0.1))\nmodel.add(Conv2D(32, (3, 3), padding='same'))\nmodel.add(BatchNormalization())\nmodel.add(LeakyReLU(alpha=0.1))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(SpatialDropout2D(0.1))\n    \n# Conv block 2\nmodel.add(Conv2D(64, (3, 3), padding='same'))\nmodel.add(BatchNormalization())\nmodel.add(LeakyReLU(alpha=0.1))\nmodel.add(Conv2D(64, (3, 3), padding='same'))\nmodel.add(BatchNormalization())\nmodel.add(LeakyReLU(alpha=0.1))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(SpatialDropout2D(0.15))\n    \n# Conv block 3 (new)\nmodel.add(Conv2D(128, (3, 3), padding='same'))\nmodel.add(BatchNormalization())\nmodel.add(LeakyReLU(alpha=0.1))\nmodel.add(Conv2D(128, (3, 3), padding='same'))\nmodel.add(BatchNormalization())\nmodel.add(LeakyReLU(alpha=0.1))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(SpatialDropout2D(0.2))\n    \n# Dense layers\nmodel.add(Flatten())\nmodel.add(Dense(256, kernel_regularizer=regularizers.l2(0.01)))\nmodel.add(BatchNormalization())\nmodel.add(LeakyReLU(alpha=0.1))\nmodel.add(Dropout(0.3))\n    \nmodel.add(Dense(128, kernel_regularizer=regularizers.l2(0.01)))\nmodel.add(BatchNormalization())\nmodel.add(LeakyReLU(alpha=0.1))\nmodel.add(Dropout(0.25))\n    \nmodel.add(Dense(1, activation='sigmoid'))\n\n\ninitial_lr = 0.001\noptimizer = tf.keras.optimizers.Adam(learning_rate=initial_lr)\n   \ndef scheduler(epoch, lr):\n    return lr * 0.9  # for example, decay by 10% every epoch\n   \nlr_callback = tf.keras.callbacks.LearningRateScheduler(scheduler)\n\n\n# Compile with updated optimizer\nmodel.compile(\n    optimizer=optimizer,\n    loss='binary_crossentropy',\n    metrics=['accuracy', tf.keras.metrics.AUC()]\n)","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetB0, MobileNetV2\nfrom tensorflow.keras import layers, models\n\nbase_model = MobileNetV2(\n    input_shape=(96, 96, 3),\n    include_top=False,\n    weights='imagenet'\n)\nbase_model.trainable = False\n\nmodel = models.Sequential([\n    base_model,\n    layers.GlobalAveragePooling2D(),\n    layers.Dropout(0.3),\n    layers.Dense(128, activation='relu'),\n    layers.Dropout(0.2),\n    layers.Dense(1, activation='sigmoid')  # Binary classification\n])\n\ninitial_lr = 0.0001\noptimizer = tf.keras.optimizers.Adam(learning_rate=initial_lr)\n\n# Compile with updated optimizer\nmodel.compile(\n    optimizer=optimizer,\n    loss='binary_crossentropy',\n    metrics=['accuracy', tf.keras.metrics.AUC()]\n)\nprint(model.summary())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-27T00:15:39.486867Z","iopub.execute_input":"2025-06-27T00:15:39.487339Z","iopub.status.idle":"2025-06-27T00:15:45.081636Z","shell.execute_reply.started":"2025-06-27T00:15:39.487317Z","shell.execute_reply":"2025-06-27T00:15:45.081012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Enhanced callbacks\ncallbacks = [\n    tf.keras.callbacks.EarlyStopping(\n        monitor='val_loss',\n        patience=5,\n        restore_best_weights=True\n    ),\n    tf.keras.callbacks.ReduceLROnPlateau(\n        monitor='val_loss',\n        factor=0.5,\n        patience=3,\n        min_lr=1e-6\n    )\n]\n\n# Train with more epochs and callbacks\nhistory = model.fit(\n    train_generator,\n    validation_data=val_generator,\n    epochs=5,\n    callbacks=callbacks,\n    verbose=1\n)\n\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='Train Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.title('Baseline Model Accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Baseline Model Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-27T00:16:09.486846Z","iopub.execute_input":"2025-06-27T00:16:09.487133Z","execution_failed":"2025-06-27T06:15:48.228Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save('/kaggle/working/my_model.keras')\nwith open('/kaggle/working/history.pkl', 'wb') as f:\n    pickle.dump(history.history, f)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-06-27T06:15:48.228Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\n# Get true labels and predicted labels\nval_generator.reset()\ny_true = val_generator.classes\ny_pred_prob = model.predict(val_generator)\ny_pred = (y_pred_prob > 0.5).astype(int).flatten()\n\n# Confusion matrix\ncm = confusion_matrix(y_true, y_pred)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=val_generator.class_indices.keys())\ndisp.plot(cmap='Blues')\nplt.title(\"Confusion Matrix\")\nplt.show()\nplt.savefig('EfficientNetB0transferlearningtrainingCM.png')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc\n\n# ROC Curve\nfpr, tpr, thresholds = roc_curve(y_true, y_pred_prob)\nroc_auc = auc(fpr, tpr)\n\nplt.figure()\nplt.plot(fpr, tpr, color='darkorange', lw=2, label=f'ROC curve (AUC = {roc_auc:.2f})')\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic (ROC)')\nplt.legend(loc='lower right')\nplt.grid(True)\nplt.show()\nplt.savefig('mobilenetAUCROC.png')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = model.predict(test_generator, verbose=1)\nsubmission = df_test.copy()\nsubmission = submission.drop(columns=['filename', 'path'])\nsubmission['label'] = predictions\nprint(submission.head())\nsubmission.to_csv('MobileNetv2.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T01:02:29.111871Z","iopub.execute_input":"2025-05-25T01:02:29.112075Z","iopub.status.idle":"2025-05-25T01:07:41.646287Z","shell.execute_reply.started":"2025-05-25T01:02:29.112061Z","shell.execute_reply":"2025-05-25T01:07:41.645662Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}