{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport keras, os\nimport glob\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, BatchNormalization, Activation, Conv2D, MaxPooling2D, LeakyReLU, SpatialDropout2D\nfrom keras import regularizers, optimizers\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom sklearn.model_selection import train_test_split\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-24T23:59:13.090866Z","iopub.execute_input":"2025-06-24T23:59:13.091168Z","iopub.status.idle":"2025-06-24T23:59:29.602045Z","shell.execute_reply.started":"2025-06-24T23:59:13.091138Z","shell.execute_reply":"2025-06-24T23:59:29.601231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n#Reading in the training data, we'll merge this with the given labels\npdir = '/kaggle/input/histopathologic-cancer-detection/'\ntrain_dir = pdir + 'train/'\n\ndf_train = pd.DataFrame({'path': glob.glob(os.path.join(train_dir, '*.tif'))}) #Take all .tif files in training folder\ndf_train['id'] = df_train['path'].str.extract(r'([^//]+).tif$') #Use filenames as ID\nlabels = pd.read_csv(pdir + 'train_labels.csv')\ndf_train = df_train.merge(labels, on='id') #merge with labels\nprint(df_train.head(5))\nprint(df_train.info())\nprint(df_train['path'][0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T23:59:33.547764Z","iopub.execute_input":"2025-06-24T23:59:33.548042Z","iopub.status.idle":"2025-06-24T23:59:39.684105Z","shell.execute_reply.started":"2025-06-24T23:59:33.548020Z","shell.execute_reply":"2025-06-24T23:59:39.683431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Creating dataframe with the test data in the same way\ntest_dir = pdir + 'test/'\ndf_test = pd.DataFrame({'path': glob.glob(os.path.join(test_dir, '*.tif'))})\ndf_test['id'] = df_test['path'].str.extract(r'([^//]+).tif$')\nprint(df_test.head(5))\nprint(df_test.info())\nprint(df_test['path'][0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T23:59:46.273637Z","iopub.execute_input":"2025-06-24T23:59:46.273912Z","iopub.status.idle":"2025-06-24T23:59:47.913352Z","shell.execute_reply.started":"2025-06-24T23:59:46.273891Z","shell.execute_reply":"2025-06-24T23:59:47.912311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Checking the distribution of classes\ncounts = df_train['label'].value_counts()\nprint(counts)\nsns.barplot(x=counts.index, y=counts.values)\nplt.xlabel('Label')\nplt.ylabel('Number of Observations')\nplt.title('Distribution of Labels in the Training Data')\nplt.xticks(ticks=[0, 1], labels=['0: No Cancer', '1: Cancer'])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T23:59:50.809859Z","iopub.execute_input":"2025-06-24T23:59:50.810416Z","iopub.status.idle":"2025-06-24T23:59:51.009690Z","shell.execute_reply.started":"2025-06-24T23:59:50.810385Z","shell.execute_reply":"2025-06-24T23:59:51.009080Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Balancing the data by undersampling class 0\nc0 = df_train[df_train['label'] == 0]\nc1 = df_train[df_train['label'] == 1]\n\nc0_bal = c0.sample(n=len(c1), random_state=1)\n\ndf_trainbal = pd.concat([c0_bal, c1], axis=0)\n\ndf_trainbal = df_trainbal.sample(frac=1, random_state=1).reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T23:59:53.450120Z","iopub.execute_input":"2025-06-24T23:59:53.450401Z","iopub.status.idle":"2025-06-24T23:59:53.539462Z","shell.execute_reply.started":"2025-06-24T23:59:53.450370Z","shell.execute_reply":"2025-06-24T23:59:53.538890Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Checking the distribution of classes\ncounts = df_trainbal['label'].value_counts()\nprint(counts)\nsns.barplot(x=counts.index, y=counts.values)\nplt.xlabel('Label')\nplt.ylabel('Number of Observations')\nplt.title('Distribution of Labels in the Balanced Training Data')\nplt.xticks(ticks=[0, 1], labels=['0: No Cancer', '1: Cancer'])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T23:59:56.788886Z","iopub.execute_input":"2025-06-24T23:59:56.789524Z","iopub.status.idle":"2025-06-24T23:59:56.906013Z","shell.execute_reply.started":"2025-06-24T23:59:56.789496Z","shell.execute_reply":"2025-06-24T23:59:56.905424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random_samples = np.random.randint(1, len(df_trainbal) + 1, size=15)\nfor i in random_samples:\n\n    image = mpimg.imread(df_trainbal['path'][i])\n    imageplot = plt.imshow(image)\n    plt.title('Label: ' + df_trainbal['label'][i].astype(str))\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T23:59:59.405543Z","iopub.execute_input":"2025-06-24T23:59:59.406109Z","iopub.status.idle":"2025-06-25T00:00:01.367358Z","shell.execute_reply.started":"2025-06-24T23:59:59.406085Z","shell.execute_reply":"2025-06-25T00:00:01.366738Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Let's add a column with the full filenames for use with imagedatagenerator and labels must be strings\ndf_trainbal['filename'] = df_trainbal['id'] + '.tif'\ndf_test['filename'] = df_test['id'] + '.tif'\ndf_trainbal['label'] = df_trainbal['label'].astype(str)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T00:00:11.219006Z","iopub.execute_input":"2025-06-25T00:00:11.219299Z","iopub.status.idle":"2025-06-25T00:00:11.316000Z","shell.execute_reply.started":"2025-06-25T00:00:11.219275Z","shell.execute_reply":"2025-06-25T00:00:11.315278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Create training and validation sets with image datagenerator\ntrain_df, val_df = train_test_split(df_trainbal, test_size=0.2, stratify=df_trainbal['label'], random_state=1)\ntrain_datagen = ImageDataGenerator(rescale=1/255, rotation_range=20, width_shift_range=0.2, height_shift_range=0.2, horizontal_flip=True,\n                                      vertical_flip=True, zoom_range=0.2, shear_range=0.2, fill_mode='nearest')\nval_datagen = ImageDataGenerator(rescale=1/255)\ntrain_generator = train_datagen.flow_from_dataframe(dataframe=train_df,\n                                                    directory='/kaggle/input/histopathologic-cancer-detection/train',\n                                                    x_col='filename',\n                                                    y_col='label',\n                                                    target_size=(96, 96),\n                                                    batch_size=32,\n                                                    class_mode='binary')\nval_generator = val_datagen.flow_from_dataframe(dataframe=val_df,\n                                                    directory='/kaggle/input/histopathologic-cancer-detection/train',\n                                                    x_col='filename',\n                                                    y_col='label',\n                                                    target_size=(96, 96),\n                                                    batch_size=32,\n                                                    class_mode='binary')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T00:00:23.288747Z","iopub.execute_input":"2025-06-25T00:00:23.289274Z","iopub.status.idle":"2025-06-25T00:04:59.151790Z","shell.execute_reply.started":"2025-06-25T00:00:23.289250Z","shell.execute_reply":"2025-06-25T00:04:59.151042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Create the test image datagenerator\ntest_datagen = ImageDataGenerator(rescale=1/255)\ntest_generator = test_datagen.flow_from_dataframe(dataframe=df_test,\n                                                    directory='/kaggle/input/histopathologic-cancer-detection/test',\n                                                    x_col='filename',\n                                                    y_col=None,\n                                                    target_size=(96, 96),\n                                                    batch_size=32,\n                                                    class_mode=None,\n                                                    shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T00:05:54.931049Z","iopub.execute_input":"2025-06-25T00:05:54.931801Z","iopub.status.idle":"2025-06-25T00:09:03.963291Z","shell.execute_reply.started":"2025-06-25T00:05:54.931773Z","shell.execute_reply":"2025-06-25T00:09:03.962774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# #Let's start with a simple baseline model using 2 convolutional blocks and relu activation\nmodel = Sequential()\n    \n# Conv block 1\nmodel.add(Conv2D(32, (3, 3), padding='same', input_shape=(96, 96, 3)))\nmodel.add(BatchNormalization())\nmodel.add(LeakyReLU(alpha=0.1))\nmodel.add(Conv2D(32, (3, 3), padding='same'))\nmodel.add(BatchNormalization())\nmodel.add(LeakyReLU(alpha=0.1))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(SpatialDropout2D(0.1))\n    \n# Conv block 2\nmodel.add(Conv2D(64, (3, 3), padding='same'))\nmodel.add(BatchNormalization())\nmodel.add(LeakyReLU(alpha=0.1))\nmodel.add(Conv2D(64, (3, 3), padding='same'))\nmodel.add(BatchNormalization())\nmodel.add(LeakyReLU(alpha=0.1))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(SpatialDropout2D(0.15))\n    \n# Conv block 3 (new)\nmodel.add(Conv2D(128, (3, 3), padding='same'))\nmodel.add(BatchNormalization())\nmodel.add(LeakyReLU(alpha=0.1))\nmodel.add(Conv2D(128, (3, 3), padding='same'))\nmodel.add(BatchNormalization())\nmodel.add(LeakyReLU(alpha=0.1))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(SpatialDropout2D(0.2))\n    \n# Dense layers\nmodel.add(Flatten())\nmodel.add(Dense(256, kernel_regularizer=regularizers.l2(0.01)))\nmodel.add(BatchNormalization())\nmodel.add(LeakyReLU(alpha=0.1))\nmodel.add(Dropout(0.3))\n    \nmodel.add(Dense(128, kernel_regularizer=regularizers.l2(0.01)))\nmodel.add(BatchNormalization())\nmodel.add(LeakyReLU(alpha=0.1))\nmodel.add(Dropout(0.25))\n    \nmodel.add(Dense(1, activation='sigmoid'))\n\n\ninitial_lr = 0.001\noptimizer = tf.keras.optimizers.Adam(learning_rate=initial_lr)\n   \n\n\n# Compile with updated optimizer\nmodel.compile(\n    optimizer=optimizer,\n    loss='binary_crossentropy',\n    metrics=['accuracy', tf.keras.metrics.AUC()]\n)\n\nprint(model.summary())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T00:10:06.204372Z","iopub.execute_input":"2025-06-25T00:10:06.204679Z","iopub.status.idle":"2025-06-25T00:10:09.169720Z","shell.execute_reply.started":"2025-06-25T00:10:06.204655Z","shell.execute_reply":"2025-06-25T00:10:09.169158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Enhanced callbacks\ncallbacks = [\n    tf.keras.callbacks.EarlyStopping(\n        monitor='val_loss',\n        patience=5,\n        restore_best_weights=True\n    ),\n    tf.keras.callbacks.ReduceLROnPlateau(\n        monitor='val_loss',\n        factor=0.5,\n        patience=3,\n        min_lr=1e-6\n    )\n]\n\n# Train with more epochs and callbacks\nhistory = model.fit(\n    train_generator,\n    validation_data=val_generator,\n    epochs=50,\n    callbacks=callbacks,\n    verbose=1\n)\n\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='Train Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.title('Baseline Model Accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Baseline Model Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T00:11:02.849203Z","iopub.execute_input":"2025-06-25T00:11:02.849826Z","iopub.status.idle":"2025-06-25T05:23:08.988462Z","shell.execute_reply.started":"2025-06-25T00:11:02.849802Z","shell.execute_reply":"2025-06-25T05:23:08.987749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.savefig('3_convblocks_LeakyReLU_LRScheduler.png')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T01:02:29.094049Z","iopub.execute_input":"2025-05-25T01:02:29.094746Z","iopub.status.idle":"2025-05-25T01:02:29.111066Z","shell.execute_reply.started":"2025-05-25T01:02:29.094724Z","shell.execute_reply":"2025-05-25T01:02:29.110400Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = model.predict(test_generator, verbose=1)\nsubmission = df_test.copy()\nsubmission = submission.drop(columns=['filename', 'path'])\nsubmission['label'] = predictions\nprint(submission.head())\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T07:29:32.203939Z","iopub.execute_input":"2025-06-25T07:29:32.204485Z","iopub.status.idle":"2025-06-25T07:29:32.270418Z","shell.execute_reply.started":"2025-06-25T07:29:32.204457Z","shell.execute_reply":"2025-06-25T07:29:32.269434Z"}},"outputs":[],"execution_count":null}]}