{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"},{"sourceId":324256,"sourceType":"modelInstanceVersion","modelInstanceId":273085,"modelId":294044}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nimport cv2\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import EfficientNetB0","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-14T02:58:06.656416Z","iopub.execute_input":"2025-04-14T02:58:06.656636Z","iopub.status.idle":"2025-04-14T02:58:06.672109Z","shell.execute_reply.started":"2025-04-14T02:58:06.656611Z","shell.execute_reply":"2025-04-14T02:58:06.671582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv')\ntest_labels = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/sample_submission.csv')\nprint(train_labels.head)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_labels.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T00:50:07.618559Z","iopub.execute_input":"2025-04-14T00:50:07.618771Z","iopub.status.idle":"2025-04-14T00:50:07.622998Z","shell.execute_reply.started":"2025-04-14T00:50:07.618754Z","shell.execute_reply":"2025-04-14T00:50:07.622411Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Brief description of the problem and data \n\n\nWe have set of images which we want to train on to do binary classification with multi layer of nodes(filters) to detarmine the cancer detection for the test images. The data has size of 220025 rows with 2 columns of image id & cancer detection label. The data has more of non-cancer images than cancer images close to 3:2 ratio.","metadata":{}},{"cell_type":"markdown","source":"#  Exploratory Data Analysis (EDA) — Inspect, Visualize and Clean the Data  \n\nWe will be removing all the rows where the labels aren't either 0 or 1. We would be taking 25k rows of each label to make the training balanced without either over sampling or under sampling.\n\nWe will start with using basic model like CNN with few layers & start changing the activation function & increasing to see the prediction. We will also use model with few layers & gradually with more layers validate the prediction. Later, we will be using pretrained model like EfficientNet & finetuning it to our dataset.","metadata":{}},{"cell_type":"code","source":"plt.hist(train_labels['label'], bins=2, edgecolor='black')  # bins=2 for binary data\nplt.title('Histogram of Train Labels')\nplt.xlabel('Label Value')\nplt.ylabel('Count')\nplt.xticks([0, 1], ['Non Cancer', 'Cancer'])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T00:50:07.623864Z","iopub.execute_input":"2025-04-14T00:50:07.624125Z","iopub.status.idle":"2025-04-14T00:50:07.925210Z","shell.execute_reply.started":"2025-04-14T00:50:07.624104Z","shell.execute_reply":"2025-04-14T00:50:07.924493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# removing the rows who label value is non 0 or 1\ntrain_labels = train_labels[train_labels['label'].isin([0, 1])]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T00:50:07.925995Z","iopub.execute_input":"2025-04-14T00:50:07.926273Z","iopub.status.idle":"2025-04-14T00:50:07.941708Z","shell.execute_reply.started":"2025-04-14T00:50:07.926233Z","shell.execute_reply":"2025-04-14T00:50:07.940984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"non_cancer_data = train_labels[train_labels['label'] == 0]\ncancer_data = train_labels[train_labels['label'] == 1]\n\nnon_cancer_samples = non_cancer_data.sample(n=25000, random_state=42, replace=False)\ncancer_samples = cancer_data.sample(n=25000, random_state=42, replace=False)\n\ntrain_data = pd.concat([non_cancer_samples, cancer_samples])\n\ntrain_data = train_data.sample(frac=1, random_state=42).reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T00:50:07.942505Z","iopub.execute_input":"2025-04-14T00:50:07.942779Z","iopub.status.idle":"2025-04-14T00:50:07.977154Z","shell.execute_reply.started":"2025-04-14T00:50:07.942753Z","shell.execute_reply":"2025-04-14T00:50:07.976307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dir = '/kaggle/input/histopathologic-cancer-detection/train/'\ntest_dir = '/kaggle/input/histopathologic-cancer-detection/test/'\ndef load_image_from_id(image_id, image_dir=train_dir):\n    image_path = image_dir + image_id + '.tif'\n    image = cv2.imread(image_path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    return image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T00:50:07.978043Z","iopub.execute_input":"2025-04-14T00:50:07.978366Z","iopub.status.idle":"2025-04-14T00:50:07.982851Z","shell.execute_reply.started":"2025-04-14T00:50:07.978338Z","shell.execute_reply":"2025-04-14T00:50:07.982086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"images = np.array([load_image_from_id(i) for i in train_data['id']])\ny = train_data['label'].values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T00:50:07.983715Z","iopub.execute_input":"2025-04-14T00:50:07.984008Z","iopub.status.idle":"2025-04-14T00:55:37.829925Z","shell.execute_reply.started":"2025-04-14T00:50:07.983941Z","shell.execute_reply":"2025-04-14T00:55:37.829327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"images.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T00:55:37.832181Z","iopub.execute_input":"2025-04-14T00:55:37.832785Z","iopub.status.idle":"2025-04-14T00:55:37.837361Z","shell.execute_reply.started":"2025-04-14T00:55:37.832764Z","shell.execute_reply":"2025-04-14T00:55:37.836669Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DModel Architecture\n\nWe will start with a basic CNN model with 2 layers of filters of 20 & 60 respectively using relu activation function & maxpooling after each layer & finally densing it out to a single array with relu & finally densing to a single value & applying sigmoid to determine cancer prediction.","metadata":{}},{"cell_type":"markdown","source":"# cnn model1","metadata":{}},{"cell_type":"code","source":"\ncnn_model1 = keras.Sequential([\n    layers.Input(shape=(96,96,3)),\n    layers.Rescaling(1./255), # normalizing the values from 0 to 1\n    layers.Cropping2D(cropping=32),\n    layers.Conv2D(20, (3, 3), activation='relu', padding='same'),\n    layers.MaxPooling2D((2, 2)),\n    layers.Conv2D(60, (3, 3), activation='relu', padding='same'),\n    layers.MaxPooling2D((2, 2)),\n    layers.Flatten(),\n    \n    layers.Dense(64, activation='relu'),\n    layers.Dropout(0.5),  # Prevent overfitting\n    \n    # Output layer\n    layers.Dense(1, activation='sigmoid')    \n])\n\ncnn_model1.compile(\n    optimizer=Adam(learning_rate=0.00005),\n    loss='binary_crossentropy',\n    metrics=['accuracy', 'auc']\n)\ncnn_model1.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T00:55:37.838102Z","iopub.execute_input":"2025-04-14T00:55:37.838301Z","iopub.status.idle":"2025-04-14T00:55:40.605633Z","shell.execute_reply.started":"2025-04-14T00:55:37.838283Z","shell.execute_reply":"2025-04-14T00:55:40.604936Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"we would be doing 1:4 split of test vs train data & would run for 50 epochs & have a stop function of 5(when the loss function hasn't improved for the 5 runs).","metadata":{}},{"cell_type":"code","source":"es = EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\ncnn_model1 = cnn_model1.fit(images, y, validation_split=0.2, epochs=50, callbacks=[es])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T00:55:40.606468Z","iopub.execute_input":"2025-04-14T00:55:40.606730Z","iopub.status.idle":"2025-04-14T00:58:15.905472Z","shell.execute_reply.started":"2025-04-14T00:55:40.606707Z","shell.execute_reply":"2025-04-14T00:58:15.904839Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now we are adding on one more layer in between with 40 filters & making the rest of the architecture of the model to be same as above model.","metadata":{}},{"cell_type":"markdown","source":"# cnn model2","metadata":{}},{"cell_type":"code","source":"# Adding one more layer in between with 40 nodes\ncnn_model2 = keras.Sequential([\n    layers.Input(shape=(96,96,3)),\n    layers.Rescaling(1./255), # normalizing the values from 0 to 1\n    layers.Cropping2D(cropping=32),\n    layers.Conv2D(20, (3, 3), activation='relu', padding='same'),\n    layers.MaxPooling2D((2, 2)),\n    layers.Conv2D(40, (3, 3), activation='relu', padding='same'),\n    layers.MaxPooling2D((2, 2)),\n    layers.Conv2D(60, (3, 3), activation='relu', padding='same'),\n    layers.MaxPooling2D((2, 2)),\n    layers.Flatten(),\n    \n    layers.Dense(64, activation='relu'),\n    layers.Dropout(0.5),  # Prevent overfitting\n    \n    # Output layer\n    layers.Dense(1, activation='sigmoid')    \n])\n\ncnn_model2.compile(\n    optimizer=Adam(learning_rate=0.00005),\n    loss='binary_crossentropy',\n    metrics=['accuracy', 'auc']\n)\ncnn_model2.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T00:58:15.906371Z","iopub.execute_input":"2025-04-14T00:58:15.906574Z","iopub.status.idle":"2025-04-14T00:58:15.973957Z","shell.execute_reply.started":"2025-04-14T00:58:15.906559Z","shell.execute_reply":"2025-04-14T00:58:15.973397Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We will still be having the same above hyper parameters, to test the impact from additional layers","metadata":{}},{"cell_type":"code","source":"es = EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\ncnn_model2 = cnn_model2.fit(images, y, validation_split=0.2, epochs=50, callbacks=[es])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T00:58:15.974615Z","iopub.execute_input":"2025-04-14T00:58:15.974803Z","iopub.status.idle":"2025-04-14T01:01:17.403960Z","shell.execute_reply.started":"2025-04-14T00:58:15.974788Z","shell.execute_reply":"2025-04-14T01:01:17.403311Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now we would like to use a pretrained advanced models like Efficient net on image training. We will use the parameters of this model & try to fit for our images of 64x64 pixels. This is 4million+ parameter model, I assume this might overfit our smaller dataset but we would still try & verify.","metadata":{}},{"cell_type":"markdown","source":"# EfficientNet model1","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetB0\n\n# Local path to the uploaded weights\nweights_path = '/kaggle/input/efficient_net/keras/default/1/efficientnetb0_notop.h5'\nbase_model = EfficientNetB0(weights=weights_path, include_top=False, input_shape=(96, 96, 3))\nbase_model.trainable = False  # freezes during initial training.\n\neffnet_model = keras.Sequential([\n    base_model,\n    layers.GlobalAveragePooling2D(),\n    layers.Dense(64, activation='relu'),\n    layers.Dropout(0.5),\n    layers.Dense(1, activation='sigmoid')\n])\neffnet_model.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T01:01:17.404880Z","iopub.execute_input":"2025-04-14T01:01:17.405107Z","iopub.status.idle":"2025-04-14T01:01:19.017669Z","shell.execute_reply.started":"2025-04-14T01:01:17.405083Z","shell.execute_reply":"2025-04-14T01:01:19.016958Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Using Adam optimizer which is suitable for efficient net models, we are using a learning rate of 0.0001 slightly higher than the above cnn model but with a lower patience & epochs.","metadata":{}},{"cell_type":"code","source":"effnet_model.compile(\n    optimizer=Adam(learning_rate=0.0001),\n    loss='binary_crossentropy',\n    metrics=['accuracy', 'auc']\n)\nes = EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\nmodel1 = effnet_model.fit(images, y, validation_split=0.20, epochs=15, callbacks=[es])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T01:01:19.018485Z","iopub.execute_input":"2025-04-14T01:01:19.018756Z","iopub.status.idle":"2025-04-14T01:05:33.768992Z","shell.execute_reply.started":"2025-04-14T01:01:19.018733Z","shell.execute_reply":"2025-04-14T01:05:33.768303Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Here we are going to build a same efficient net model but with tuning hyper parameters with higher learning rate of 0.001, different split of data between training & testing. Additionally, we are decreasing epochs as well.","metadata":{}},{"cell_type":"markdown","source":"# EfficientNet model2 (hyper parameters tuning from the above model)","metadata":{}},{"cell_type":"code","source":"effnet_model.compile(\n    optimizer=Adam(learning_rate=0.001),\n    loss='binary_crossentropy',\n    metrics=['accuracy', 'auc']\n)\nes = EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\nmodel2 = effnet_model.fit(images, y, validation_split=0.20, epochs=15, callbacks=[es])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T01:05:33.769982Z","iopub.execute_input":"2025-04-14T01:05:33.770195Z","iopub.status.idle":"2025-04-14T01:09:46.452991Z","shell.execute_reply.started":"2025-04-14T01:05:33.770179Z","shell.execute_reply":"2025-04-14T01:09:46.452428Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Result & analysis\nThe initial CNN model with 2 layers has gave us val_accuracy: 0.8015 - val_auc: 0.8821 - val_loss: 0.4299\nThe 2nd model with an additional layer has gave us val_accuracy: 0.7983 - val_auc: 0.8761 - val_loss: 0.4403 (with same hyper parameters as above)\n\nThen we have used a better model efficient net but we think it might overfit with more parameters & small dataset.\nThis model has val_accuracy: 0.8836 - val_auc: 0.9506 - val_loss: 0.2931. this is far better than the above both models.\n\nThen we have increased the learning rate to see how it performs & have a different of data.\nThis one gave val_accuracy: 0.8788 - val_auc: 0.9465 - val_loss: 0.3049\n\n\n\n\n\n","metadata":{}},{"cell_type":"markdown","source":"# Evaluation\n","metadata":{}},{"cell_type":"code","source":"def plot_validation_accuracy_condensed(model, model_name=None):\n    \n    val_accuracy = model.history['val_accuracy']\n\n    epochs = range(1, len(val_accuracy) + 1)\n\n    plt.figure(figsize=(7, 4)) # Slightly smaller figure\n    plt.plot(epochs, val_accuracy, marker='.', linestyle='-', label='Val Accuracy') # Smaller marker\n    plt.title(f'{model_name} - Val Accuracy' if model_name else 'Validation Accuracy')\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    \n    if len(epochs) <= 15: plt.xticks(epochs)\n    else: plt.locator_params(axis='x', integer=True)\n    plt.legend()\n    plt.grid(axis='y', linestyle='--', alpha=0.7) # Grid on y-axis only, lighter\n    plt.tight_layout() # Adjust layout\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T01:09:46.454011Z","iopub.execute_input":"2025-04-14T01:09:46.454280Z","iopub.status.idle":"2025-04-14T01:09:46.459754Z","shell.execute_reply.started":"2025-04-14T01:09:46.454238Z","shell.execute_reply":"2025-04-14T01:09:46.459127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_validation_accuracy_condensed(cnn_model1, \"efficient net cnn_model1\")\nplot_validation_accuracy_condensed(cnn_model2, \"efficient net cnn_model2\")\nplot_validation_accuracy_condensed(model1, \"efficient net cnn_model1\")\nplot_validation_accuracy_condensed(model2, \"efficient net cnn_model2\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T01:09:46.460469Z","iopub.execute_input":"2025-04-14T01:09:46.460652Z","iopub.status.idle":"2025-04-14T01:09:47.241602Z","shell.execute_reply.started":"2025-04-14T01:09:46.460638Z","shell.execute_reply":"2025-04-14T01:09:47.240870Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Conclusion\n\nThe 2nd CNN model has an additional layer of filters but it has less parameters & it's less accurate than the 1st model. The efficient net model is advanced & it's pretrained & we are finetuning it's existing parameters. So this has the better accuracy & performance. Later, we have tuned the hyper parameters by increasing the learning rate, it has decreased the accuracy.\nThe key learning is to use a pretrained model with it's existing weights & finetune it to our dataset to have the best model & accuracy. We need to have a balanced learning rate to have better accuracy. ","metadata":{}},{"cell_type":"code","source":"test_images = np.array([load_image_from_id(i, image_dir=test_dir) for i in test_labels['id']])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T01:10:11.120548Z","iopub.execute_input":"2025-04-14T01:10:11.121130Z","iopub.status.idle":"2025-04-14T01:20:01.501796Z","shell.execute_reply.started":"2025-04-14T01:10:11.121108Z","shell.execute_reply":"2025-04-14T01:20:01.501146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_output = effnet_model.predict(test_images)\ny_pred_output = y_pred_output.ravel()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T01:21:05.472537Z","iopub.execute_input":"2025-04-14T01:21:05.472817Z","iopub.status.idle":"2025-04-14T01:21:38.782116Z","shell.execute_reply.started":"2025-04-14T01:21:05.472792Z","shell.execute_reply":"2025-04-14T01:21:38.781311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!rm -f /kaggle/working/submission_cnn.npy","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_cnn_df = pd.DataFrame({\n            'id':test_labels[\"id\"],\n            'label':y_pred_output })\nsubmission_cnn_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T02:49:24.105337Z","iopub.execute_input":"2025-04-14T02:49:24.105654Z","iopub.status.idle":"2025-04-14T02:49:24.253851Z","shell.execute_reply.started":"2025-04-14T02:49:24.105626Z","shell.execute_reply":"2025-04-14T02:49:24.253294Z"}},"outputs":[],"execution_count":null}]}