{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30886,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\"\"\"\n# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\"\"\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-02-15T00:35:55.944549Z","iopub.execute_input":"2025-02-15T00:35:55.944896Z","iopub.status.idle":"2025-02-15T00:35:55.951794Z","shell.execute_reply.started":"2025-02-15T00:35:55.944859Z","shell.execute_reply":"2025-02-15T00:35:55.950845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#imports \n\nimport os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport keras\nimport tensorflow as tf\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T00:35:55.953023Z","iopub.execute_input":"2025-02-15T00:35:55.953338Z","iopub.status.idle":"2025-02-15T00:35:55.974844Z","shell.execute_reply.started":"2025-02-15T00:35:55.953308Z","shell.execute_reply":"2025-02-15T00:35:55.974153Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Step One. Importing and reading data, problem description**\n\nFirst, we are going to look at the provided dataset.\n\nWe have in our hands a set of 220.025 histological images with typical H-E tinction labeled as cancer or not cancer. No info about types of cancer, and no demographic data about patients. \n\nIn that way of reasoning, we have a binary classification problem with a lot of data, and pretty complex data as those histological images are similar between each other, not homogeneous and probably coming from different hospitals, care centers, patients, and even organs.\n\nThe dataset images are on tiff format and size is 96px X 96px X 3chanels (RGB).\n","metadata":{}},{"cell_type":"code","source":"#Read\ndf_train = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv') \n\nprint (df_train.head())\nprint (df_train.shape)\n\n#As we are going to need URL to images we are going to create new colummn with this path\n#path to actual images in new colummn\ndf_train['location'] = '/kaggle/input/histopathologic-cancer-detection/train/' + df_train['id'] + '.tif'\n\n#Most networks will need label as string so lets convert it\ndf_train[\"label\"] = df_train[\"label\"].astype(str)  # Convert to string\n\n#We are first going to test various networks so lest get a random sample, this is for reducing the time of trining\ndf_sampled = df_train.sample(n=1024, random_state=42)  # Set seed for reproducibility\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T00:35:55.976365Z","iopub.execute_input":"2025-02-15T00:35:55.976648Z","iopub.status.idle":"2025-02-15T00:35:56.293313Z","shell.execute_reply.started":"2025-02-15T00:35:55.976628Z","shell.execute_reply":"2025-02-15T00:35:56.292225Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Step Two. Basic EDA.**\n\nWe are just looking at some images and its label, so not much here to explore.\n\nLest take a look at distribution of labels, check for duplicates or null values, and some examples of pictures with diferent labels.\n\n","metadata":{}},{"cell_type":"code","source":"#NN\n\nprint ('Overview of dataset')\nprint(df_train.info())  \nprint()\nduplicates = df_train.duplicated(subset=['id']).sum()\nprint(f'Duplicates\\n{duplicates}')\nprint()\nprint('Check missing values')\nprint(df_train.isnull().sum())\nprint()\nprint('Labels')\ndf_train['label'].value_counts()\n\n# Count values\ntotal = len(df_train)  # Total number of rows\nplt.figure(figsize=(6,6))\n\nax = sns.countplot(data=df_train, x='label', palette='viridis')\n\n# Add percentages on bars\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / total:.2f}%'  # Calculate %\n    ax.annotate(percentage, (p.get_x() + p.get_width() / 2, p.get_height()), \n                ha='center', va='bottom', fontsize=12, color='black')\n\nplt.title('Class Distribution with Percentages')\nplt.show()\n\n#preview of first 10 images ant its labels\nfig, axes = plt.subplots(2, 5, figsize=(15, 6))  # 2 rows, 5 columns\n\nfor i, ax in enumerate(axes.flat):\n    img = mpimg.imread(df_train['location'].iloc[i])  # Load image\n    ax.imshow(img)\n    ax.axis('off')  # Hide axes\n    ax.set_title(f\"Label: {df_train.loc[i, 'label']}\")  # Show label\n\nplt.tight_layout()\nplt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T00:35:56.295192Z","iopub.execute_input":"2025-02-15T00:35:56.295638Z","iopub.status.idle":"2025-02-15T00:35:57.892013Z","shell.execute_reply.started":"2025-02-15T00:35:56.295599Z","shell.execute_reply":"2025-02-15T00:35:57.891061Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Step Three. DModel Architecture**\n\nAs we expected images are very complex, labels are not distributed perfectly so we have more benign than malignant samples, but this is not a problem because in the medical centers there is going to be usually a distribution where benign pathology is far more common than cancer.\n\nThis complex data is going to be examined using CNN. So, we are going to take sampled data and train some networks with default parameters.\n\nAfter researching a little bit, it seems for this type of histological images DenseNet-169, ResNet-50 and EfficientNetB0 are the top three recommended.\nLest try them (with small data) and the one with better performance will be fine-tuned.\n","metadata":{}},{"cell_type":"code","source":"#imports for CNNs\n\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.applications import DenseNet169\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D\nfrom tensorflow.keras.optimizers import Adam, SGD, RMSprop\nfrom tensorflow.keras.losses import BinaryCrossentropy, Hinge\nfrom tensorflow.keras.metrics import AUC\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T00:35:57.893038Z","iopub.execute_input":"2025-02-15T00:35:57.893355Z","iopub.status.idle":"2025-02-15T00:35:57.898426Z","shell.execute_reply.started":"2025-02-15T00:35:57.893322Z","shell.execute_reply":"2025-02-15T00:35:57.897476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cheking similar distribution between sample and complete df.\n\ntotal = len(df_sampled)  # Total number of rows\nplt.figure(figsize=(6,6))\n\nax = sns.countplot(data=df_sampled, x='label', palette='viridis')\n\n# Add percentages on bars\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / total:.2f}%'  # Calculate %\n    ax.annotate(percentage, (p.get_x() + p.get_width() / 2, p.get_height()), \n                ha='center', va='bottom', fontsize=12, color='black')\n\nplt.title('Class Distribution with Percentages')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T00:35:57.899451Z","iopub.execute_input":"2025-02-15T00:35:57.899774Z","iopub.status.idle":"2025-02-15T00:35:58.089250Z","shell.execute_reply.started":"2025-02-15T00:35:57.899740Z","shell.execute_reply":"2025-02-15T00:35:58.088238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load EfficientNetB0  (Feature Extractor)\nEfficientNet_model = EfficientNetB0(weights='imagenet', include_top=False, input_shape=(224, 224, 3))  \nEfficientNet_model.trainable = True  # Allow fine-tuning\n\n# Add Custom Classifier for Binary Classification\nx = GlobalAveragePooling2D()(EfficientNet_model.output)  # Convert feature maps to vector\nx = Dense(128, activation=\"relu\")(x)  # Optional Dense Layer\nx = Dense(1, activation=\"sigmoid\")(x)  # Binary Classification (Sigmoid Output)\n\n# Create Model\nmodel_Eff = Model(inputs=EfficientNet_model.input, outputs=x)\n\n# Compile Model (Use Binary Loss & AUC Metric)\nmodel_Eff.compile(optimizer=Adam(learning_rate=1e-5), \n              loss=\"binary_crossentropy\", \n              metrics=[\"binary_accuracy\", AUC(name=\"auc\")])\n\ndatagen_Eff = ImageDataGenerator(\n    rescale=1./255,  \n    validation_split=0.2  # 80-20 train-val split\n)\ntrain_generator_Eff = datagen_Eff.flow_from_dataframe(\n    dataframe=df_sampled, directory=None, x_col=\"location\", y_col=\"label\",\n    target_size=(224, 224), batch_size=64, subset=\"training\",\n    class_mode=\"binary\"  # Important for Binary Classification\n)\n\nval_generator_Eff = datagen_Eff.flow_from_dataframe(\n    dataframe=df_sampled, directory=None, x_col=\"location\", y_col=\"label\",\n    target_size=(224, 224), batch_size=64, subset=\"validation\",\n    class_mode=\"binary\"\n)\n\n# Train the Model\nmodel_Eff.fit(train_generator_Eff, validation_data=val_generator_Eff, epochs=10)\n\n# Evaluate on Validation Set\nval_loss, val_acc, val_auc = model_Eff.evaluate(val_generator_Eff)\nprint(f\"✅ EfficientNetB0 Validation Accuracy: {val_acc:.4f}, AUC: {val_auc:.4f}\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T00:35:58.090309Z","iopub.execute_input":"2025-02-15T00:35:58.090686Z","iopub.status.idle":"2025-02-15T00:39:26.047911Z","shell.execute_reply.started":"2025-02-15T00:35:58.090653Z","shell.execute_reply":"2025-02-15T00:39:26.046782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#ResNet50 \nRes_model = ResNet50(weights='imagenet', include_top=False, input_shape=(96, 96, 3))\nRes_model.trainable = True  # Allow fine-tuning\n\n# Add Custom Classifier for Binary Classification\nx = GlobalAveragePooling2D()(Res_model.output)  # Convert feature maps to vector\nx = Dense(128, activation=\"relu\")(x)  # Optional Dense Layer\nx = Dense(1, activation=\"sigmoid\")(x)  # Binary Classification (Sigmoid Output)\n\n# Create Model\nmodel_Res = Model(inputs=Res_model.input, outputs=x)\n\n# Compile Model (Use Binary Loss & AUC Metric)\nmodel_Res.compile(optimizer=Adam(learning_rate=1e-5), \n              loss=\"binary_crossentropy\", \n              metrics=[\"binary_accuracy\", AUC(name=\"auc\")])\n\ndatagen_Res = ImageDataGenerator(\n    rescale=1./255,  \n    validation_split=0.2  # 80-20 train-val split\n)\ntrain_generator_Res = datagen_Res.flow_from_dataframe(\n    dataframe=df_sampled, directory=None, x_col=\"location\", y_col=\"label\",\n    target_size=(96, 96), batch_size=64, subset=\"training\",\n    class_mode=\"binary\"\n)\n\nval_generator_Res = datagen_Res.flow_from_dataframe(\n    dataframe=df_sampled, directory=None, x_col=\"location\", y_col=\"label\",\n    target_size=(96, 96), batch_size=64, subset=\"validation\",\n    class_mode=\"binary\"\n)\n\n# Train the Model\nmodel_Res.fit(train_generator_Res, validation_data=val_generator_Res, epochs=10)\n\n# Evaluate on Validation Set\nval_loss, val_acc, val_auc = model_Res.evaluate(val_generator_Res)\nprint(f\"✅ ResNet50 Validation Accuracy: {val_acc:.4f}, AUC: {val_auc:.4f}\")# Load ResNet50 WITHOUT Top Layers (Feature Extractor)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T00:39:26.048960Z","iopub.execute_input":"2025-02-15T00:39:26.049227Z","iopub.status.idle":"2025-02-15T00:41:28.210292Z","shell.execute_reply.started":"2025-02-15T00:39:26.049204Z","shell.execute_reply":"2025-02-15T00:41:28.209491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Load DenseNet-169 WITHOUT Top Layers (Feature Extractor)\nDenseN_model = DenseNet169(weights='imagenet', include_top=False, input_shape=(96, 96, 3))\nDenseN_model.trainable = True  # Allow fine-tuning\n\n# Add Custom Classifier for Binary Classification\nx = GlobalAveragePooling2D()(DenseN_model.output)  # Convert feature maps to vector\nx = Dense(128, activation=\"relu\")(x)  # Optional Dense Layer\nx = Dense(1, activation=\"sigmoid\")(x)  # Binary Classification (Sigmoid Output)\n\n# Create Model\nmodel_DenseN = Model(inputs=DenseN_model.input, outputs=x)\n\n# Compile Model (Use Binary Loss & AUC Metric)\nmodel_DenseN.compile(optimizer=Adam(learning_rate=1e-5), \n              loss=\"binary_crossentropy\", \n              metrics=[\"binary_accuracy\", AUC(name=\"auc\")])\n\ndatagen_DenseN = ImageDataGenerator(\n    rescale=1./255,  \n    validation_split=0.2  # 80-20 train-val split\n)\ntrain_generator_DenseN = datagen_DenseN.flow_from_dataframe(\n    dataframe=df_sampled, directory=None, x_col=\"location\", y_col=\"label\",\n    target_size=(96, 96), batch_size=64, subset=\"training\",\n    class_mode=\"binary\"\n)\n\nval_generator_DenseN = datagen_DenseN.flow_from_dataframe(\n    dataframe=df_sampled, directory=None, x_col=\"location\", y_col=\"label\",\n    target_size=(96, 96), batch_size=64, subset=\"validation\",\n    class_mode=\"binary\"\n)\n\n# Train the Model\nmodel_DenseN.fit(train_generator_DenseN, validation_data=val_generator_DenseN, epochs=10)\n\n# Evaluate on Validation Set\nval_loss, val_acc, val_auc = model_DenseN.evaluate(val_generator_DenseN)\nprint(f\"✅ DenseNet-169 Validation Accuracy: {val_acc:.4f}, AUC: {val_auc:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T00:41:28.212813Z","iopub.execute_input":"2025-02-15T00:41:28.213046Z","iopub.status.idle":"2025-02-15T00:47:59.031167Z","shell.execute_reply.started":"2025-02-15T00:41:28.213026Z","shell.execute_reply":"2025-02-15T00:47:59.030217Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Step Four Results and Analysis**\n\nSo taking a look epoch by epoch in three trained CNN.","metadata":{}},{"cell_type":"code","source":"import time\n\n# Train EfficientNetB0\nstart_time = time.time()\nhistory_efficient = model_Eff.fit(train_generator_Eff, validation_data=val_generator_Eff, epochs=10)\nefficient_time = time.time() - start_time\n\n# Train ResNet50\nstart_time = time.time()\nhistory_resnet = model_Res.fit(train_generator_Res, validation_data=val_generator_Res, epochs=10)\nresnet_time = time.time() - start_time\n\n# Train DenseNet-169\nstart_time = time.time()\nhistory_densenet = model_DenseN.fit(train_generator_DenseN, validation_data=val_generator_DenseN, epochs=10)\ndensenet_time = time.time() - start_time\n\n# Extract Performance Metrics\nepochs = range(1, 11)\n\nefficient_acc = history_efficient.history[\"val_binary_accuracy\"]\nefficient_auc = history_efficient.history[\"val_auc\"]\n\nresnet_acc = history_resnet.history[\"val_binary_accuracy\"]\nresnet_auc = history_resnet.history[\"val_auc\"]\n\ndensenet_acc = history_densenet.history[\"val_binary_accuracy\"]\ndensenet_auc = history_densenet.history[\"val_auc\"]\n\n# 📌 Plot Accuracy\nplt.figure(figsize=(12, 5))\n\nplt.subplot(1, 2, 1)\nplt.plot(epochs, efficient_acc, label=\"EfficientNetB0\", marker=\"o\")\nplt.plot(epochs, resnet_acc, label=\"ResNet50\", marker=\"o\")\nplt.plot(epochs, densenet_acc, label=\"DenseNet-169\", marker=\"o\")\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Validation Accuracy\")\nplt.title(\"Model Comparison: Validation Accuracy\")\nplt.legend()\n\n# 📌 Plot AUC\nplt.subplot(1, 2, 2)\nplt.plot(epochs, efficient_auc, label=\"EfficientNetB0\", marker=\"o\")\nplt.plot(epochs, resnet_auc, label=\"ResNet50\", marker=\"o\")\nplt.plot(epochs, densenet_auc, label=\"DenseNet-169\", marker=\"o\")\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Validation AUC\")\nplt.title(\"Model Comparison: AUC Score\")\nplt.legend()\n\nplt.show()\n\n# 📌 Compare Training Speed\nprint(f\"⏳ Training Time (Seconds):\")\nprint(f\"✅ EfficientNetB0: {efficient_time:.2f} sec\")\nprint(f\"✅ ResNet50: {resnet_time:.2f} sec\")\nprint(f\"✅ DenseNet-169: {densenet_time:.2f} sec\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T00:47:59.032535Z","iopub.execute_input":"2025-02-15T00:47:59.032881Z","iopub.status.idle":"2025-02-15T00:49:51.583134Z","shell.execute_reply.started":"2025-02-15T00:47:59.032848Z","shell.execute_reply":"2025-02-15T00:49:51.582130Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Fine Tuning**\n\nHere we have a the same winner by time and by performace DenseNet-169. So lest play with some parameters an see if we can improve even more.\n","metadata":{}},{"cell_type":"code","source":"import keras_tuner as kt\n\ndef build_model(hp):\n    base_model = DenseNet169(weights=\"imagenet\", include_top=False, input_shape=(96, 96, 3))\n    base_model.trainable = True  # Allow fine-tuning\n    \n    x = GlobalAveragePooling2D()(base_model.output)\n    x = Dense(128, activation=\"relu\")(x)\n    x = Dense(1, activation=\"sigmoid\")(x)  # Binary classification\n    \n    model = Model(inputs=base_model.input, outputs=x)\n    \n    # 📌 Hyperparameters for tuning\n    learning_rate = hp.Choice(\"learning_rate\", [1e-3, 1e-4, 1e-5])\n    optimizer = hp.Choice(\"optimizer\", [\"adam\", \"sgd\", \"rmsprop\"])\n    loss = hp.Choice(\"loss\", [\"binary_crossentropy\", \"hinge\"])\n    \n    # Select optimizer\n    if optimizer == \"adam\":\n        opt = Adam(learning_rate=learning_rate)\n    elif optimizer == \"sgd\":\n        opt = SGD(learning_rate=learning_rate, momentum=0.9)\n    elif optimizer == \"rmsprop\":\n        opt = RMSprop(learning_rate=learning_rate)\n    \n    # Compile the model\n    model.compile(optimizer=opt, loss=loss, metrics=[\"binary_accuracy\", tf.keras.metrics.AUC(name=\"auc\")])\n    \n    return model\n\ntuner = kt.GridSearch(\n    hypermodel=build_model,\n    objective=\"val_binary_accuracy\",\n    max_trials=10,  # Number of different combinations to try\n    executions_per_trial=1  # Run each configuration once\n)\ndatagen = tf.keras.preprocessing.image.ImageDataGenerator(\n    rescale=1./255,  \n    validation_split=0.2\n)\n\ntrain_generator = datagen.flow_from_dataframe(\n    dataframe=df_sampled, directory=None, x_col=\"location\", y_col=\"label\",\n    target_size=(96, 96), batch_size=64, subset=\"training\",\n    class_mode=\"binary\"\n)\n\nval_generator = datagen.flow_from_dataframe(\n    dataframe=df_sampled, directory=None, x_col=\"location\", y_col=\"label\",\n    target_size=(96, 96), batch_size=64, subset=\"validation\",\n    class_mode=\"binary\"\n)\n\n# Perform hyperparameter search\ntuner.search(train_generator, validation_data=val_generator, epochs=5)\n\n# Get the best model\nbest_hps = tuner.get_best_hyperparameters(num_trials=1)[0]\nbest_model = tuner.get_best_models(num_models=1)[0]\n\nprint(f\"Best Learning Rate: {best_hps.get('learning_rate')}\")\nprint(f\"Best Optimizer: {best_hps.get('optimizer')}\")\nprint(f\"Best Loss Function: {best_hps.get('loss')}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T00:49:51.584458Z","iopub.execute_input":"2025-02-15T00:49:51.584839Z","iopub.status.idle":"2025-02-15T00:50:01.914784Z","shell.execute_reply.started":"2025-02-15T00:49:51.584809Z","shell.execute_reply":"2025-02-15T00:50:01.913766Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Step 5. Conclusion**\n\nSo after fine tuning we got some info of the best parameters and we will use all images.\nBe patient.\n\nBest val_binary_accuracy So Far: 0.8235294222831726\n\nTotal elapsed time: 00h 50m 51s\n\nBest Learning Rate: 0.0001\n\nBest Optimizer: adam\n\nBest Loss Function: hinge\n","metadata":{}},{"cell_type":"code","source":"#Data for final model\n\ndatagen_final = ImageDataGenerator(\n    rescale=1./255,  \n    validation_split=0.2  # 80-20 train-val split\n)\ntrain_generator_final = datagen.flow_from_dataframe(\n    dataframe=df_train, directory=None, x_col=\"location\", y_col=\"label\",\n    target_size=(96, 96), batch_size=64, subset=\"training\",\n    class_mode=\"binary\"\n)\n\nval_generator_final = datagen.flow_from_dataframe(\n    dataframe=df_train, directory=None, x_col=\"location\", y_col=\"label\",\n    target_size=(96, 96), batch_size=64, subset=\"validation\",\n    class_mode=\"binary\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T00:56:32.104917Z","iopub.execute_input":"2025-02-15T00:56:32.105215Z","iopub.status.idle":"2025-02-15T00:59:47.488622Z","shell.execute_reply.started":"2025-02-15T00:56:32.105194Z","shell.execute_reply":"2025-02-15T00:59:47.487880Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Final model with best features\nfinal_model = DenseNet169(weights='imagenet', include_top=False, input_shape=(96, 96, 3))\nfinal_model.trainable = True  # Allow fine-tuning\n\n# Add Custom Classifier for Binary Classification\nx = GlobalAveragePooling2D()(final_model.output)  # Convert feature maps to vector\nx = Dense(128, activation=\"relu\")(x)  # Optional Dense Layer\nx = Dense(1, activation=\"sigmoid\")(x)  # Binary Classification (Sigmoid Output)\n\n# Create Model\nmodel_final = Model(inputs=final_model.input, outputs=x)\n\n# Compile Model (Use Binary Loss & AUC Metric)\nmodel_final.compile(optimizer=Adam(learning_rate=0.0001), \n              loss=\"hinge\", \n              metrics=[\"binary_accuracy\", AUC(name=\"auc\")])\n\n# Train the Model\nmodel_final.fit(train_generator_final, validation_data=val_generator_final, epochs=5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T00:59:54.446478Z","iopub.execute_input":"2025-02-15T00:59:54.446820Z","iopub.status.idle":"2025-02-15T02:01:00.559428Z","shell.execute_reply.started":"2025-02-15T00:59:54.446794Z","shell.execute_reply":"2025-02-15T02:01:00.558170Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_loss, val_acc, val_auc = model_final.evaluate(val_generator)\nprint(f\"✅ FINAL MODEL on Validation Accuracy: {val_acc:.4f}, AUC: {val_auc:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T02:04:53.026439Z","iopub.execute_input":"2025-02-15T02:04:53.026736Z","iopub.status.idle":"2025-02-15T02:05:00.217351Z","shell.execute_reply.started":"2025-02-15T02:04:53.026713Z","shell.execute_reply":"2025-02-15T02:05:00.216567Z"}},"outputs":[],"execution_count":null}]}