{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":6243,"databundleVersionId":868544,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import libraries\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nimport os, glob, cv2\nfrom PIL import Image, ImageFile \n%matplotlib inline\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.applications import VGG16, ResNet50\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Dense, Flatten, Dropout\nfrom tensorflow.keras.models import Model, Sequential\nfrom tensorflow.keras.applications.vgg16 import preprocess_input\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.optimizers import Adam\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom tensorflow.keras.metrics import Recall\nfrom sklearn.metrics import accuracy_score, recall_score, precision_score, classification_report, confusion_matrix\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:01:58.558153Z","iopub.execute_input":"2025-06-06T11:01:58.558441Z","iopub.status.idle":"2025-06-06T11:02:08.603576Z","shell.execute_reply.started":"2025-06-06T11:01:58.558417Z","shell.execute_reply":"2025-06-06T11:02:08.602957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import filepaths\nroot_dir = '../input/intel-mobileodt-cervical-cancer-screening'\ntrain_dir = os.path.join(root_dir,'train', 'train')\n\ntype1_dir = os.path.join(train_dir, 'Type_1')\ntype2_dir = os.path.join(train_dir, 'Type_2')\ntype3_dir = os.path.join(train_dir, 'Type_3')\n\ntrain_type1_files = glob.glob(type1_dir+'/*.jpg')\ntrain_type2_files = glob.glob(type2_dir+'/*.jpg')\ntrain_type3_files = glob.glob(type3_dir+'/*.jpg')\n\nadded_type1_files = glob.glob(os.path.join(root_dir, \"additional_Type_1_v2\", \"Type_1\")+'/*.jpg')\nadded_type2_files = glob.glob(os.path.join(root_dir, \"additional_Type_2_v2\", \"Type_2\")+'/*.jpg')\nadded_type3_files = glob.glob(os.path.join(root_dir, \"additional_Type_3_v2\", \"Type_3\")+'/*.jpg')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:02:08.604837Z","iopub.execute_input":"2025-06-06T11:02:08.605385Z","iopub.status.idle":"2025-06-06T11:02:08.711494Z","shell.execute_reply.started":"2025-06-06T11:02:08.605364Z","shell.execute_reply":"2025-06-06T11:02:08.710695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"type1_files = train_type1_files + added_type1_files\ntype2_files = train_type2_files + added_type2_files\ntype3_files = train_type3_files + added_type3_files\n\nprint(f'''Type 1 files for training: {len(type1_files)} \nType 2 files for training: {len(type2_files)}\nType 3 files for training: {len(type3_files)}''')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:02:08.712336Z","iopub.execute_input":"2025-06-06T11:02:08.712612Z","iopub.status.idle":"2025-06-06T11:02:08.717351Z","shell.execute_reply.started":"2025-06-06T11:02:08.712592Z","shell.execute_reply":"2025-06-06T11:02:08.716638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# get data for testing\n\ntest_dir = os.path.join(root_dir,'test', 'test')\n\ntest_files = glob.glob(test_dir+'/*.jpg')\n\nprint(f'Test files for training: {len(test_files)}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:02:08.718709Z","iopub.execute_input":"2025-06-06T11:02:08.718962Z","iopub.status.idle":"2025-06-06T11:02:08.751254Z","shell.execute_reply.started":"2025-06-06T11:02:08.718945Z","shell.execute_reply":"2025-06-06T11:02:08.750542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# create dataframe of file and labels\nfiles = {'filepath': type1_files + type2_files + type3_files,\n          'label': ['Type 1']* len(type1_files) + ['Type 2']* len(type2_files) + ['Type 3']* len(type3_files)}\n\nfiles_df = pd.DataFrame(files).sample(frac=1, random_state= 1).reset_index(drop=True)\nfiles_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:02:08.842600Z","iopub.execute_input":"2025-06-06T11:02:08.843225Z","iopub.status.idle":"2025-06-06T11:02:08.866209Z","shell.execute_reply.started":"2025-06-06T11:02:08.843201Z","shell.execute_reply":"2025-06-06T11:02:08.865638Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DATA EXPLORATION","metadata":{}},{"cell_type":"code","source":"files_df.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:02:15.534720Z","iopub.execute_input":"2025-06-06T11:02:15.535285Z","iopub.status.idle":"2025-06-06T11:02:15.555789Z","shell.execute_reply.started":"2025-06-06T11:02:15.535256Z","shell.execute_reply":"2025-06-06T11:02:15.555116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check for damaged files\nbad_files = []\nfor path in (files_df['filepath'].values):\n    try:\n        img = Image.open(path)\n    except:\n        index = files_df[files_df['filepath']==path].index.values[0]\n        bad_files.append(index)\nprint(len(bad_files))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:02:22.431674Z","iopub.execute_input":"2025-06-06T11:02:22.432038Z","iopub.status.idle":"2025-06-06T11:04:06.379697Z","shell.execute_reply.started":"2025-06-06T11:02:22.432011Z","shell.execute_reply":"2025-06-06T11:04:06.378741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# drop the damaged files\nfiles_df.drop(bad_files, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:04:06.381135Z","iopub.execute_input":"2025-06-06T11:04:06.381652Z","iopub.status.idle":"2025-06-06T11:04:06.387485Z","shell.execute_reply.started":"2025-06-06T11:04:06.381632Z","shell.execute_reply":"2025-06-06T11:04:06.386848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# get count of each type \ntype_count = pd.DataFrame(files_df['label'].value_counts(normalize=True)*100)\ntype_count","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:04:06.388343Z","iopub.execute_input":"2025-06-06T11:04:06.388631Z","iopub.status.idle":"2025-06-06T11:04:06.406714Z","shell.execute_reply.started":"2025-06-06T11:04:06.388606Z","shell.execute_reply":"2025-06-06T11:04:06.406173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# display barplot of type count\nplt.figure(figsize = (15, 6))\nsns.barplot(x= type_count['proportion'], y= type_count.index.to_list())\nplt.title('Cervical Cancer Type Distribution')\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:04:06.408730Z","iopub.execute_input":"2025-06-06T11:04:06.409044Z","iopub.status.idle":"2025-06-06T11:04:06.593525Z","shell.execute_reply.started":"2025-06-06T11:04:06.409025Z","shell.execute_reply":"2025-06-06T11:04:06.592686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# display sample images of types\nfor label in ('Type 1', 'Type 2', 'Type 3'):\n    filepaths = files_df[files_df['label']==label]['filepath'].values[:5]\n    fig = plt.figure(figsize= (15, 6))\n    for i, path in enumerate(filepaths):\n        img = cv2.imread(path)\n        img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR)\n        img = cv2.resize(img, (224, 224))\n        fig.add_subplot(1, 5, i+1)\n        plt.imshow(img)\n        plt.subplots_adjust(hspace=0.5)\n        plt.axis(False)\n        plt.title(label)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T00:00:49.315562Z","iopub.execute_input":"2025-06-06T00:00:49.315768Z","iopub.status.idle":"2025-06-06T00:00:52.404169Z","shell.execute_reply.started":"2025-06-06T00:00:49.315753Z","shell.execute_reply":"2025-06-06T00:00:52.403537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#  split the data into train  and validation set\ntrain_df, eval_df = train_test_split(files_df, test_size= 0.2, stratify= files_df['label'], random_state= 1)\nval_df, test_df = train_test_split(eval_df, test_size= 0.5, stratify= eval_df['label'], random_state= 1)\nprint(len(train_df), len(val_df), len(test_df))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:04:06.594425Z","iopub.execute_input":"2025-06-06T11:04:06.594762Z","iopub.status.idle":"2025-06-06T11:04:06.609641Z","shell.execute_reply.started":"2025-06-06T11:04:06.594743Z","shell.execute_reply":"2025-06-06T11:04:06.608982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# loads images from dataframe\ndef load_images(dataframe):\n    features = []\n    filepaths = dataframe['filepath'].values\n    labels = dataframe['label'].values\n    \n    for path in filepaths:\n        img = cv2.imread(path)\n        resized_img = cv2.resize(img, (180, 180))\n        features.append(np.array(resized_img))\n    return np.array(features), np.array(labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:04:06.610464Z","iopub.execute_input":"2025-06-06T11:04:06.610650Z","iopub.status.idle":"2025-06-06T11:04:06.618413Z","shell.execute_reply.started":"2025-06-06T11:04:06.610636Z","shell.execute_reply":"2025-06-06T11:04:06.617825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# load training and evaluation data\ntrain_features, train_labels = load_images(train_df)\nval_features, val_labels = load_images(val_df)\ntest_features, test_labels = load_images(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:04:06.619145Z","iopub.execute_input":"2025-06-06T11:04:06.619396Z","iopub.status.idle":"2025-06-06T11:20:14.257690Z","shell.execute_reply.started":"2025-06-06T11:04:06.619378Z","shell.execute_reply":"2025-06-06T11:20:14.257010Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check lengths of training and evaluation sets\nprint(f'''train features:{len(train_features)},train labels:{len(train_labels)}\n    val features:{len(val_features)}, val labels:{len(val_labels)}\n    test features:{len(test_features)}, test labels:{len(test_labels)}''') ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:20:14.258564Z","iopub.execute_input":"2025-06-06T11:20:14.258760Z","iopub.status.idle":"2025-06-06T11:20:14.264082Z","shell.execute_reply.started":"2025-06-06T11:20:14.258746Z","shell.execute_reply":"2025-06-06T11:20:14.263141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# get image shape\nInputShape = train_features[766].shape\nprint(InputShape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:20:14.264888Z","iopub.execute_input":"2025-06-06T11:20:14.265136Z","iopub.status.idle":"2025-06-06T11:20:14.283984Z","shell.execute_reply.started":"2025-06-06T11:20:14.265115Z","shell.execute_reply":"2025-06-06T11:20:14.283230Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# encode the labels\nle = LabelEncoder().fit(['Type 1', 'Type 2', 'Type 3'])\ny_train = le.transform(train_labels)\ny_val = le.transform(val_labels)\ny_test = le.transform(test_labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:20:14.286084Z","iopub.execute_input":"2025-06-06T11:20:14.286294Z","iopub.status.idle":"2025-06-06T11:20:14.304396Z","shell.execute_reply.started":"2025-06-06T11:20:14.286278Z","shell.execute_reply":"2025-06-06T11:20:14.303590Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# normalize the features\nX_train = train_features/255\nX_val  = val_features/255\nX_test  = test_features/255","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:20:14.305290Z","iopub.execute_input":"2025-06-06T11:20:14.305521Z","iopub.status.idle":"2025-06-06T11:20:21.560596Z","shell.execute_reply.started":"2025-06-06T11:20:14.305497Z","shell.execute_reply":"2025-06-06T11:20:21.559904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"conv_base = VGG16(weights= 'imagenet',\n                  include_top= False,\n                  input_shape= (180, 180, 3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:20:21.561395Z","iopub.execute_input":"2025-06-06T11:20:21.561667Z","iopub.status.idle":"2025-06-06T11:20:24.577415Z","shell.execute_reply.started":"2025-06-06T11:20:21.561642Z","shell.execute_reply":"2025-06-06T11:20:24.576816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the ResNet50 base model with ImageNet weights, excluding the top classifier layers\nconv_base_2 = ResNet50(weights='imagenet',\n                       include_top=False,\n                       input_shape=(180, 180, 3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:20:24.578153Z","iopub.execute_input":"2025-06-06T11:20:24.578368Z","iopub.status.idle":"2025-06-06T11:20:26.286385Z","shell.execute_reply.started":"2025-06-06T11:20:24.578351Z","shell.execute_reply":"2025-06-06T11:20:26.285553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for layer in conv_base.layers[:-5]:\n    layer.trainable= False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:20:26.287419Z","iopub.execute_input":"2025-06-06T11:20:26.287644Z","iopub.status.idle":"2025-06-06T11:20:26.292074Z","shell.execute_reply.started":"2025-06-06T11:20:26.287626Z","shell.execute_reply":"2025-06-06T11:20:26.291230Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Using VGG16 model\nconv_base.trainable = False\n\nmodel = Sequential([conv_base,\n                   Flatten(),\n                    Dense(180, activation='relu'),\n                    Dropout(0.5),\n                    Dense(3, activation='softmax')\n                   ])\n\nmodel.compile(optimizer= Adam(0.0001),\n              loss= 'sparse_categorical_crossentropy',\n              metrics= [\"accuracy\"]\n             )\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:20:26.293047Z","iopub.execute_input":"2025-06-06T11:20:26.293316Z","iopub.status.idle":"2025-06-06T11:20:26.354826Z","shell.execute_reply.started":"2025-06-06T11:20:26.293297Z","shell.execute_reply":"2025-06-06T11:20:26.354096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"early_stopping = EarlyStopping(\n    monitor='val_loss',\n    patience=5,\n    restore_best_weights=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:20:26.355657Z","iopub.execute_input":"2025-06-06T11:20:26.356026Z","iopub.status.idle":"2025-06-06T11:20:26.359650Z","shell.execute_reply.started":"2025-06-06T11:20:26.356000Z","shell.execute_reply":"2025-06-06T11:20:26.358905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(\n        X_train, y_train,\n        validation_data=(X_val, y_val),\n        epochs=100,\n        batch_size=32,\n        callbacks= [early_stopping])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:20:26.360566Z","iopub.execute_input":"2025-06-06T11:20:26.360824Z","iopub.status.idle":"2025-06-06T11:27:04.662511Z","shell.execute_reply.started":"2025-06-06T11:20:26.360802Z","shell.execute_reply":"2025-06-06T11:27:04.661689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Using ResNet50 model\nfor layer in conv_base.layers[:-5]:\n    layer.trainable= False\n\nconv_base_2.trainable = False\n\nmodel_2 = Sequential([conv_base_2,\n                      Flatten(),\n                     Dense(180, activation='relu'),\n                      Dropout(0.5),\n                     Dense(3, activation='softmax')\n                     ])\n\nmodel_2.compile(optimizer=Adam(learning_rate=1e-4),\n              loss='sparse_categorical_crossentropy',\n              metrics=['accuracy']\n             )\nmodel_2.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:27:04.663736Z","iopub.execute_input":"2025-06-06T11:27:04.663984Z","iopub.status.idle":"2025-06-06T11:27:04.729374Z","shell.execute_reply.started":"2025-06-06T11:27:04.663964Z","shell.execute_reply":"2025-06-06T11:27:04.728770Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the model\nhistory_2 = model_2.fit(X_train, y_train,\n    validation_data=(X_val, y_val),\n    epochs=100,\n    batch_size=32,\n    callbacks=[early_stopping]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:27:04.730098Z","iopub.execute_input":"2025-06-06T11:27:04.730400Z","iopub.status.idle":"2025-06-06T11:29:09.262831Z","shell.execute_reply.started":"2025-06-06T11:27:04.730357Z","shell.execute_reply":"2025-06-06T11:29:09.261912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# read training history into dataframe\nhistory_df = pd.DataFrame(history.history)\n\nhistory_df_2 = pd.DataFrame(history_2.history)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:32:23.344452Z","iopub.execute_input":"2025-06-06T11:32:23.345208Z","iopub.status.idle":"2025-06-06T11:32:23.351682Z","shell.execute_reply.started":"2025-06-06T11:32:23.345180Z","shell.execute_reply":"2025-06-06T11:32:23.350743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# display training and validation history\n\n# display history of accuracy\nplt.figure(figsize= (15,6))\nplt.subplot(1,2,1)\nplt.plot(history_df['accuracy'], label= 'accracy' )\nplt.plot(history_df['val_accuracy'], label= 'val_accuracy')\n# history_df[['accuracy', 'val_accuracy']]\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.title('Training and Validation Accuracy History')\nplt.legend()\n\n# display history of loss\nplt.subplot(1,2,2)\nplt.plot(history_df['loss'], label= 'loss')\nplt.plot(history_df['val_loss'], label= 'val_loss')\n# history_df[['loss', 'val_loss']].plot()\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.title('Training and Validation Loss History')\nplt.legend()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:32:23.640461Z","iopub.execute_input":"2025-06-06T11:32:23.640795Z","iopub.status.idle":"2025-06-06T11:32:24.070570Z","shell.execute_reply.started":"2025-06-06T11:32:23.640767Z","shell.execute_reply":"2025-06-06T11:32:24.069769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# display training and validation history\n\n# display history of accurracy\nplt.figure(figsize= (15,6))\nplt.subplot(1,2,1)\nplt.plot(history_df_2['accuracy'], label= 'accuracy' )\nplt.plot(history_df_2['val_accuracy'], label= 'val_accuracy')\n# history_df[['accuracy', 'val_acc']]\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.title('Training and Validation Accuracy History')\nplt.legend()\n\n# display history of loss\nplt.subplot(1,2,2)\nplt.plot(history_df_2['loss'], label= 'loss')\nplt.plot(history_df_2['val_loss'], label= 'val_loss')\n# history_df[['loss', 'val_loss']].plot()\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.title('Training and Validation Loss History')\nplt.legend()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:33:37.688713Z","iopub.execute_input":"2025-06-06T11:33:37.689473Z","iopub.status.idle":"2025-06-06T11:33:38.069679Z","shell.execute_reply.started":"2025-06-06T11:33:37.689445Z","shell.execute_reply":"2025-06-06T11:33:38.068828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.evaluate(X_test, y_test)\nmodel_2.evaluate(X_test, y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T11:38:17.307140Z","iopub.execute_input":"2025-06-06T11:38:17.307997Z","iopub.status.idle":"2025-06-06T11:38:21.859344Z","shell.execute_reply.started":"2025-06-06T11:38:17.307969Z","shell.execute_reply":"2025-06-06T11:38:21.858611Z"}},"outputs":[],"execution_count":null}]}