{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n#import numpy as np # linear algebra\n#import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport numpy as np              # For numerical operations\nimport pandas as pd             # For working with CSVs and DataFrames\nimport matplotlib.pyplot as plt # For plotting\nimport cv2                      # For image processing\nimport seaborn as sns           # For nicer plots\nfrom sklearn.utils import shuffle                # To shuffle datasets\nfrom sklearn.metrics import confusion_matrix     # To evaluate classification results\nfrom sklearn.model_selection import train_test_split # For splitting data into train/test\nimport itertools               # Useful for looping combinations (e.g., confusion matrix plotting)\nimport shutil                  # File operations like copy, move\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-24T23:55:58.301704Z","iopub.execute_input":"2025-05-24T23:55:58.302404Z","iopub.status.idle":"2025-05-24T23:55:58.306892Z","shell.execute_reply.started":"2025-05-24T23:55:58.302379Z","shell.execute_reply":"2025-05-24T23:55:58.306050Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ndata_dir = \"/kaggle/input/histopathologic-cancer-detection\"\n\nprint(os.listdir(data_dir))\n\n# Total Samples Available\nprint('Train Images =', len(os.listdir(os.path.join(data_dir, 'train'))))\nprint('Test Images =', len(os.listdir(os.path.join(data_dir, 'test'))))\n\n# Read train labels CSV\ndf = pd.read_csv(os.path.join(data_dir, 'train_labels.csv'))\nprint('Shape of DataFrame:', df.shape)\ndf.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T23:56:07.450189Z","iopub.execute_input":"2025-05-24T23:56:07.450468Z","iopub.status.idle":"2025-05-24T23:56:10.085777Z","shell.execute_reply.started":"2025-05-24T23:56:07.450446Z","shell.execute_reply":"2025-05-24T23:56:10.085054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_DIR = '/kaggle/input/histopathologic-cancer-detection/train/'\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T23:56:14.265568Z","iopub.execute_input":"2025-05-24T23:56:14.266365Z","iopub.status.idle":"2025-05-24T23:56:14.269689Z","shell.execute_reply.started":"2025-05-24T23:56:14.266341Z","shell.execute_reply":"2025-05-24T23:56:14.268895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig = plt.figure(figsize = (20,8))\nindex = 1\nfor i in np.random.randint(low = 0, high = df.shape[0], size = 10):\n    file = TRAIN_DIR + df.iloc[i]['id'] + '.tif'\n    img = cv2.imread(file)\n    ax = fig.add_subplot(2, 5, index)\n    ax.imshow(img, cmap = 'gray')\n    index = index + 1\n    color = ['green' if df.iloc[i].label == 1 else 'red'][0]\n    ax.set_title(df.iloc[i].label, fontsize = 18, color = color)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T23:56:17.053563Z","iopub.execute_input":"2025-05-24T23:56:17.053825Z","iopub.status.idle":"2025-05-24T23:56:18.413474Z","shell.execute_reply.started":"2025-05-24T23:56:17.053807Z","shell.execute_reply":"2025-05-24T23:56:18.412677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Removing problematic images\ndf = df[df['id'] != 'dd6dfed324f9fcb6f93f46f32fc800f2ec196be2']  # corrupted image\ndf = df[df['id'] != '9369c7278ec8bcc6c880d99194de09fc2bd4efbe']  # black image\n\nprint(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T23:56:23.393976Z","iopub.execute_input":"2025-05-24T23:56:23.394284Z","iopub.status.idle":"2025-05-24T23:56:23.444331Z","shell.execute_reply.started":"2025-05-24T23:56:23.394261Z","shell.execute_reply":"2025-05-24T23:56:23.443638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels_count = df.label.value_counts()\n\nplt.pie(labels_count, labels=['No Cancer', 'Cancer'], startangle=180, \n        autopct='%1.1f', colors=['#00ff99','#FF96A7'], shadow=True)\nplt.figure(figsize=(16,16))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T23:56:34.349904Z","iopub.execute_input":"2025-05-24T23:56:34.350622Z","iopub.status.idle":"2025-05-24T23:56:34.436138Z","shell.execute_reply.started":"2025-05-24T23:56:34.350591Z","shell.execute_reply":"2025-05-24T23:56:34.435322Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SAMPLE_SIZE = 80000\n# take a random sample of class 0 with size equal to num samples in class 1\ndf_0 = df[df['label'] == 0].sample(SAMPLE_SIZE, random_state = 0)\n# filter out class 1\ndf_1 = df[df['label'] == 1].sample(SAMPLE_SIZE, random_state = 0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T23:56:46.345708Z","iopub.execute_input":"2025-05-24T23:56:46.346015Z","iopub.status.idle":"2025-05-24T23:56:46.382397Z","shell.execute_reply.started":"2025-05-24T23:56:46.345976Z","shell.execute_reply":"2025-05-24T23:56:46.381812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# concat the dataframes\ndf_train = pd.concat([df_0, df_1], axis = 0).reset_index(drop = True)\n# shuffle\ndf_train = shuffle(df_train)\n\nprint(df_train['label'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T23:56:56.985305Z","iopub.execute_input":"2025-05-24T23:56:56.985943Z","iopub.status.idle":"2025-05-24T23:56:57.018217Z","shell.execute_reply.started":"2025-05-24T23:56:56.985918Z","shell.execute_reply":"2025-05-24T23:56:57.017621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport os\n\n# Target labels for stratification\ny = df_train['label']\n\n# Stratified split\ndf_train, df_val = train_test_split(df_train, test_size=0.1, random_state=0, stratify=y)\n\n# Update base_dir path for Kaggle writable directory\nbase_dir = '/kaggle/working/base_dir'\ntrain_dir = os.path.join(base_dir, 'train_dir')\nval_dir = os.path.join(base_dir, 'val_dir')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T23:57:11.849294Z","iopub.execute_input":"2025-05-24T23:57:11.850026Z","iopub.status.idle":"2025-05-24T23:57:11.913569Z","shell.execute_reply.started":"2025-05-24T23:57:11.849973Z","shell.execute_reply":"2025-05-24T23:57:11.913029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nbase_dir = '/kaggle/working/base_dir'\nos.makedirs(base_dir, exist_ok=True)  # Create base_dir if it doesn't exist\n\ntrain_dir = os.path.join(base_dir, 'train_dir')\nos.makedirs(train_dir, exist_ok=True)\n\nval_dir = os.path.join(base_dir, 'val_dir')\nos.makedirs(val_dir, exist_ok=True)\n\n# Create class subfolders inside train_dir\nos.makedirs(os.path.join(train_dir, '0'), exist_ok=True)\nos.makedirs(os.path.join(train_dir, '1'), exist_ok=True)\n\n# Create class subfolders inside val_dir\nos.makedirs(os.path.join(val_dir, '0'), exist_ok=True)\nos.makedirs(os.path.join(val_dir, '1'), exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T23:57:24.737743Z","iopub.execute_input":"2025-05-24T23:57:24.738280Z","iopub.status.idle":"2025-05-24T23:57:24.744359Z","shell.execute_reply.started":"2025-05-24T23:57:24.738253Z","shell.execute_reply":"2025-05-24T23:57:24.743730Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(os.listdir('/kaggle/working/base_dir/train_dir'))\nprint(os.listdir('/kaggle/working/base_dir/val_dir'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T23:57:38.933826Z","iopub.execute_input":"2025-05-24T23:57:38.934403Z","iopub.status.idle":"2025-05-24T23:57:38.938749Z","shell.execute_reply.started":"2025-05-24T23:57:38.934373Z","shell.execute_reply":"2025-05-24T23:57:38.938024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.set_index('id', inplace=True)\n\ntrain_list = list(df_train['id'])\nval_list = list(df_val['id'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T23:57:49.121697Z","iopub.execute_input":"2025-05-24T23:57:49.122420Z","iopub.status.idle":"2025-05-24T23:57:49.161149Z","shell.execute_reply.started":"2025-05-24T23:57:49.122393Z","shell.execute_reply":"2025-05-24T23:57:49.160424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dir = '/kaggle/input/histopathologic-cancer-detection'\n\nfor image in train_list:\n    file_name = image + '.tif'\n    target = df.loc[image, 'label']\n\n    label = '0' if target == 0 else '1'\n\n    src = os.path.join(data_dir, 'train', file_name)  # Kaggle input path\n    dest = os.path.join(train_dir, label, file_name)  # Kaggle working path\n\n    shutil.copyfile(src, dest)\n\nfor image in val_list:\n    file_name = image + '.tif'\n    target = df.loc[image, 'label']\n\n    label = '0' if target == 0 else '1'\n\n    src = os.path.join(data_dir, 'train', file_name)\n    dest = os.path.join(val_dir, label, file_name)\n\n    shutil.copyfile(src, dest)\n\nprint(len(os.listdir(os.path.join(train_dir, '0'))))\nprint(len(os.listdir(os.path.join(train_dir, '1'))))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T23:58:05.285431Z","iopub.execute_input":"2025-05-24T23:58:05.286031Z","iopub.status.idle":"2025-05-25T00:12:34.779917Z","shell.execute_reply.started":"2025-05-24T23:58:05.286003Z","shell.execute_reply":"2025-05-25T00:12:34.779090Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport numpy as np\n\nIMAGE_SIZE = 96\n\ntrain_path = '/kaggle/working/base_dir/train_dir'\nvalid_path = '/kaggle/working/base_dir/val_dir'\ntest_path = '/kaggle/input/histopathologic-cancer-detection/test'\n\nnum_train_samples = len(df_train)\nnum_val_samples = len(df_val)\ntrain_batch_size = 32\nval_batch_size = 32\n\ntrain_steps = np.ceil(num_train_samples / train_batch_size)\nval_steps = np.ceil(num_val_samples / val_batch_size)\n\ndatagen = ImageDataGenerator(rescale=1.0/255)\n\ntrain_gen = datagen.flow_from_directory(\n    train_path,\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n    batch_size=train_batch_size,\n    class_mode='categorical'\n)\n\nval_gen = datagen.flow_from_directory(\n    valid_path,\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n    batch_size=val_batch_size,\n    class_mode='categorical'\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T00:12:40.710244Z","iopub.execute_input":"2025-05-25T00:12:40.710529Z","iopub.status.idle":"2025-05-25T00:12:54.983775Z","shell.execute_reply.started":"2025-05-25T00:12:40.710508Z","shell.execute_reply":"2025-05-25T00:12:54.983255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, Dropout, MaxPooling2D, Flatten, Dense\nfrom tensorflow.keras.layers import BatchNormalization, SeparableConv2D, Activation","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T00:13:08.394847Z","iopub.execute_input":"2025-05-25T00:13:08.395698Z","iopub.status.idle":"2025-05-25T00:13:08.402322Z","shell.execute_reply.started":"2025-05-25T00:13:08.395673Z","shell.execute_reply":"2025-05-25T00:13:08.401498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Net:\n    @staticmethod\n    def build(width, height, depth, classes):\n            \n            #initializa model\n            model = Sequential()\n            \n            inputShape = (height, width, depth)\n            \n            #Add First Layer CONV => ReLU => Pooling\n            model.add(Conv2D(filters = 32, kernel_size = (5,5), padding=\"same\", activation='relu', input_shape= inputShape))\n            model.add(Conv2D(filters = 32, kernel_size = (3,3), padding=\"same\", activation='relu'))\n            model.add(Conv2D(filters = 32, kernel_size = (3,3), padding=\"same\", activation='relu'))\n            model.add(MaxPooling2D(pool_size=(2, 2)))\n            model.add(Dropout(0.2))\n            \n            #Add Second Layer CONV => ReLU => Pooling\n            model.add(Conv2D(filters = 64, kernel_size = (3,3), padding=\"same\", activation='relu'))\n            model.add(Conv2D(filters = 64, kernel_size = (3,3), padding=\"same\", activation='relu'))\n            model.add(Conv2D(filters = 64, kernel_size = (3,3), padding=\"same\", activation='relu'))\n            model.add(MaxPooling2D(pool_size=(2, 2)))\n            model.add(Dropout(0.2))\n            \n            #Add Third Layer CONV => ReLU => Pooling\n            model.add(Conv2D(filters = 128, kernel_size = (3,3), padding=\"same\", activation='relu'))\n            model.add(Conv2D(filters = 128, kernel_size = (3,3), padding=\"same\", activation='relu'))\n            model.add(Conv2D(filters = 128, kernel_size = (3,3), padding=\"same\", activation='relu'))\n            model.add(MaxPooling2D(pool_size=(2, 2)))\n            model.add(Dropout(0.25))\n            \n            #FC => ReLU\n            model.add(Flatten())\n            model.add(Dense(units = 500, activation = 'relu'))\n            model.add(Dropout(0.2))\n            #FC => Output\n            model.add(Dense(classes, activation='softmax'))\n            \n            model.summary()\n            \n            return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T00:13:12.365021Z","iopub.execute_input":"2025-05-25T00:13:12.365792Z","iopub.status.idle":"2025-05-25T00:13:12.373722Z","shell.execute_reply.started":"2025-05-25T00:13:12.365761Z","shell.execute_reply":"2025-05-25T00:13:12.373041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = Net.build(width = 96, height = 96, depth = 3, classes = 2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T00:13:22.492955Z","iopub.execute_input":"2025-05-25T00:13:22.493279Z","iopub.status.idle":"2025-05-25T00:13:25.094676Z","shell.execute_reply.started":"2025-05-25T00:13:22.493258Z","shell.execute_reply":"2025-05-25T00:13:25.094130Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.optimizers import Adam\n\nmodel.compile(optimizer=Adam(learning_rate=0.0001), \n              loss='categorical_crossentropy', \n              metrics=['accuracy'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T00:13:29.016943Z","iopub.execute_input":"2025-05-25T00:13:29.017546Z","iopub.status.idle":"2025-05-25T00:13:29.033330Z","shell.execute_reply.started":"2025-05-25T00:13:29.017523Z","shell.execute_reply":"2025-05-25T00:13:29.032677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau\n\nfilepath = \"checkpoint.h5\"\ncheckpoint = ModelCheckpoint(filepath, monitor='val_accuracy', verbose=1, \n                             save_best_only=True, mode='max')\n\nreduce_lr = ReduceLROnPlateau(monitor='val_accuracy', factor=0.5, patience=2, \n                              verbose=1, mode='max', min_lr=1e-5)\n\ncallbacks_list = [checkpoint, reduce_lr]\n\ntrain_steps = int(np.ceil(num_train_samples / train_batch_size))\nval_steps = int(np.ceil(num_val_samples / val_batch_size))\n\nhistory = model.fit(\n    train_gen,\n    steps_per_epoch=train_steps,\n    validation_data=val_gen,\n    validation_steps=val_steps,\n    epochs=11,\n    verbose=1,\n    callbacks=callbacks_list\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T00:13:41.670105Z","iopub.execute_input":"2025-05-25T00:13:41.670393Z","iopub.status.idle":"2025-05-25T00:34:43.714109Z","shell.execute_reply.started":"2025-05-25T00:13:41.670364Z","shell.execute_reply":"2025-05-25T00:34:43.713512Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot training & validation accuracy values\n\nplt.plot(history.history['accuracy'])       # instead of 'acc'\nplt.plot(history.history['val_accuracy'])   # instead of 'val_acc'\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Validation'], loc='best')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T00:34:51.510180Z","iopub.execute_input":"2025-05-25T00:34:51.510870Z","iopub.status.idle":"2025-05-25T00:34:51.679148Z","shell.execute_reply.started":"2025-05-25T00:34:51.510844Z","shell.execute_reply":"2025-05-25T00:34:51.678365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot training & validation loss values\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('Model loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Test'], loc='best')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T00:34:59.902119Z","iopub.execute_input":"2025-05-25T00:34:59.902408Z","iopub.status.idle":"2025-05-25T00:35:00.062882Z","shell.execute_reply.started":"2025-05-25T00:34:59.902382Z","shell.execute_reply":"2025-05-25T00:35:00.062182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load best model weights\nmodel.load_weights('checkpoint.h5')\n\n# Evaluate the model on the validation set\nval_loss, val_acc = model.evaluate(val_gen, steps=val_steps)\nprint('val_loss:', val_loss)\nprint('val_acc:', val_acc)\n\n# Get predictions (probabilities for each class)\npredictions = model.predict(val_gen, steps=val_steps, verbose=1)\n\n# Convert predictions to DataFrame\ndf_preds = pd.DataFrame(predictions, columns=['no_tumor', 'has_tumor'])\n\n# Get true labels from generator\ny_true = val_gen.classes\n\n# Predicted probabilities for the positive class (has_tumor, i.e., class 1)\ny_pred = df_preds['has_tumor'].values\n\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T00:47:27.898889Z","iopub.execute_input":"2025-05-25T00:47:27.899631Z","iopub.status.idle":"2025-05-25T00:47:48.064372Z","shell.execute_reply.started":"2025-05-25T00:47:27.899605Z","shell.execute_reply":"2025-05-25T00:47:48.063783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ROC AUC Score\nfrom sklearn.metrics import roc_auc_score, roc_curve, auc\nprint('ROC AUC Score =', roc_auc_score(y_true, y_pred))\n\n# ROC curve\nfpr_keras, tpr_keras, thresholds_keras = roc_curve(y_true, y_pred)\nauc_keras = auc(fpr_keras, tpr_keras)\n# Plot ROC Curve\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(8,6))\nplt.plot([0, 1], [0, 1], 'k--')\nplt.plot(fpr_keras, tpr_keras, label='AUC = {:.2f}'.format(auc_keras))\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('ROC Curve')\nplt.legend(loc='best')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T00:49:17.816973Z","iopub.execute_input":"2025-05-25T00:49:17.817250Z","iopub.status.idle":"2025-05-25T00:49:17.992900Z","shell.execute_reply.started":"2025-05-25T00:49:17.817231Z","shell.execute_reply":"2025-05-25T00:49:17.992195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(1)\nplt.plot([0, 1], [0, 1], 'k--')\nplt.plot(fpr_keras, tpr_keras, label='area = {:.2f}'.format(auc_keras))\nplt.xlabel('False positive rate')\nplt.ylabel('True positive rate')\nplt.title('ROC curve')\nplt.legend(loc='best')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T00:50:32.767004Z","iopub.execute_input":"2025-05-25T00:50:32.767698Z","iopub.status.idle":"2025-05-25T00:50:32.923548Z","shell.execute_reply.started":"2025-05-25T00:50:32.767676Z","shell.execute_reply":"2025-05-25T00:50:32.922887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n\ny_pred_binary = predictions.argmax(axis=1)\ncm = confusion_matrix(y_true, y_pred_binary)\n\nfrom mlxtend.plotting import plot_confusion_matrix\nfig, ax = plot_confusion_matrix(conf_mat=cm,\n                                show_absolute=True,\n                                show_normed=True,\n                                colorbar=True,\n                               cmap = 'Dark2')\nplt.show()\n\nfrom sklearn.metrics import classification_report\n# Generate a classification report\n\nreport = classification_report(y_true, y_pred_binary, target_names = ['no_tumor', 'has_tumor'])\nprint(report)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T01:01:49.343859Z","iopub.execute_input":"2025-05-25T01:01:49.344280Z","iopub.status.idle":"2025-05-25T01:01:49.694695Z","shell.execute_reply.started":"2025-05-25T01:01:49.344250Z","shell.execute_reply":"2025-05-25T01:01:49.694044Z"}},"outputs":[],"execution_count":null}]}