{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":1353811,"sourceType":"datasetVersion","datasetId":762203},{"sourceId":11684072,"sourceType":"datasetVersion","datasetId":7333342},{"sourceId":36238573,"sourceType":"kernelVersion"}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import NASNetMobile\nfrom tensorflow import keras\nfrom tensorflow.keras import layers , models\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Flatten, Dropout, Activation , MaxPooling2D, BatchNormalization\nfrom sklearn.model_selection import train_test_split\nfrom PIL import Image\nimport os\nimport numpy as np\nimport shutil\nimport pandas as pd\nimport random\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:30.939414Z","iopub.execute_input":"2025-05-05T22:03:30.939784Z","iopub.status.idle":"2025-05-05T22:03:46.920203Z","shell.execute_reply.started":"2025-05-05T22:03:30.939727Z","shell.execute_reply":"2025-05-05T22:03:46.919334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load first CSV (target is already 0/1)\ndf1 = pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/train.csv\")  # has columns 'image_name', 'target'\ndf1['filepath'] = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/train/\" + df1['image_name'] + \".jpg\"\n\n# Upsample malignant (target=1) 7x\ndf1_pos = df1[df1['target'] == 1]\ndf1 = pd.concat([df1, pd.concat([df1_pos] * 7, ignore_index=True)], ignore_index=True)\n\n# Load second CSV (with diagnosis)\ndf2_raw = pd.read_csv(\"/kaggle/input/isic-2019-training-groundtruth/ISIC_2019_Training_GroundTruth (2).csv\")  # has columns 'image_name', 'diagnosis'\n\n\n# Convert one-hot to single label\ndiagnosis_columns = ['MEL','NV','BCC','AK','BKL','DF','VASC','SCC','UNK']\ndf2_raw['diagnosis'] = df2_raw[diagnosis_columns].idxmax(axis=1)\n\n# Map diagnosis to binary target\nbenign_labels = ['NV', 'BKL', 'DF', 'VASC', 'UNK']\ndf2_raw['target'] = df2_raw['diagnosis'].apply(lambda x: 0 if x in benign_labels else 1)\n\n# Final columns + filepath\ndf2_raw['filepath'] = \"/kaggle/input/jpeg-isic2019-512x512/train/\" + df2_raw['image'] + \".jpg\"\ndf2 = df2_raw[['image', 'target', 'filepath']].rename(columns={'image': 'image_name'})\n\n\n# Combine datasets\ndf_all = pd.concat([df1[['image_name', 'target', 'filepath']], df2], ignore_index=True)\ndf_all = df_all.sample(frac=1, random_state=42).reset_index(drop=True)  # shuffle\n\n# Split into train, val, test (70/15/15)\ntrain_df, temp_df = train_test_split(df_all, test_size=0.3, stratify=df_all['target'], random_state=42)\nval_df, test_df = train_test_split(temp_df, test_size=0.5, stratify=temp_df['target'], random_state=42)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:46.921438Z","iopub.execute_input":"2025-05-05T22:03:46.921941Z","iopub.status.idle":"2025-05-05T22:03:47.167135Z","shell.execute_reply.started":"2025-05-05T22:03:46.921920Z","shell.execute_reply":"2025-05-05T22:03:47.166261Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_df.shape)\nprint(val_df.shape)\nprint(test_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:47.168018Z","iopub.execute_input":"2025-05-05T22:03:47.168285Z","iopub.status.idle":"2025-05-05T22:03:47.173216Z","shell.execute_reply.started":"2025-05-05T22:03:47.168262Z","shell.execute_reply":"2025-05-05T22:03:47.172401Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_all.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:47.174852Z","iopub.execute_input":"2025-05-05T22:03:47.175121Z","iopub.status.idle":"2025-05-05T22:03:47.203700Z","shell.execute_reply.started":"2025-05-05T22:03:47.175104Z","shell.execute_reply":"2025-05-05T22:03:47.202971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nfilepath = df_all.iloc[14][\"filepath\"]  # Or 'filepath' if the column name is that\nif os.path.exists(filepath):\n    print(\"File exists:\", filepath)\nelse:\n    print(\"File does NOT exist:\", filepath)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:47.204478Z","iopub.execute_input":"2025-05-05T22:03:47.204691Z","iopub.status.idle":"2025-05-05T22:03:47.211236Z","shell.execute_reply.started":"2025-05-05T22:03:47.204675Z","shell.execute_reply":"2025-05-05T22:03:47.210488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Parameters\nIMAGE_SIZE = (224,224)\nBATCH_SIZE = 8\nAUTOTUNE = tf.data.AUTOTUNE\nAUG_COPIES = 7  # Number of augmented copies for target==1\n\n# ------------------------------------------\n# Step 1: Duplicate class 1 rows in the DataFrame\ndf_pos = train_df[train_df['target'] == 1]\ndf_pos_aug = pd.concat([df_pos] * AUG_COPIES, ignore_index=True)\ntrain_df_augmented = pd.concat([train_df, df_pos_aug], ignore_index=True)\ntrain_df_augmented = train_df_augmented.sample(frac=1, random_state=42).reset_index(drop=True)\n\n# ------------------------------------------\n# Step 2: Shades of Gray implementation\n# Shades of Gray implementation using TensorFlow operations\ndef shades_of_gray_tf(img, power=6, gamma=None):\n    img = tf.cast(img, tf.float32) # Use tf.cast instead of .astype\n    if gamma is not None:\n        img = tf.pow(img, (1.0 / gamma)) # Use tf.pow\n    img_power = tf.pow(img, power) # Use tf.pow\n    # Use tf.reduce_mean for mean calculation\n    mean_per_channel = tf.pow(tf.reduce_mean(img_power, axis=[0, 1]), 1 / power)\n    # Use tf.square and tf.reduce_sum for norm calculation\n    norm = tf.sqrt(tf.reduce_sum(tf.square(mean_per_channel)))\n    scaling_factors = norm / mean_per_channel\n    img = img * scaling_factors\n    img = tf.clip_by_value(img, 0, 255) # Use tf.clip_by_value\n    return tf.cast(img, tf.uint8) # Use tf.cast\n\n\n\n# TensorFlow wrapper\ndef tf_shades_of_gray(image):\n    image = tf.py_function(func=shades_of_gray, inp=[image], Tout=tf.uint8)\n    image.set_shape([None, None, 3])\n    return image\n\n# ------------------------------------------\n# Step 3: Preprocessing\ndef preprocess_image(filepath, label):\n    image = tf.io.read_file(filepath)\n    image = tf.image.decode_jpeg(image, channels=3)\n    # Apply Shades of Gray here using the TensorFlow version\n    #image = shades_of_gray_tf(image)\n    image = tf.image.resize(image, IMAGE_SIZE)\n    image = tf.cast(image, tf.float32) / 255.0  # Normalize\n    return image, label\n\n\n# ------------------------------------------\n# Step 4: Augmentation (only for target == 1)\ndef augment(image, label):\n    image = tf.image.random_flip_left_right(image)\n\n    def augment_if_positive():\n        img = tf.image.random_brightness(image, max_delta=0.2)\n        img = tf.image.random_contrast(img, lower=0.8, upper=1.2)\n        img = tf.image.random_saturation(img, lower=0.8, upper=1.2)\n        img = tf.image.random_hue(img, max_delta=0.05)\n        return img\n\n    def no_extra_augment():\n        return image\n\n    image = tf.cond(tf.equal(label, 1),\n                    true_fn=augment_if_positive,\n                    false_fn=no_extra_augment)\n\n    return image, label\n\n# ------------------------------------------\n# Step 5: Dataset builder\ndef df_to_dataset(df, augment_data=False):\n    filepaths = df['filepath'].values\n    labels = df['target'].values\n    ds = tf.data.Dataset.from_tensor_slices((filepaths, labels))\n    ds = ds.map(preprocess_image, num_parallel_calls=AUTOTUNE)\n    if augment_data:\n        ds = ds.map(augment, num_parallel_calls=AUTOTUNE)\n        ds = ds.shuffle(1024)\n    ds = ds.batch(BATCH_SIZE).prefetch(AUTOTUNE)\n    return ds\n\n# ------------------------------------------\n# Step 6: Final datasets\ntrain_ds = df_to_dataset(train_df_augmented, augment_data=True)\nval_ds   = df_to_dataset(val_df)\ntest_ds  = df_to_dataset(test_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:47.212229Z","iopub.execute_input":"2025-05-05T22:03:47.212813Z","iopub.status.idle":"2025-05-05T22:03:49.252729Z","shell.execute_reply.started":"2025-05-05T22:03:47.212786Z","shell.execute_reply":"2025-05-05T22:03:49.252168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_samples = len(train_df)\nnum_batches = (num_samples + BATCH_SIZE - 1) // BATCH_SIZE\nprint(\"Total samples:\", num_samples)\nprint(\"Approximate batches:\", num_batches)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:49.253684Z","iopub.execute_input":"2025-05-05T22:03:49.253869Z","iopub.status.idle":"2025-05-05T22:03:49.257927Z","shell.execute_reply.started":"2025-05-05T22:03:49.253854Z","shell.execute_reply":"2025-05-05T22:03:49.257091Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetB4\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom sklearn.metrics import classification_report, confusion_matrix, roc_auc_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:49.258772Z","iopub.execute_input":"2025-05-05T22:03:49.259009Z","iopub.status.idle":"2025-05-05T22:03:49.275147Z","shell.execute_reply.started":"2025-05-05T22:03:49.258983Z","shell.execute_reply":"2025-05-05T22:03:49.274572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, models, Model\n\n# Recommended input size for EfficientNetB6 is typically 528x528\n# Ensure IMAGE_SIZE is defined and set accordingly, e.g., IMAGE_SIZE = (528, 528)\nBATCH_SIZE = 16\nNUM_CLASSES = 1 # Binary classification\n\n\nbase_model = tf.keras.applications.EfficientNetB0(\n    weights='imagenet',\n    include_top=False,\n    input_shape=(IMAGE_SIZE[0], IMAGE_SIZE[1], 3)\n)\nbase_model.trainable = False # Freeze base initially","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:49.275905Z","iopub.execute_input":"2025-05-05T22:03:49.276160Z","iopub.status.idle":"2025-05-05T22:03:51.984483Z","shell.execute_reply.started":"2025-05-05T22:03:49.276144Z","shell.execute_reply":"2025-05-05T22:03:51.983958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.optimizers.schedules import CosineDecayRestarts\nlr_schedule = CosineDecayRestarts(\n    initial_learning_rate=1e-3,\n    first_decay_steps=1000,\n    t_mul=2.0,\n    m_mul=1.0,\n    alpha=1e-5\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:51.986343Z","iopub.execute_input":"2025-05-05T22:03:51.986553Z","iopub.status.idle":"2025-05-05T22:03:51.995105Z","shell.execute_reply.started":"2025-05-05T22:03:51.986537Z","shell.execute_reply":"2025-05-05T22:03:51.994424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming base_model is loaded and frozen as shown above\n\n# --- Build Model using Functional API ---\nx = base_model.output\nx = layers.GlobalAveragePooling2D()(x)\nx = layers.Dropout(0.4)(x) # Add dropout layer as in your sequential model\noutputs = layers.Dense(NUM_CLASSES, activation='sigmoid')(x) # Use NUM_CLASSES variable\n\nmodel = Model(inputs=base_model.input, outputs=outputs)\n#model.summary()\n\nmetrics=[\n    'accuracy',\n    tf.keras.metrics.Precision(name='precision'),\n    tf.keras.metrics.Recall(name='recall'),\n    tf.keras.metrics.AUC(name='auc')\n]\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=lr_schedule),\n    loss='binary_crossentropy',\n    metrics=metrics\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:51.995881Z","iopub.execute_input":"2025-05-05T22:03:51.996117Z","iopub.status.idle":"2025-05-05T22:03:52.050000Z","shell.execute_reply.started":"2025-05-05T22:03:51.996097Z","shell.execute_reply":"2025-05-05T22:03:52.049267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.utils import class_weight\ny_original = df_all['target']\nclasses = np.unique(y_original)\nclass_weights = class_weight.compute_class_weight('balanced',\n                                                  classes=classes,\n                                                  y=y_original)\n\nclass_weight_dict = dict(zip(classes, class_weights))\n\nprint(\"Calculated Class Weights (based on original data):\", class_weight_dict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:52.050759Z","iopub.execute_input":"2025-05-05T22:03:52.050991Z","iopub.status.idle":"2025-05-05T22:03:52.069451Z","shell.execute_reply.started":"2025-05-05T22:03:52.050968Z","shell.execute_reply":"2025-05-05T22:03:52.068940Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Callbacks: early stopping + model checkpoint\ncallbacks = [\n    tf.keras.callbacks.EarlyStopping(monitor='val_auc', patience=5, mode='max', restore_best_weights=True),\n    tf.keras.callbacks.ModelCheckpoint('best_model.keras', monitor='val_auc', save_best_only=True, mode='max')\n]\n\n# Training\nmodel.fit(\n    train_ds,\n    validation_data=val_ds,\n    epochs=5,\n    callbacks=callbacks,\n    class_weight=None  # imbalance handling\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T22:03:52.070158Z","iopub.execute_input":"2025-05-05T22:03:52.070399Z","execution_failed":"2025-05-06T00:02:13.132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save(\"/kaggle/working/final_model.keras\")","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-06T00:02:13.133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.metrics import roc_curve, auc, confusion_matrix, classification_report, f1_score\nimport matplotlib.pyplot as plt\n\n# Get true labels and predictions\ny_true = []\ny_pred_probs = []\n\nfor images, labels in train_ds:\n    preds = model.predict(images).ravel()\n    y_pred_probs.extend(preds)\n    y_true.extend(labels.numpy())\n\ny_true = np.array(y_true)\ny_pred_probs = np.array(y_pred_probs)\n\n# Convert predicted probabilities to class labels (threshold = 0.5)\ny_pred = (y_pred_probs >= 0.5).astype(int)\nfpr, tpr, _ = roc_curve(y_true, y_pred_probs)\nroc_auc = auc(fpr, tpr)\n\nplt.figure(figsize=(6, 6))\nplt.plot(fpr, tpr, color='darkorange', lw=2, label='AUC = %0.3f' % roc_auc)\nplt.plot([0, 1], [0, 1], color='navy', lw=1, linestyle='--')\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('ROC Curve on Validation Data')\nplt.legend(loc='lower right')\nplt.grid()\nplt.show()\n# Confusion matrix\ncm = confusion_matrix(y_true, y_pred)\nprint(\"Confusion Matrix:\\n\", cm)\n\n# Classification report includes precision, recall, f1\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_true, y_pred, digits=4))\n\n# F1 score separately if needed\nf1 = f1_score(y_true, y_pred)\nprint(\"\\nF1 Score:\", f1)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-06T00:02:13.133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Evaluation ---\ny_true = []\ny_pred = []\n\nfor images, labels in test_ds:\n    preds = model.predict(images)\n    y_pred.extend(preds.flatten())\n    y_true.extend(labels.numpy().flatten())\n\ny_pred_binary = np.array(y_pred) > 0.5\n\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_true, y_pred_binary))\nprint(\"\\nConfusion Matrix:\")\nprint(confusion_matrix(y_true, y_pred_binary))\nprint(\"\\nAUC:\", roc_auc_score(y_true, y_pred))","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-06T00:02:13.133Z"},"_kg_hide-input":false},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import (confusion_matrix, classification_report, \n                            roc_curve, auc, precision_recall_curve, \n                            average_precision_score)\n\ndef plot_confusion_matrix(y_true, y_pred, classes, \n                          normalize=False, title=None, cmap=plt.cm.Blues):\n    \"\"\"\n    This function prints and plots the confusion matrix.\n    Normalization can be applied by setting `normalize=True`.\n    \"\"\"\n    if not title:\n        if normalize:\n            title = 'Normalized confusion matrix'\n        else:\n            title = 'Confusion matrix, without normalization'\n\n    # Compute confusion matrix\n    cm = confusion_matrix(y_true, y_pred)\n    \n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n        fmt = '.2f'\n    else:\n        fmt = 'd'\n\n    fig, ax = plt.subplots(figsize=(8, 6))\n    im = ax.imshow(cm, interpolation='nearest', cmap=cmap)\n    ax.figure.colorbar(im, ax=ax)\n    \n    # Show all ticks and label them\n    ax.set(xticks=np.arange(cm.shape[1]),\n           yticks=np.arange(cm.shape[0]),\n           xticklabels=classes, yticklabels=classes,\n           title=title,\n           ylabel='True label',\n           xlabel='Predicted label')\n\n    # Rotate the tick labels and set their alignment\n    plt.setp(ax.get_xticklabels(), rotation=45, ha=\"right\",\n             rotation_mode=\"anchor\")\n\n    # Loop over data dimensions and create text annotations\n    thresh = cm.max() / 2.\n    for i in range(cm.shape[0]):\n        for j in range(cm.shape[1]):\n            ax.text(j, i, format(cm[i, j], fmt),\n                    ha=\"center\", va=\"center\",\n                    color=\"white\" if cm[i, j] > thresh else \"black\")\n    \n    fig.tight_layout()\n    return ax\n\ndef plot_metrics(y_true, y_pred_probs, y_pred_binary):\n    \"\"\"Plot ROC curve and Precision-Recall curve\"\"\"\n    # Calculate metrics\n    report = classification_report(y_true, y_pred_binary, output_dict=True)\n    fpr, tpr, _ = roc_curve(y_true, y_pred_probs)\n    roc_auc = auc(fpr, tpr)\n    precision, recall, _ = precision_recall_curve(y_true, y_pred_probs)\n    avg_precision = average_precision_score(y_true, y_pred_probs)\n    \n    # Create figure\n    fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(16, 6))\n    \n    # ROC Curve\n    ax1.plot(fpr, tpr, color='darkorange', lw=2, \n             label=f'ROC curve (AUC = {roc_auc:.2f})')\n    ax1.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\n    ax1.set_xlim([0.0, 1.0])\n    ax1.set_ylim([0.0, 1.05])\n    ax1.set_xlabel('False Positive Rate')\n    ax1.set_ylabel('True Positive Rate')\n    ax1.set_title('Receiver Operating Characteristic')\n    ax1.legend(loc=\"lower right\")\n    \n    # Precision-Recall Curve\n    ax2.plot(recall, precision, color='blue', lw=2, \n             label=f'Precision-Recall (AP = {avg_precision:.2f})')\n    ax2.set_xlim([0.0, 1.0])\n    ax2.set_ylim([0.0, 1.05])\n    ax2.set_xlabel('Recall')\n    ax2.set_ylabel('Precision')\n    ax2.set_title('Precision-Recall Curve')\n    ax2.legend(loc=\"upper right\")\n    \n    plt.tight_layout()\n    \n    # Print metrics\n    print(\"\\nDetailed Classification Metrics:\")\n    print(f\"Accuracy: {report['accuracy']:.4f}\")\n    print(f\"Precision (Class 1): {report['1']['precision']:.4f}\")\n    print(f\"Recall (Class 1): {report['1']['recall']:.4f}\")\n    print(f\"F1-Score (Class 1): {report['1']['f1-score']:.4f}\")\n    print(f\"AUC: {roc_auc:.4f}\")\n    \n    return fig\n\n# Example usage\n# Assuming you have:\n# y_true - true labels (0 or 1)\n# y_pred_probs - predicted probabilities (continuous between 0-1)\n# y_pred_binary - binary predictions (0 or 1)\n\n# Generate metrics and plots\nplot_confusion_matrix(y_true, y_pred_binary, classes=['Negative', 'Positive'], \n                     normalize=True, title='Normalized Confusion Matrix')\nplt.show()\n\nplot_metrics(y_true, y_pred, y_pred_binary)\nplt.show()","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-06T00:02:13.133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import (confusion_matrix, classification_report, \n                            roc_curve, auc, precision_recall_curve, \n                            average_precision_score)\n\ndef plot_confusion_matrix(y_true, y_pred, classes, title='Confusion Matrix', cmap=plt.cm.Blues):\n    \"\"\"\n    This function prints and plots the confusion matrix with actual counts.\n    \"\"\"\n    # Compute confusion matrix\n    cm = confusion_matrix(y_true, y_pred)\n    \n    fig, ax = plt.subplots(figsize=(8, 6))\n    im = ax.imshow(cm, interpolation='nearest', cmap=cmap)\n    ax.figure.colorbar(im, ax=ax)\n    \n    # Show all ticks and label them\n    ax.set(xticks=np.arange(cm.shape[1]),\n           yticks=np.arange(cm.shape[0]),\n           xticklabels=classes, yticklabels=classes,\n           title=title,\n           ylabel='True label',\n           xlabel='Predicted label')\n\n    # Rotate the tick labels and set their alignment\n    plt.setp(ax.get_xticklabels(), rotation=45, ha=\"right\",\n             rotation_mode=\"anchor\")\n\n    # Loop over data dimensions and create text annotations\n    thresh = cm.max() / 2.\n    for i in range(cm.shape[0]):\n        for j in range(cm.shape[1]):\n            ax.text(j, i, format(cm[i, j], 'd'),  # 'd' means integer format\n                    ha=\"center\", va=\"center\",\n                    color=\"white\" if cm[i, j] > thresh else \"black\")\n    \n    fig.tight_layout()\n    return ax\n\ndef plot_metrics(y_true, y_pred_probs, y_pred_binary):\n    \"\"\"Plot ROC curve and Precision-Recall curve with metrics\"\"\"\n    # Calculate metrics\n    report = classification_report(y_true, y_pred_binary, output_dict=True)\n    fpr, tpr, _ = roc_curve(y_true, y_pred_probs)\n    roc_auc = auc(fpr, tpr)\n    precision, recall, _ = precision_recall_curve(y_true, y_pred_probs)\n    avg_precision = average_precision_score(y_true, y_pred_probs)\n    \n    # Create figure\n    fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(16, 6))\n    \n    # ROC Curve\n    ax1.plot(fpr, tpr, color='darkorange', lw=2, \n             label=f'ROC curve (AUC = {roc_auc:.2f})')\n    ax1.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\n    ax1.set_xlim([0.0, 1.0])\n    ax1.set_ylim([0.0, 1.05])\n    ax1.set_xlabel('False Positive Rate')\n    ax1.set_ylabel('True Positive Rate')\n    ax1.set_title('Receiver Operating Characteristic')\n    ax1.legend(loc=\"lower right\")\n    \n    # Precision-Recall Curve\n    ax2.plot(recall, precision, color='blue', lw=2, \n             label=f'Precision-Recall (AP = {avg_precision:.2f})')\n    ax2.set_xlim([0.0, 1.0])\n    ax2.set_ylim([0.0, 1.05])\n    ax2.set_xlabel('Recall')\n    ax2.set_ylabel('Precision')\n    ax2.set_title('Precision-Recall Curve')\n    ax2.legend(loc=\"upper right\")\n    \n    plt.tight_layout()\n    \n    # Print metrics\n    print(\"\\nDetailed Classification Metrics:\")\n    print(f\"Accuracy: {report['accuracy']:.4f}\")\n    print(f\"Precision (Positive Class): {report['1']['precision']:.4f}\")\n    print(f\"Recall (Positive Class): {report['1']['recall']:.4f}\")\n    print(f\"F1-Score (Positive Class): {report['1']['f1-score']:.4f}\")\n    print(f\"AUC: {roc_auc:.4f}\")\n    \n    return fig\n\n# Example usage\n# Assuming you have:\n# y_true = true labels (0 or 1)\n# y_pred_probs = predicted probabilities (continuous between 0-1)\n# y_pred_binary = binary predictions (0 or 1, threshold=0.5)\n\n# Generate confusion matrix with actual counts\nplot_confusion_matrix(y_true, y_pred_binary, \n                     classes=['Negative', 'Positive'], \n                     title='Confusion Matrix (Actual Counts)')\nplt.show()\n\n# Generate ROC and Precision-Recall curves with metrics\nplot_metrics(y_true, y_pred, y_pred_binary)\nplt.show()","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-06T00:02:13.133Z"}},"outputs":[],"execution_count":null}]}