{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"}],"dockerImageVersionId":30299,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    print(os.path.join(dirname))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:38:02.223716Z","iopub.execute_input":"2025-11-08T06:38:02.224177Z","iopub.status.idle":"2025-11-08T06:42:11.743887Z","shell.execute_reply.started":"2025-11-08T06:38:02.224067Z","shell.execute_reply":"2025-11-08T06:42:11.742911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import re\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly\nimport plotly.graph_objects as go\nimport cv2\nimport tensorflow as tf\nfrom kaggle_datasets import KaggleDatasets\nfrom functools import partial\nimport sklearn\nfrom tqdm import tqdm_notebook as tqdm\nimport gc\n%matplotlib inline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:11.745581Z","iopub.execute_input":"2025-11-08T06:42:11.745895Z","iopub.status.idle":"2025-11-08T06:42:22.233214Z","shell.execute_reply.started":"2025-11-08T06:42:11.745868Z","shell.execute_reply":"2025-11-08T06:42:22.232268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept:\n    strategy = tf.distribute.get_strategy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:22.234367Z","iopub.execute_input":"2025-11-08T06:42:22.234972Z","iopub.status.idle":"2025-11-08T06:42:22.250442Z","shell.execute_reply.started":"2025-11-08T06:42:22.234935Z","shell.execute_reply":"2025-11-08T06:42:22.249681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Number of replicas:', strategy.num_replicas_in_sync)\nprint(\"Version of Tensorflow used : \", tf.__version__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:22.251361Z","iopub.execute_input":"2025-11-08T06:42:22.251573Z","iopub.status.idle":"2025-11-08T06:42:22.256047Z","shell.execute_reply.started":"2025-11-08T06:42:22.251553Z","shell.execute_reply":"2025-11-08T06:42:22.255090Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"AUTOTUNE = tf.data.experimental.AUTOTUNE\nGCS_PATH = \"/kaggle/input/siim-isic-melanoma-classification\"\n# BATCH_SIZE = 16 * strategy.num_replicas_in_sync\n# IMAGE_SIZE = [1024, 1024]\n# SHAPE = [256, 256] \nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nIMAGE_SIZE = [1024, 1024]\nSHAPE = [384, 384] ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:22.257249Z","iopub.execute_input":"2025-11-08T06:42:22.257508Z","iopub.status.idle":"2025-11-08T06:42:22.268444Z","shell.execute_reply.started":"2025-11-08T06:42:22.257484Z","shell.execute_reply":"2025-11-08T06:42:22.267590Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Batch Size = \", BATCH_SIZE)\nprint(\"GCS Path = \", GCS_PATH)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:22.269381Z","iopub.execute_input":"2025-11-08T06:42:22.269646Z","iopub.status.idle":"2025-11-08T06:42:22.279315Z","shell.execute_reply.started":"2025-11-08T06:42:22.269610Z","shell.execute_reply":"2025-11-08T06:42:22.278287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.DataFrame(pd.read_csv(\"../input/siim-isic-melanoma-classification/train.csv\"))\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:22.283084Z","iopub.execute_input":"2025-11-08T06:42:22.283402Z","iopub.status.idle":"2025-11-08T06:42:22.425390Z","shell.execute_reply.started":"2025-11-08T06:42:22.283355Z","shell.execute_reply":"2025-11-08T06:42:22.424415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pd.DataFrame(pd.read_csv(\"../input/siim-isic-melanoma-classification/test.csv\"))\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:22.426368Z","iopub.execute_input":"2025-11-08T06:42:22.426596Z","iopub.status.idle":"2025-11-08T06:42:22.464813Z","shell.execute_reply.started":"2025-11-08T06:42:22.426576Z","shell.execute_reply":"2025-11-08T06:42:22.464071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:22.465974Z","iopub.execute_input":"2025-11-08T06:42:22.466305Z","iopub.status.idle":"2025-11-08T06:42:22.513101Z","shell.execute_reply.started":"2025-11-08T06:42:22.466275Z","shell.execute_reply":"2025-11-08T06:42:22.512356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:22.514071Z","iopub.execute_input":"2025-11-08T06:42:22.514418Z","iopub.status.idle":"2025-11-08T06:42:22.527146Z","shell.execute_reply.started":"2025-11-08T06:42:22.514384Z","shell.execute_reply":"2025-11-08T06:42:22.526207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dir = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/train/\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:22.528414Z","iopub.execute_input":"2025-11-08T06:42:22.529387Z","iopub.status.idle":"2025-11-08T06:42:22.538898Z","shell.execute_reply.started":"2025-11-08T06:42:22.529347Z","shell.execute_reply":"2025-11-08T06:42:22.537842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_names = train[\"image_name\"].values + \".jpg\"\nrandom_images = [np.random.choice(image_names) for i in range(4)] # Generates a random sample from a given 1-D array\nrandom_images ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:22.539921Z","iopub.execute_input":"2025-11-08T06:42:22.540273Z","iopub.status.idle":"2025-11-08T06:42:22.555112Z","shell.execute_reply.started":"2025-11-08T06:42:22.540238Z","shell.execute_reply":"2025-11-08T06:42:22.554084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_images = []","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:22.556350Z","iopub.execute_input":"2025-11-08T06:42:22.556622Z","iopub.status.idle":"2025-11-08T06:42:22.565872Z","shell.execute_reply.started":"2025-11-08T06:42:22.556598Z","shell.execute_reply":"2025-11-08T06:42:22.565182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nfor i in range(4) : \n    plt.subplot(2, 2, i + 1) \n    image = cv2.imread(os.path.join(train_dir, random_images[i]))\n    # cv2 reads images in BGR format. Hence we convert it to RGB\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    sample_images.append(image)\n    plt.imshow(image, cmap = \"gray\")\n    plt.grid(True)\n# Automatically adjust subplot parameters to give specified padding.\nplt.tight_layout()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:22.566786Z","iopub.execute_input":"2025-11-08T06:42:22.567034Z","iopub.status.idle":"2025-11-08T06:42:34.050099Z","shell.execute_reply.started":"2025-11-08T06:42:22.567011Z","shell.execute_reply":"2025-11-08T06:42:34.049007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split \ntraining_files, validation_files = train_test_split(tf.io.gfile.glob(GCS_PATH + \"/tfrecords/train*.tfrec\"),\n                                                   test_size = 0.1, random_state = 42)\n\ntesting_files = tf.io.gfile.glob(GCS_PATH + \"/tfrecords/test*.tfrec\")\n\nprint(\"Number of training files = \", len(training_files))\nprint(\"Number of validation files = \", len(validation_files))\nprint(\"Number of test files = \", len(testing_files))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:34.051279Z","iopub.execute_input":"2025-11-08T06:42:34.051542Z","iopub.status.idle":"2025-11-08T06:42:34.245872Z","shell.execute_reply.started":"2025-11-08T06:42:34.051518Z","shell.execute_reply":"2025-11-08T06:42:34.245020Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def decode_image(image) : \n    image = tf.image.decode_jpeg(image, channels = 3)\n    image = tf.cast(image, tf.float32)\n    image = image / 255.0\n    image = tf.reshape(image, [IMAGE_SIZE[0], IMAGE_SIZE[1], 3])\n    return image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:34.246934Z","iopub.execute_input":"2025-11-08T06:42:34.247233Z","iopub.status.idle":"2025-11-08T06:42:34.252221Z","shell.execute_reply.started":"2025-11-08T06:42:34.247207Z","shell.execute_reply":"2025-11-08T06:42:34.251317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_images[0].shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:34.253318Z","iopub.execute_input":"2025-11-08T06:42:34.254157Z","iopub.status.idle":"2025-11-08T06:42:34.265914Z","shell.execute_reply.started":"2025-11-08T06:42:34.254100Z","shell.execute_reply":"2025-11-08T06:42:34.264803Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_files","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:34.266970Z","iopub.execute_input":"2025-11-08T06:42:34.267260Z","iopub.status.idle":"2025-11-08T06:42:34.280728Z","shell.execute_reply.started":"2025-11-08T06:42:34.267237Z","shell.execute_reply":"2025-11-08T06:42:34.279980Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_picked = training_files[0]\nsample_picked","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:34.281960Z","iopub.execute_input":"2025-11-08T06:42:34.282247Z","iopub.status.idle":"2025-11-08T06:42:34.291587Z","shell.execute_reply.started":"2025-11-08T06:42:34.282224Z","shell.execute_reply":"2025-11-08T06:42:34.290666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file = tf.data.TFRecordDataset(sample_picked)\nfile","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:34.292723Z","iopub.execute_input":"2025-11-08T06:42:34.292968Z","iopub.status.idle":"2025-11-08T06:42:39.929689Z","shell.execute_reply.started":"2025-11-08T06:42:34.292946Z","shell.execute_reply":"2025-11-08T06:42:39.928694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_description = {\"image\" : tf.io.FixedLenFeature([], tf.string), \n                      \"target\" : tf.io.FixedLenFeature([], tf.int64)}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:39.930726Z","iopub.execute_input":"2025-11-08T06:42:39.931003Z","iopub.status.idle":"2025-11-08T06:42:39.935366Z","shell.execute_reply.started":"2025-11-08T06:42:39.930977Z","shell.execute_reply":"2025-11-08T06:42:39.934446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def parse_function(example) : \n    # The example supplied is parsed based on the feature_description above.\n    return tf.io.parse_single_example(example, feature_description)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:39.940555Z","iopub.execute_input":"2025-11-08T06:42:39.940800Z","iopub.status.idle":"2025-11-08T06:42:39.945804Z","shell.execute_reply.started":"2025-11-08T06:42:39.940778Z","shell.execute_reply":"2025-11-08T06:42:39.944948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"parsed_dataset = file.map(parse_function)\nparsed_dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:39.946719Z","iopub.execute_input":"2025-11-08T06:42:39.946954Z","iopub.status.idle":"2025-11-08T06:42:39.990536Z","shell.execute_reply.started":"2025-11-08T06:42:39.946930Z","shell.execute_reply":"2025-11-08T06:42:39.989756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def read_tfrecord(example, labeled) : \n    if labeled == True : \n        tfrecord_format = {\"image\" : tf.io.FixedLenFeature([], tf.string),\n                           \"target\" : tf.io.FixedLenFeature([], tf.int64)}\n    else:\n        tfrecord_format = {\"image\" : tf.io.FixedLenFeature([], tf.string),\n                          \"image_name\" : tf.io.FixedLenFeature([], tf.string)}\n    \n    example = tf.io.parse_single_example(example, tfrecord_format)\n    image = decode_image(example[\"image\"])\n    if labeled == True : \n        label = tf.cast(example[\"target\"], tf.int32)\n        return image, label\n    else:\n        image_name = example[\"image_name\"]\n        return image, image_name     ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:39.991431Z","iopub.execute_input":"2025-11-08T06:42:39.991674Z","iopub.status.idle":"2025-11-08T06:42:39.997637Z","shell.execute_reply.started":"2025-11-08T06:42:39.991652Z","shell.execute_reply":"2025-11-08T06:42:39.996758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_dataset(filenames, labeled, ordered):\n    ignore_order = tf.data.Options()\n    if ordered == False: # dataset is unordered, so we ignore the order to load data quickly.\n        ignore_order.experimental_deterministic = False # This disables the order and enhances the speed\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTOTUNE) \n    dataset = dataset.with_options(ignore_order) \n    dataset = dataset.map(partial(read_tfrecord, labeled=labeled), num_parallel_calls=AUTOTUNE)\n    return dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:39.998844Z","iopub.execute_input":"2025-11-08T06:42:39.999111Z","iopub.status.idle":"2025-11-08T06:42:40.017329Z","shell.execute_reply.started":"2025-11-08T06:42:39.999087Z","shell.execute_reply":"2025-11-08T06:42:40.016427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def image_augmentation(image, label) :     \n    image = tf.image.resize(image, SHAPE)\n    image = tf.image.random_flip_left_right(image)\n    return image, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:40.018376Z","iopub.execute_input":"2025-11-08T06:42:40.018689Z","iopub.status.idle":"2025-11-08T06:42:40.029352Z","shell.execute_reply.started":"2025-11-08T06:42:40.018657Z","shell.execute_reply":"2025-11-08T06:42:40.028527Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load The Datasets : ","metadata":{}},{"cell_type":"code","source":"def get_training_dataset() : \n    dataset = load_dataset(training_files, labeled = True, ordered = False)\n    dataset = dataset.map(image_augmentation, num_parallel_calls=AUTOTUNE)\n    dataset = dataset.repeat()\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTOTUNE) \n    return dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:40.030482Z","iopub.execute_input":"2025-11-08T06:42:40.030747Z","iopub.status.idle":"2025-11-08T06:42:40.043579Z","shell.execute_reply.started":"2025-11-08T06:42:40.030708Z","shell.execute_reply":"2025-11-08T06:42:40.042930Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_validation_dataset() : \n    dataset = load_dataset(validation_files, labeled = True, ordered = False)\n    dataset = dataset.map(image_augmentation, num_parallel_calls=AUTOTUNE)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(AUTOTUNE) \n    return dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:40.044630Z","iopub.execute_input":"2025-11-08T06:42:40.044868Z","iopub.status.idle":"2025-11-08T06:42:40.055558Z","shell.execute_reply.started":"2025-11-08T06:42:40.044846Z","shell.execute_reply":"2025-11-08T06:42:40.054927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_test_dataset() : \n    dataset = load_dataset(testing_files, labeled = False, ordered = True)\n    dataset = dataset.map(image_augmentation, num_parallel_calls=AUTOTUNE)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(AUTOTUNE) \n    return dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:40.056545Z","iopub.execute_input":"2025-11-08T06:42:40.057276Z","iopub.status.idle":"2025-11-08T06:42:40.069182Z","shell.execute_reply.started":"2025-11-08T06:42:40.057242Z","shell.execute_reply":"2025-11-08T06:42:40.068316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_dataset = get_training_dataset()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:40.070257Z","iopub.execute_input":"2025-11-08T06:42:40.070574Z","iopub.status.idle":"2025-11-08T06:42:40.274365Z","shell.execute_reply.started":"2025-11-08T06:42:40.070542Z","shell.execute_reply":"2025-11-08T06:42:40.273447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"validation_dataset = get_validation_dataset()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:40.275322Z","iopub.execute_input":"2025-11-08T06:42:40.275580Z","iopub.status.idle":"2025-11-08T06:42:40.314781Z","shell.execute_reply.started":"2025-11-08T06:42:40.275556Z","shell.execute_reply":"2025-11-08T06:42:40.314136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def count_data_items(filenames):\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) for filename in filenames]\n    return np.sum(n)\n\nnum_training_images = count_data_items(training_files)\nnum_validation_images = count_data_items(validation_files)\nnum_testing_images = count_data_items(testing_files)\n\nSTEPS_PER_EPOCH_TRAIN = num_training_images // BATCH_SIZE\nSTEPS_PER_EPOCH_VAL = num_validation_images // BATCH_SIZE\n\nprint(\"Number of Training Images = \", num_training_images)\nprint(\"Number of Validation Images = \", num_validation_images)\nprint(\"Number of Testing Images = \", num_testing_images)\nprint(\"\\n\")\nprint(\"Numer of steps per epoch in Train = \", STEPS_PER_EPOCH_TRAIN)\nprint(\"Numer of steps per epoch in Validation = \", STEPS_PER_EPOCH_VAL)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:40.316239Z","iopub.execute_input":"2025-11-08T06:42:40.316607Z","iopub.status.idle":"2025-11-08T06:42:40.323821Z","shell.execute_reply.started":"2025-11-08T06:42:40.316572Z","shell.execute_reply":"2025-11-08T06:42:40.322905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_batch, label_batch = next(iter(training_dataset))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:40.324819Z","iopub.execute_input":"2025-11-08T06:42:40.325059Z","iopub.status.idle":"2025-11-08T06:42:40.802023Z","shell.execute_reply.started":"2025-11-08T06:42:40.325037Z","shell.execute_reply":"2025-11-08T06:42:40.801024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def show_batch(image_batch, label_batch) :\n    plt.figure(figsize = (20, 20))\n    for n in range(8) : \n        ax = plt.subplot(2,4,n+1)\n        plt.imshow(image_batch[n])\n        if label_batch[n] == 0 : \n            plt.title(\"BENIGN\")\n        else:\n            plt.title(\"MALIGNANT\")\n    plt.grid(False)\n    plt.tight_layout()       ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:40.803328Z","iopub.execute_input":"2025-11-08T06:42:40.803592Z","iopub.status.idle":"2025-11-08T06:42:40.809024Z","shell.execute_reply.started":"2025-11-08T06:42:40.803567Z","shell.execute_reply":"2025-11-08T06:42:40.808020Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show_batch(image_batch.numpy(), label_batch.numpy())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:40.810024Z","iopub.execute_input":"2025-11-08T06:42:40.810364Z","iopub.status.idle":"2025-11-08T06:42:42.978508Z","shell.execute_reply.started":"2025-11-08T06:42:40.810330Z","shell.execute_reply":"2025-11-08T06:42:42.977173Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's free up some memory","metadata":{}},{"cell_type":"code","source":"del image_batch\ndel label_batch\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:42.980272Z","iopub.execute_input":"2025-11-08T06:42:42.980614Z","iopub.status.idle":"2025-11-08T06:42:43.162942Z","shell.execute_reply.started":"2025-11-08T06:42:42.980582Z","shell.execute_reply":"2025-11-08T06:42:43.161999Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Construction : ","metadata":{}},{"cell_type":"code","source":"malignant = len(train[train[\"target\"] == 1])\nbenign = len(train[train[\"target\"] == 0 ])\ntotal = len(train) \n\nprint(\"Malignant Cases in Train Data = \", malignant)\nprint(\"Benign Cases In Train Dataset = \",benign)\nprint(\"Total Cases In Train Dataset = \",total)\nprint(\"Ratio of Malignant to Benign = \",malignant/benign)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:43.164081Z","iopub.execute_input":"2025-11-08T06:42:43.164390Z","iopub.status.idle":"2025-11-08T06:42:43.184513Z","shell.execute_reply.started":"2025-11-08T06:42:43.164355Z","shell.execute_reply":"2025-11-08T06:42:43.183591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"weight_malignant = (total/malignant)/2.0\nweight_benign = (total/benign)/2.0\n\nclass_weight = {0 : weight_benign , 1 : weight_malignant}\n\nprint(\"Weight for benign cases = \", class_weight[0])\nprint(\"Weight for malignant cases = \", class_weight[1])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:43.185819Z","iopub.execute_input":"2025-11-08T06:42:43.186099Z","iopub.status.idle":"2025-11-08T06:42:43.191408Z","shell.execute_reply.started":"2025-11-08T06:42:43.186075Z","shell.execute_reply":"2025-11-08T06:42:43.190548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"callback_early_stopping = tf.keras.callbacks.EarlyStopping(patience = 15, verbose = 0, restore_best_weights = True)\n\ncallbacks_lr_reduce = tf.keras.callbacks.ReduceLROnPlateau(monitor = \"val_auc\", factor = 0.1, patience = 10, \n                                                          verbose = 0, min_lr = 1e-6)\n\ncallback_checkpoint = tf.keras.callbacks.ModelCheckpoint(\"melanoma_weights.h5\",\n                                                         save_weights_only=True, monitor='val_auc',\n                                                         mode='max', save_best_only = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:43.192551Z","iopub.execute_input":"2025-11-08T06:42:43.192797Z","iopub.status.idle":"2025-11-08T06:42:44.319540Z","shell.execute_reply.started":"2025-11-08T06:42:43.192774Z","shell.execute_reply":"2025-11-08T06:42:44.318818Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model Design : MobileNetV2\n\nA supercool resource : **https://machinethink.net/blog/mobilenet-v2/**","metadata":{},"attachments":{}},{"cell_type":"markdown","source":"## Bias Initialization : \n\nSince the dataset is heavily imbalanced, we may want to assign different weights to different classes. Setting an initial bias is important in such cases.","metadata":{}},{"cell_type":"code","source":"# with strategy.scope() : \n    \n#     # Khởi tạo bias cho lớp cuối\n#     bias = np.log(malignant/benign)\n#     bias = tf.keras.initializers.Constant(bias)\n    \n#     # TẠO MÔ HÌNH (Giai đoạn 1: Freeze)\n#     base_model = tf.keras.applications.MobileNetV2(input_shape = (SHAPE[0], SHAPE[1], 3), \n#                                                    include_top = False,\n#                                                    weights = \"imagenet\")\n#     base_model.trainable = False\n    \n#     global model, history, history_fine_tune\n    \n#     model = tf.keras.Sequential([base_model,\n#                                  tf.keras.layers.GlobalAveragePooling2D(),\n#                                  tf.keras.layers.Dense(20, activation = \"relu\"), \n#                                  tf.keras.layers.Dropout(0.4),\n#                                  tf.keras.layers.Dense(10, activation = \"relu\"),\n#                                  tf.keras.layers.Dropout(0.3),\n#                                  tf.keras.layers.Dense(1, activation = \"sigmoid\", bias_initializer = bias)\n#                                ])\n    \n#     # COMPILE MÔ HÌNH GIAI ĐOẠN 1 (LR = 1e-4)\n#     model.compile(optimizer = tf.keras.optimizers.Adam(lr = 1e-4), \n#                   loss = \"binary_crossentropy\", \n#                   metrics = [tf.keras.metrics.AUC(name = 'auc')])\n    \n#     print(\"--- BẮT ĐẦU GIAI ĐOẠN 1 (FREEZE) ---\")\n\n#     # HUẤN LUYỆN GIAI ĐOẠN 1: FREEZE (8 EPOCHS)\n#     EPOCHS_FREEZE = 100\n#     history = model.fit(training_dataset, \n#                         epochs = EPOCHS_FREEZE, \n#                         steps_per_epoch = STEPS_PER_EPOCH_TRAIN,\n#                         validation_data = validation_dataset, \n#                         validation_steps = STEPS_PER_EPOCH_VAL,\n#                         callbacks = [callback_early_stopping, callbacks_lr_reduce, callback_checkpoint],\n#                         class_weight = class_weight)\n\n#     # --- GIAI ĐOẠN 2: FINE-TUNING ---\n    \n#     # Mở đóng băng một phần của mô hình Base\n#     base_model.trainable = True \n#     for layer in base_model.layers[:-20]: \n#         layer.trainable = False\n    \n#     # # COMPILE LẠI MÔ HÌNH GIAI ĐOẠN 2 (LR = 1e-6)\n#     # model.compile(optimizer = tf.keras.optimizers.Adam(lr = 1e-6), \n#     #               loss = \"binary_crossentropy\", \n#     #               metrics = [tf.keras.metrics.AUC(name = 'auc')])\n\n#     print(\"\\n--- BẮT ĐẦU GIAI ĐOẠN 2 (FINE-TUNING) ---\")\n    \n#     # HUẤN LUYỆN GIAI ĐOẠN 2: FINE-TUNE (18 EPOCHS)\n#     FINE_TUNE_EPOCHS = 200\n#     TOTAL_EPOCHS = EPOCHS_FREEZE + FINE_TUNE_EPOCHS\n#     START_EPOCH = history.epoch[-1] + 1 if history.epoch else 0 \n    \n#     history_fine_tune = model.fit(training_dataset, \n#                                   epochs = TOTAL_EPOCHS,\n#                                   initial_epoch = START_EPOCH,\n#                                   steps_per_epoch = STEPS_PER_EPOCH_TRAIN,\n#                                   validation_data = validation_dataset, \n#                                   validation_steps = STEPS_PER_EPOCH_VAL,\n#                                   callbacks = [callback_early_stopping, callbacks_lr_reduce, callback_checkpoint],\n#                                   class_weight = class_weight)\n    \n# # KẾT THÚC khối strategy.scope()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:44.320829Z","iopub.execute_input":"2025-11-08T06:42:44.321105Z","iopub.status.idle":"2025-11-08T06:42:44.327194Z","shell.execute_reply.started":"2025-11-08T06:42:44.321080Z","shell.execute_reply":"2025-11-08T06:42:44.326080Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with strategy.scope():\n\n    # --- KHỞI TẠO bias CHO LỚP CUỐI ---\n    bias = np.log(malignant / benign)\n    bias = tf.keras.initializers.Constant(bias)\n\n    # --- GIAI ĐOẠN 1: FREEZE ---\n    base_model = tf.keras.applications.MobileNetV2(\n        input_shape=(SHAPE[0], SHAPE[1], 3),\n        include_top=False,\n        weights=\"imagenet\"\n    )\n    base_model.trainable = False  # đóng băng toàn bộ\n\n    global model, history, history_fine_tune\n\n    model = tf.keras.Sequential([\n        base_model,\n        tf.keras.layers.GlobalAveragePooling2D(),\n        tf.keras.layers.Dense(20, activation=\"relu\"),\n        tf.keras.layers.Dropout(0.4),\n        tf.keras.layers.Dense(10, activation=\"relu\"),\n        tf.keras.layers.Dropout(0.3),\n        tf.keras.layers.Dense(1, activation=\"sigmoid\", bias_initializer=bias)\n    ])\n\n    # --- COMPILE GIAI ĐOẠN 1 ---\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4),\n        loss=\"binary_crossentropy\",\n        metrics=[tf.keras.metrics.AUC(name='auc')]\n    )\n\n    print(\"\\n--- BẮT ĐẦU GIAI ĐOẠN 1 (FREEZE) ---\")\n    EPOCHS_FREEZE = 10\n\n    history = model.fit(\n        training_dataset,\n        epochs=EPOCHS_FREEZE,\n        steps_per_epoch=STEPS_PER_EPOCH_TRAIN,\n        validation_data=validation_dataset,\n        validation_steps=STEPS_PER_EPOCH_VAL,\n        callbacks=[callback_early_stopping, callbacks_lr_reduce, callback_checkpoint],\n        class_weight=class_weight\n    )\n\n    # --- GIAI ĐOẠN 2: FINE-TUNING ---\n    print(\"\\n--- BẮT ĐẦU GIAI ĐOẠN 2 (FINE-TUNING) ---\")\n\n    # Mở trainable cho base_model\n    base_model.trainable = True\n\n    # Giữ nguyên phần lớn layer freeze (chỉ mở 20 lớp cuối)\n    for layer in base_model.layers[:-20]:\n        layer.trainable = False\n\n    # Giữ BatchNorm không trainable để tránh lệch thống kê\n    for layer in base_model.layers:\n        if isinstance(layer, tf.keras.layers.BatchNormalization):\n            layer.trainable = False\n\n    # Tạo optimizer mới, LR rất nhỏ\n    fine_tune_optimizer = tf.keras.optimizers.Adam(learning_rate=1e-6)\n\n    # COMPILE LẠI MÔ HÌNH\n    model.compile(\n        optimizer=fine_tune_optimizer,\n        loss=\"binary_crossentropy\",\n        metrics=[tf.keras.metrics.AUC(name='auc')]\n    )\n\n    # Tiếp tục train\n    FINE_TUNE_EPOCHS = 10\n    TOTAL_EPOCHS = EPOCHS_FREEZE + FINE_TUNE_EPOCHS\n    START_EPOCH = history.epoch[-1] + 1 if history.epoch else 0\n\n    history_fine_tune = model.fit(\n        training_dataset,\n        epochs=TOTAL_EPOCHS,\n        initial_epoch=START_EPOCH,\n        steps_per_epoch=STEPS_PER_EPOCH_TRAIN,\n        validation_data=validation_dataset,\n        validation_steps=STEPS_PER_EPOCH_VAL,\n        callbacks=[callback_early_stopping, callbacks_lr_reduce, callback_checkpoint],\n        class_weight=class_weight\n    )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:42:44.328728Z","iopub.execute_input":"2025-11-08T06:42:44.329073Z","iopub.status.idle":"2025-11-08T07:46:44.388903Z","shell.execute_reply.started":"2025-11-08T06:42:44.329040Z","shell.execute_reply":"2025-11-08T07:46:44.377397Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_epochs_it_ran_for = len(history.history['loss'])\nn_epochs_it_ran_for","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T07:46:44.399974Z","iopub.execute_input":"2025-11-08T07:46:44.401672Z","iopub.status.idle":"2025-11-08T07:46:44.426633Z","shell.execute_reply.started":"2025-11-08T07:46:44.401642Z","shell.execute_reply":"2025-11-08T07:46:44.425809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = np.arange(0,n_epochs_it_ran_for,1)\nplt.figure(1, figsize = (20, 12))\nplt.subplot(1,2,1)\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.plot(X, history.history[\"loss\"], label = \"Training Loss\")\nplt.plot(X, history.history[\"val_loss\"], label = \"Validation Loss\")\nplt.grid(True)\nplt.legend()\n\nplt.subplot(1,2,2)\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.plot(X, history.history[\"auc\"], label = \"Training Accuracy\")\nplt.plot(X, history.history[\"val_auc\"], label = \"Validation Accuracy\")\nplt.grid(True)\nplt.legend()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T07:46:44.427764Z","iopub.execute_input":"2025-11-08T07:46:44.427989Z","iopub.status.idle":"2025-11-08T07:46:44.973226Z","shell.execute_reply.started":"2025-11-08T07:46:44.427968Z","shell.execute_reply":"2025-11-08T07:46:44.972234Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Due to callbacks, best weights are automatically restored!","metadata":{}},{"cell_type":"code","source":"resulting_probabilities = model.predict(testing_dataset_images, verbose = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T07:46:44.974346Z","iopub.execute_input":"2025-11-08T07:46:44.974612Z","iopub.status.idle":"2025-11-08T07:46:45.558984Z","shell.execute_reply.started":"2025-11-08T07:46:44.974588Z","shell.execute_reply":"2025-11-08T07:46:45.557609Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(resulting_probabilities)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T07:46:45.560036Z","iopub.status.idle":"2025-11-08T07:46:45.560534Z","shell.execute_reply.started":"2025-11-08T07:46:45.560285Z","shell.execute_reply":"2025-11-08T07:46:45.560308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_submission_file = pd.read_csv(\"../input/siim-isic-melanoma-classification/sample_submission.csv\")\nsample_submission_file.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T07:47:42.441605Z","iopub.execute_input":"2025-11-08T07:47:42.442448Z","iopub.status.idle":"2025-11-08T07:47:42.540531Z","shell.execute_reply.started":"2025-11-08T07:47:42.442415Z","shell.execute_reply":"2025-11-08T07:47:42.539560Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del sample_submission_file[\"target\"]\nsample_submission_file.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T07:47:45.438033Z","iopub.execute_input":"2025-11-08T07:47:45.438392Z","iopub.status.idle":"2025-11-08T07:47:45.464330Z","shell.execute_reply.started":"2025-11-08T07:47:45.438361Z","shell.execute_reply":"2025-11-08T07:47:45.463160Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"testing_image_names","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T07:47:47.688740Z","iopub.execute_input":"2025-11-08T07:47:47.689361Z","iopub.status.idle":"2025-11-08T07:47:47.708702Z","shell.execute_reply.started":"2025-11-08T07:47:47.689327Z","shell.execute_reply":"2025-11-08T07:47:47.707393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"testing_image_names = np.concatenate([x for x in testing_image_names], axis=0)\ntesting_image_names = np.array(testing_image_names)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T07:48:26.670485Z","iopub.execute_input":"2025-11-08T07:48:26.671499Z","iopub.status.idle":"2025-11-08T07:48:26.690293Z","shell.execute_reply.started":"2025-11-08T07:48:26.671463Z","shell.execute_reply":"2025-11-08T07:48:26.689162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"decoded_test_names = []\nfor names in testing_image_names : \n    names = names.decode('utf-8')\n    decoded_test_names.append(names)\ndecoded_test_names = np.array(decoded_test_names)\ndel testing_image_names","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T07:48:28.958839Z","iopub.execute_input":"2025-11-08T07:48:28.959680Z","iopub.status.idle":"2025-11-08T07:48:28.979337Z","shell.execute_reply.started":"2025-11-08T07:48:28.959645Z","shell.execute_reply":"2025-11-08T07:48:28.978139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(decoded_test_names), type(decoded_test_names), decoded_test_names.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T07:46:45.571353Z","iopub.status.idle":"2025-11-08T07:46:45.571793Z","shell.execute_reply.started":"2025-11-08T07:46:45.571556Z","shell.execute_reply":"2025-11-08T07:46:45.571577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"decoded_test_names","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T07:46:45.573427Z","iopub.status.idle":"2025-11-08T07:46:45.573742Z","shell.execute_reply.started":"2025-11-08T07:46:45.573588Z","shell.execute_reply":"2025-11-08T07:46:45.573602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"testing_image_names = pd.DataFrame(decoded_test_names, columns=[\"image_name\"])\ntesting_image_names.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T07:46:45.575193Z","iopub.status.idle":"2025-11-08T07:46:45.575504Z","shell.execute_reply.started":"2025-11-08T07:46:45.575356Z","shell.execute_reply":"2025-11-08T07:46:45.575371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_dataframe = pd.DataFrame({\"image_name\" : decoded_test_names, \n                               \"target\" : np.concatenate(resulting_probabilities)})\npred_dataframe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T07:46:45.576915Z","iopub.status.idle":"2025-11-08T07:46:45.577256Z","shell.execute_reply.started":"2025-11-08T07:46:45.577068Z","shell.execute_reply":"2025-11-08T07:46:45.577082Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_submission_file = sample_submission_file.merge(pred_dataframe, on = \"image_name\")\nsample_submission_file.to_csv(\"submission.csv\", index = False)\nsample_submission_file.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T07:46:45.579028Z","iopub.status.idle":"2025-11-08T07:46:45.579481Z","shell.execute_reply.started":"2025-11-08T07:46:45.579250Z","shell.execute_reply":"2025-11-08T07:46:45.579271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save(\"melanoma_model.h5\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T07:48:34.691580Z","iopub.execute_input":"2025-11-08T07:48:34.692289Z","iopub.status.idle":"2025-11-08T07:48:35.005612Z","shell.execute_reply.started":"2025-11-08T07:48:34.692247Z","shell.execute_reply":"2025-11-08T07:48:35.004760Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===========================================\n# 🔍 ĐÁNH GIÁ MÔ HÌNH MELANOMA TRÊN DỮ LIỆU THẬT (KAGGLE)\n# ===========================================\nimport tensorflow as tf\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix, classification_report, roc_curve, auc\nimport os\n\nprint(\"\\n🚀 BẮT ĐẦU ĐÁNH GIÁ MÔ HÌNH TRÊN DỮ LIỆU THẬT\")\n\n# ====== 1️⃣ CẤU HÌNH ======\nDATA_DIR = \"/kaggle/input/siim-isic-melanoma-classification\"\nTRAIN_CSV = os.path.join(DATA_DIR, \"train.csv\")\nTRAIN_DIR = os.path.join(DATA_DIR, \"jpeg/train\")\n\nIMG_SIZE = 384\nBATCH_SIZE = 32\n\n# ====== 2️⃣ TẠO DATASET KIỂM ĐỊNH TỪ DỮ LIỆU THẬT ======\nprint(\"\\nĐang đọc file train.csv...\")\ndf = pd.read_csv(TRAIN_CSV)\nprint(f\"Tổng số ảnh trong dataset: {len(df)}\")\n\n# 🧩 Lấy mẫu nhỏ để đánh giá nhanh (bạn có thể tăng nếu muốn)\ndf_sample = df.sample(2000, random_state=42).reset_index(drop=True)\n\n# Đường dẫn ảnh thật\nimage_paths = [os.path.join(TRAIN_DIR, f\"{img_id}.jpg\") for img_id in df_sample[\"image_name\"]]\nlabels = df_sample[\"target\"].values\n\n# Hàm xử lý ảnh\ndef decode_image(filename, label):\n    bits = tf.io.read_file(filename)\n    image = tf.image.decode_jpeg(bits, channels=3)\n    image = tf.image.resize(image, [IMG_SIZE, IMG_SIZE])\n    image = image / 255.0\n    return image, label\n\n# Tạo dataset TensorFlow\neval_dataset = tf.data.Dataset.from_tensor_slices((image_paths, labels))\neval_dataset = eval_dataset.map(decode_image, num_parallel_calls=tf.data.AUTOTUNE)\neval_dataset = eval_dataset.batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\nnum_validation_images = len(df_sample)\nprint(f\"Số lượng ảnh được dùng để đánh giá: {num_validation_images}\")\n\n# ====== 3️⃣ TẢI LẠI MÔ HÌNH HUẤN LUYỆN ======\nprint(\"\\nĐang tải mô hình đầy đủ từ 'melanoma_model.h5'...\")\nmodel_eval = tf.keras.models.load_model(\"/kaggle/working/melanoma_model.h5\")\nprint(\"✅ Mô hình tải thành công.\")\n\n# ====== 4️⃣ DỰ ĐOÁN ======\nprint(\"\\nĐang chạy dự đoán...\")\ny_true = np.array(labels)\ny_pred_probs = model_eval.predict(eval_dataset.map(lambda x, y: x), verbose=1)\ny_pred_labels = (y_pred_probs > 0.5).astype(int)\n\n# ====== 5️⃣ ĐÁNH GIÁ HIỆU SUẤT ======\nprint(\"\\n📊 KẾT QUẢ ĐÁNH GIÁ MÔ HÌNH\")\n\n# Báo cáo phân loại\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_true, y_pred_labels, target_names=['Benign (0)', 'Malignant (1)']))\n\n# Ma trận nhầm lẫn\ncm = confusion_matrix(y_true, y_pred_labels)\ntn, fp, fn, tp = cm.ravel()\nspecificity = tn / (tn + fp)\nprint(f\"Sensitivity (Recall): {tp / (tp + fn):.4f}\")\nprint(f\"Specificity:          {specificity:.4f}\")\n\n# Vẽ Confusion Matrix\nplt.figure(figsize=(8, 6))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n            xticklabels=['Pred Benign (0)', 'Pred Malignant (1)'],\n            yticklabels=['True Benign (0)', 'True Malignant (1)'])\nplt.title('Confusion Matrix (Evaluation Set)')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.show()\n\n# ROC & AUC\nfpr, tpr, thresholds = roc_curve(y_true, y_pred_probs)\nroc_auc = auc(fpr, tpr)\nprint(f\"\\nAUC (từ sklearn): {roc_auc:.4f}\")\n\nplt.figure(figsize=(10, 8))\nplt.plot(fpr, tpr, color='darkorange', lw=2, label=f'ROC (AUC = {roc_auc:.4f})')\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate (1 - Specificity)')\nplt.ylabel('True Positive Rate (Sensitivity)')\nplt.title('ROC Curve')\nplt.legend(loc=\"lower right\")\nplt.grid(True)\nplt.show()\n\nprint(\"\\n✅ Đánh giá hoàn tất.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T08:04:50.188996Z","iopub.execute_input":"2025-11-08T08:04:50.189359Z","iopub.status.idle":"2025-11-08T08:05:46.556901Z","shell.execute_reply.started":"2025-11-08T08:04:50.189327Z","shell.execute_reply":"2025-11-08T08:05:46.555986Z"}},"outputs":[],"execution_count":null}]}