{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"},{"sourceId":1322485,"sourceType":"datasetVersion","datasetId":758225},{"sourceId":1322494,"sourceType":"datasetVersion","datasetId":688574},{"sourceId":1322517,"sourceType":"datasetVersion","datasetId":689329},{"sourceId":1322552,"sourceType":"datasetVersion","datasetId":688719},{"sourceId":1322612,"sourceType":"datasetVersion","datasetId":689578},{"sourceId":1324333,"sourceType":"datasetVersion","datasetId":762256},{"sourceId":1324349,"sourceType":"datasetVersion","datasetId":762108},{"sourceId":1324366,"sourceType":"datasetVersion","datasetId":762176},{"sourceId":1324385,"sourceType":"datasetVersion","datasetId":762138},{"sourceId":1324412,"sourceType":"datasetVersion","datasetId":762168},{"sourceId":1324430,"sourceType":"datasetVersion","datasetId":768238},{"sourceId":1324483,"sourceType":"datasetVersion","datasetId":758409},{"sourceId":1339671,"sourceType":"datasetVersion","datasetId":758489}],"dockerImageVersionId":30299,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install --upgrade \"tensorflow>=2.9.0\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport random\nimport re\nimport math\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport tensorflow.keras.backend as K\nimport matplotlib.pyplot as plt\nimport albumentations as A\nimport cv2\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score\nfrom tensorflow.keras.layers import Input, Dense, GlobalAveragePooling2D, Concatenate, Dropout\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.applications import efficientnet_v2\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:49:34.341012Z","iopub.execute_input":"2024-12-07T13:49:34.341477Z","iopub.status.idle":"2024-12-07T13:49:39.422149Z","shell.execute_reply.started":"2024-12-07T13:49:34.341427Z","shell.execute_reply":"2024-12-07T13:49:39.421407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================================\n# Configuration\n# =========================================\nSEED = 42\nFOLDS = 5\nIMG_SIZE = 384\nBATCH_SIZE = 32\nEPOCHS = 12\nTTA = 5\nLR = 1e-4\n\n# Set seeds for reproducibility\ndef seed_everything(seed=42):\n    np.random.seed(seed)\n    random.seed(seed)\n    tf.random.set_seed(seed)\n\nseed_everything(SEED)\n\n# Detect and initialize TPU (if available)\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept:\n    strategy = tf.distribute.get_strategy()\n\nREPLICAS = strategy.num_replicas_in_sync\nAUTO = tf.data.experimental.AUTOTUNE","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:49:39.423127Z","iopub.execute_input":"2024-12-07T13:49:39.423597Z","iopub.status.idle":"2024-12-07T13:49:39.430683Z","shell.execute_reply.started":"2024-12-07T13:49:39.423572Z","shell.execute_reply":"2024-12-07T13:49:39.429726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# =========================================\n# Load metadata and create triple stratified folds\n# =========================================\ntrain_df = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\ntrain_df['sex'] = train_df['sex'].fillna('unknown')\ntrain_df['age_approx'] = train_df['age_approx'].fillna(train_df['age_approx'].median())\ntrain_df['anatom_site_general_challenge'] = train_df['anatom_site_general_challenge'].fillna('unknown')\n\n# Encode categorical variables\nsex_map = {v: i for i, v in enumerate(train_df['sex'].unique())}\nanatom_map = {v: i for i, v in enumerate(train_df['anatom_site_general_challenge'].unique())}\n\ntrain_df['sex_enc'] = train_df['sex'].map(sex_map)\ntrain_df['anatom_enc'] = train_df['anatom_site_general_challenge'].map(anatom_map)\n\n# Custom triple stratification:\n# We want to ensure stratification by target primarily, \n# then we can create bins for age and combine sex and anatom for complexity.\n# A simple approach: create a combined strata key from target, sex_enc, and anatom_enc\n# and stratify on that. For a more sophisticated approach, you might cluster or use other techniques.\nstrata = train_df['target'].astype(str) + \"_\" + train_df['sex_enc'].astype(str) + \"_\" + train_df['anatom_enc'].astype(str)\n\nskf = StratifiedKFold(n_splits=FOLDS, shuffle=True, random_state=SEED)\ntrain_df['fold'] = -1\nfor i, (trn_idx, val_idx) in enumerate(skf.split(train_df, strata)):\n    train_df.loc[val_idx, 'fold'] = i\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:49:39.432562Z","iopub.execute_input":"2024-12-07T13:49:39.432818Z","iopub.status.idle":"2024-12-07T13:49:41.041827Z","shell.execute_reply.started":"2024-12-07T13:49:39.432796Z","shell.execute_reply":"2024-12-07T13:49:41.040829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================================\n# TFRecords Paths (Adjust to your dataset)\n# =========================================\nGCS_PATH = '../input/melanoma-384x384'  # Example path\nfiles_train_all = tf.io.gfile.glob(GCS_PATH + '/train*.tfrec')\nfiles_test = tf.io.gfile.glob(GCS_PATH + '/test*.tfrec')\nfiles_test = np.sort(np.array(files_test))\n\ndef count_data_items(filenames):\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(fname).group(1)) for fname in filenames]\n    return np.sum(n)\n\nNUM_TEST = count_data_items(files_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:49:41.042997Z","iopub.execute_input":"2024-12-07T13:49:41.043268Z","iopub.status.idle":"2024-12-07T13:49:41.06097Z","shell.execute_reply.started":"2024-12-07T13:49:41.043244Z","shell.execute_reply":"2024-12-07T13:49:41.060131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================================\n# Albumentations Augmentation\n# =========================================\n# We can use Albumentations for more advanced augmentations\ndef build_augmenter():\n    return A.Compose([\n        A.RandomResizedCrop(IMG_SIZE, IMG_SIZE, scale=(0.8,1.0), p=0.5),\n        A.HorizontalFlip(p=0.5),\n        A.VerticalFlip(p=0.5),\n        A.RandomBrightnessContrast(p=0.5),\n        A.ShiftScaleRotate(shift_limit=0.1, scale_limit=0.1, rotate_limit=45, p=0.5)\n    ])\n\naugmenter = build_augmenter()\n\ndef aug_fn(image):\n    # image is a numpy array (H, W, C) in float32 or uint8\n    # Apply Albumentations\n    data = {\"image\": image}\n    aug_data = augmenter(**data)\n    aug_img = aug_data[\"image\"]  # This might be uint8 by default\n    \n    # Convert to float32 here\n    aug_img = aug_img.astype(np.float32)  # ensure float32\n    # Now the function returns float32 array\n    return aug_img\n\n\ndef read_tfrecord(example, labeled=True):\n    if labeled:\n        tfrec_format = {\n            'image': tf.io.FixedLenFeature([], tf.string),\n            'target': tf.io.FixedLenFeature([], tf.int64),\n            'sex': tf.io.FixedLenFeature([], tf.int64),\n            'age_approx': tf.io.FixedLenFeature([], tf.int64),\n            'anatom_site_general_challenge': tf.io.FixedLenFeature([], tf.int64)\n        }\n        example = tf.io.parse_single_example(example, tfrec_format)\n        return example['image'], example['target'], example['sex'], example['age_approx'], example['anatom_site_general_challenge']\n    else:\n        tfrec_format = {\n            'image': tf.io.FixedLenFeature([], tf.string),\n            'image_name': tf.io.FixedLenFeature([], tf.string)\n        }\n        example = tf.io.parse_single_example(example, tfrec_format)\n        return example['image'], example['image_name']\n\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32)/255.0\n    image = tf.reshape(image, [IMG_SIZE, IMG_SIZE, 3])\n    return image\n\ndef augment_image(image):\n    image = tf.numpy_function(func=aug_fn, inp=[image], Tout=tf.float32)\n    # Now image is float32, we can divide by 255 if needed\n    image = image / 255.0\n    image = tf.reshape(image, [IMG_SIZE, IMG_SIZE, 3])\n    return image\n\ndef get_training_dataset(filenames, batch_size, repeat=True, shuffle=True):\n    ds = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    ds = ds.map(lambda ex: read_tfrecord(ex, labeled=True), num_parallel_calls=AUTO)\n    ds = ds.map(lambda img, tgt, sex, age, site: (decode_image(img), (tgt, sex, age, site)), num_parallel_calls=AUTO)\n\n    if shuffle:\n        ds = ds.shuffle(2048)\n    if repeat:\n        ds = ds.repeat()\n    # Apply augmentation on-the-fly\n    ds = ds.map(lambda img, label_tuple: (augment_image(img), label_tuple), num_parallel_calls=AUTO)\n    ds = ds.batch(batch_size * REPLICAS)\n    ds = ds.prefetch(AUTO)\n    return ds\n\ndef get_validation_dataset(filenames, batch_size):\n    ds = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    ds = ds.map(lambda ex: read_tfrecord(ex, labeled=True), num_parallel_calls=AUTO)\n    ds = ds.map(lambda img, tgt, sex, age, site: (decode_image(img), (tgt, sex, age, site)), num_parallel_calls=AUTO)\n    ds = ds.batch(batch_size * REPLICAS)\n    ds = ds.prefetch(AUTO)\n    return ds\n\ndef get_test_dataset(filenames, batch_size, repeat=False, augment=False, labeled=False):\n    ds = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    ds = ds.map(lambda ex: read_tfrecord(ex, labeled=False), num_parallel_calls=AUTO)\n    ds = ds.map(lambda img, name: (decode_image(img), name), num_parallel_calls=AUTO)\n    if augment:\n        ds = ds.map(lambda img, name: (augment_image(img), name), num_parallel_calls=AUTO)\n    if repeat:\n        ds = ds.repeat()\n    ds = ds.batch(batch_size * REPLICAS)\n    ds = ds.prefetch(AUTO)\n    return ds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:49:41.062147Z","iopub.execute_input":"2024-12-07T13:49:41.062394Z","iopub.status.idle":"2024-12-07T13:49:41.080368Z","shell.execute_reply.started":"2024-12-07T13:49:41.062372Z","shell.execute_reply":"2024-12-07T13:49:41.079461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================================\n# Build a Model with Meta-Features\n# =========================================\n# We will have two inputs:\n# 1) Image input through EfficientNetV2\n# 2) Meta input (sex, age, site) that will be concatenated before the final dense layer\n\ndef build_model(dim=384):\n    img_input = Input(shape=(dim,dim,3), name='img_input')\n    base_model = efficientnet_v2.EfficientNetV2B0(include_top=False, input_tensor=img_input, weights='imagenet')\n    x = GlobalAveragePooling2D()(base_model.output)\n\n    meta_input = Input(shape=(3,), name='meta_input')\n    m = Dense(16, activation='relu')(meta_input)\n    m = Dropout(0.1)(m)\n\n    combined = Concatenate()([x, m])\n    combined = Dense(32, activation='relu')(combined)\n    combined = Dropout(0.1)(combined)\n    out = Dense(1, activation='sigmoid')(combined)\n\n    model = Model(inputs=[img_input, meta_input], outputs=out)\n    return model\n\n# Custom data generator function to separate image and meta features\ndef process_batch(features, label_tuple):\n    # label_tuple = (tgt, sex, age, site)\n    tgt = label_tuple[0]\n    sex = tf.cast(label_tuple[1], tf.float32)\n    age = tf.cast(label_tuple[2], tf.float32)\n    site = tf.cast(label_tuple[3], tf.float32)\n    meta = tf.stack([sex, age, site], axis=1)\n    return (features, meta), tgt\n\n# Custom Focal Loss\ndef focal_loss(gamma=2.0, alpha=0.25):\n    def focal_crossentropy(y_true, y_pred):\n        y_true = tf.cast(y_true, tf.float32)\n        bce = K.binary_crossentropy(y_true, y_pred)\n        bce = K.flatten(bce)\n        y_pred = K.flatten(y_pred)\n        y_true = K.flatten(y_true)\n        pt = tf.where(tf.equal(y_true,1), y_pred, 1-y_pred)\n        loss = bce * ((1-pt)**gamma)\n        if alpha is not None:\n            alpha_factor = tf.where(tf.equal(y_true,1), alpha, 1-alpha)\n            loss = alpha_factor * loss\n        return tf.reduce_mean(loss)\n    return focal_crossentropy\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:49:41.08167Z","iopub.execute_input":"2024-12-07T13:49:41.082016Z","iopub.status.idle":"2024-12-07T13:49:41.096549Z","shell.execute_reply.started":"2024-12-07T13:49:41.081985Z","shell.execute_reply":"2024-12-07T13:49:41.095668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================================\n# Training & Inference\n# =========================================\noof_preds = []\noof_targets = []\ntest_preds = np.zeros((NUM_TEST,1))\n\nfor fold in range(FOLDS):\n    print(\"=\"*30)\n    print(f\"Starting Fold {fold+1}\")\n    val_df = train_df[train_df.fold == fold]\n    trn_df = train_df[train_df.fold != fold]\n\n    # Match TFRecords by image_name if you have that info in TFRecords\n    # For demonstration, assume files train00.tfrec ... each corresponds to partial data.\n    # In practice, you'd filter TFRecords based on image ids in these subsets.\n    # A simpler approach (not perfect) is to use all training tfrecords and rely on the model to\n    # only get a correct validation set. A more accurate method is needed in real scenario.\n\n    # For simplicity, let's assume that each TFRecord is a chunk of data and we can just \n    # re-split by fold assigned at generation time. If not, you'd need a dictionary mapping.\n\n    # Example: If your TFRecords are created per fold, something like:\n    trn_files = [f for f in files_train_all if any(img_id in f for img_id in trn_df.image_name.values)]\n    val_files = [f for f in files_train_all if any(img_id in f for img_id in val_df.image_name.values)]\n\n    if len(val_files) == 0:\n        # fallback: just for demonstration (not ideal)\n        # In a real scenario, ensure TFRecords are mapped properly.\n        val_files = files_train_all[fold::FOLDS]\n        trn_files = list(set(files_train_all)-set(val_files))\n\n    print(\"Num Train Files:\", len(trn_files))\n    print(\"Num Val Files:\", len(val_files))\n\n    train_ds = get_training_dataset(trn_files, BATCH_SIZE)\n    val_ds = get_validation_dataset(val_files, BATCH_SIZE)\n\n    # Map dataset to include meta inputs\n    train_ds = train_ds.map(process_batch, num_parallel_calls=AUTO)\n    val_ds = val_ds.map(process_batch, num_parallel_calls=AUTO)\n\n    with strategy.scope():\n        model = build_model(IMG_SIZE)\n        model.compile(optimizer=Adam(LR), loss=focal_loss(gamma=2.0, alpha=0.25), metrics=['AUC'])\n\n    ckpt = ModelCheckpoint(f'fold-{fold}.h5', monitor='val_AUC', mode='max', save_best_only=True, verbose=1)\n    es = EarlyStopping(monitor='val_AUC', mode='max', patience=4, restore_best_weights=True)\n    rlr = ReduceLROnPlateau(monitor='val_AUC', mode='max', factor=0.5, patience=2, verbose=1)\n\n    steps_per_epoch = count_data_items(trn_files)//(BATCH_SIZE)\n    val_steps = count_data_items(val_files)//(BATCH_SIZE)\n\n    history = model.fit(\n        train_ds,\n        validation_data=val_ds,\n        epochs=EPOCHS,\n        steps_per_epoch=steps_per_epoch,\n        validation_steps=val_steps,\n        callbacks=[ckpt, es, rlr],\n        verbose=1\n    )\n\n    # OOF Predictions\n    model.load_weights(f'fold-{fold}.h5')\n    # Create a non-augmented validation set for OOF predictions\n    val_ds_oof = get_validation_dataset(val_files, BATCH_SIZE)\n    val_ds_oof = val_ds_oof.map(process_batch, num_parallel_calls=AUTO)\n\n    val_preds = model.predict(val_ds_oof, verbose=1)\n    val_targets = val_df.target.values[:len(val_preds)]\n\n    oof_preds.append(val_preds)\n    oof_targets.append(val_targets)\n\n    # Test Predictions with TTA\n    test_ds = get_test_dataset(files_test, BATCH_SIZE, repeat=True, augment=True, labeled=False)\n    # For test, we have no meta, so we can set meta to zeros or mean. \n    # In a real scenario, if no meta for test, handle accordingly. Suppose test_meta = zeros:\n    def add_dummy_meta(img, name):\n        # Just add dummy meta as (sex=0, age=50, site=0) example\n        meta = tf.convert_to_tensor([[0.0, 50.0, 0.0]], dtype=tf.float32)\n        meta = tf.repeat(meta, tf.shape(img)[0], axis=0)\n        return (img, meta), name\n    test_ds = test_ds.map(add_dummy_meta, num_parallel_calls=AUTO)\n\n    ct_test = NUM_TEST // BATCH_SIZE\n    test_preds_fold = np.zeros((NUM_TEST,1))\n    for t in range(TTA):\n        preds_t = model.predict(test_ds, steps=ct_test, verbose=1)\n        test_preds_fold += preds_t/TTA\n    test_preds += test_preds_fold/FOLDS\n\n# Compute OOF AUC\noof_preds = np.concatenate(oof_preds)\noof_targets = np.concatenate(oof_targets)\nauc = roc_auc_score(oof_targets, oof_preds)\nprint('OOF AUC:', auc)\n\n# Create submission\ntest_ds_final = get_test_dataset(files_test, BATCH_SIZE, repeat=False, augment=False, labeled=False)\nnames = []\nfor img, name in test_ds_final.unbatch():\n    names.append(name.numpy().decode(\"utf-8\"))\n\nsubmission = pd.DataFrame({'image_name': names, 'target': test_preds.ravel()})\nsubmission.sort_values('image_name', inplace=True)\nsubmission.to_csv('submission.csv', index=False)\nprint(submission.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T13:49:41.097847Z","iopub.execute_input":"2024-12-07T13:49:41.098411Z","iopub.status.idle":"2024-12-07T13:50:10.087009Z","shell.execute_reply.started":"2024-12-07T13:49:41.098379Z","shell.execute_reply":"2024-12-07T13:50:10.085342Z"}},"outputs":[],"execution_count":null}]}