{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":7251,"sourceType":"datasetVersion","datasetId":2798},{"sourceId":13637649,"sourceType":"datasetVersion","datasetId":8668593}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv) \n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:58:30.836189Z","iopub.execute_input":"2025-12-03T13:58:30.836462Z","iopub.status.idle":"2025-12-03T13:58:32.424342Z","shell.execute_reply.started":"2025-12-03T13:58:30.836434Z","shell.execute_reply":"2025-12-03T13:58:32.423735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# improved_training_tpu_ready.py\n# Run in Kaggle / Colab. Adjust DS_PATH and NUM_TRAINING_IMAGES accordingly.\n\nimport os\nimport math\nimport logging\nimport warnings\nimport numpy as np\nimport tensorflow as tf\nfrom matplotlib import pyplot as plt\nfrom sklearn.metrics import (\n    accuracy_score, precision_score, recall_score, f1_score,\n    roc_auc_score, classification_report, confusion_matrix\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:58:32.425678Z","iopub.execute_input":"2025-12-03T13:58:32.426024Z","iopub.status.idle":"2025-12-03T13:58:49.376444Z","shell.execute_reply.started":"2025-12-03T13:58:32.426005Z","shell.execute_reply":"2025-12-03T13:58:49.375869Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# BAGIAN 1: KONFIGURASI TPU/GPU & DATASET","metadata":{}},{"cell_type":"code","source":"print(\"\\n\" + \"=\"*60)\nprint(\"GPU INFORMATION\")\nprint(\"=\"*60)\n\n# Cek GPU tersedia\ngpus = tf.config.list_physical_devices('GPU')\nprint(f\"Number of GPUs Available: {len(gpus)}\")\n\nfor i, gpu in enumerate(gpus):\n    print(f\"  GPU {i}: {gpu}\")\n    \n# Cek apakah TensorFlow menggunakan GPU\nif len(gpus) > 0:\n    print(\"\\n✅ TensorFlow WILL use GPU for training!\")\n    # Set memory growth untuk menghindari OOM error\n    for gpu in gpus:\n        tf.config.experimental.set_memory_growth(gpu, True)\nelse:\n    print(\"\\n⚠️ WARNING: No GPU detected! Training will use CPU (very slow)\")\n\nprint(\"=\"*60 + \"\\n\")\n\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '2'\nwarnings.filterwarnings('ignore')\nlogging.getLogger('tensorflow').setLevel(logging.ERROR)\n\nprint(\"TensorFlow:\", tf.__version__)\nprint(\"=\" * 60)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:58:49.377236Z","iopub.execute_input":"2025-12-03T13:58:49.377756Z","iopub.status.idle":"2025-12-03T13:58:50.01345Z","shell.execute_reply.started":"2025-12-03T13:58:49.377729Z","shell.execute_reply":"2025-12-03T13:58:50.012661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"try:\n    resolver = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print(\"Found TPU:\", resolver.cluster_spec().as_dict() if resolver.cluster_spec() else \"TPU resolver ok\")\n    tf.config.experimental_connect_to_cluster(resolver)\n    tf.tpu.experimental.initialize_tpu_system(resolver)\n    strategy = tf.distribute.TPUStrategy(resolver)\n    print(\"✅ Using TPU with TPUStrategy\")\nexcept Exception as e:\n    print(\"TPU not available:\", e)\n    gpus = tf.config.list_physical_devices('GPU')\n    if gpus:\n        strategy = tf.distribute.MirroredStrategy()\n        print(\"✅ Using MirroredStrategy with GPUs (num_replicas_in_sync):\", strategy.num_replicas_in_sync)\n    else:\n        strategy = tf.distribute.get_strategy()\n        print(\"⚠️ Using default strategy (CPU)\")\n        \nBATCH_SIZE_PER_REPLICA = 16\nBATCH_SIZE = BATCH_SIZE_PER_REPLICA * max(1, strategy.num_replicas_in_sync)\nIMAGE_SIZE = [224, 224]\nDS_PATH = '/kaggle/input/tpu-getting-started'   # change if needed\nNUM_CLASSES = 104    # your dataset classes\nNUM_TRAINING_IMAGES = 12753   # please set to actual\nSTEPS_PER_EPOCH = max(1, NUM_TRAINING_IMAGES // BATCH_SIZE)\nEPOCHS = 25\n\nprint(f\"BATCH_SIZE: {BATCH_SIZE}, STEPS_PER_EPOCH: {STEPS_PER_EPOCH}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:58:50.014407Z","iopub.execute_input":"2025-12-03T13:58:50.014725Z","iopub.status.idle":"2025-12-03T13:58:50.497722Z","shell.execute_reply.started":"2025-12-03T13:58:50.014697Z","shell.execute_reply":"2025-12-03T13:58:50.497025Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# BAGIAN 2: TFRecord parsing & preprocessing","metadata":{}},{"cell_type":"code","source":"def decode_image(image_data, size=IMAGE_SIZE):\n    \"\"\"Decode jpeg bytes -> float32 image in range [0,255], resized.\"\"\"\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.image.convert_image_dtype(image, dtype=tf.float32)  # [0,1]\n    image = tf.image.resize(image, size)\n    image = image * 255.0  # keep consistent with Keras preprocess_input expectations\n    return image\n\ndef read_labeled_tfrecord(example):\n    fmt = {\"image\": tf.io.FixedLenFeature([], tf.string),\n           \"class\": tf.io.FixedLenFeature([], tf.int64)}\n    ex = tf.io.parse_single_example(example, fmt)\n    image = decode_image(ex['image'])\n    label = tf.cast(ex['class'], tf.int32)\n    return image, label\n\ndef read_test_tfrecord(example):\n    fmt = {\"image\": tf.io.FixedLenFeature([], tf.string),\n           \"id\": tf.io.FixedLenFeature([], tf.string)}\n    ex = tf.io.parse_single_example(example, fmt)\n    image = decode_image(ex['image'])\n    idnum = ex['id']\n    return image, idnum\n\n# conservative augmentations (avoid extreme transforms for MLP rubric)\ndef augment(image, label):\n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_brightness(image, max_delta=0.08)\n    image = tf.image.random_contrast(image, 0.95, 1.05)\n    # clip to [0,255] (preprocess_input will handle further mapping)\n    image = tf.clip_by_value(image, 0.0, 255.0)\n    return image, label\n\ndef load_dataset(filenames, labeled=True, ordered=False, augment_data=False, preprocess_fn=None):\n    options = tf.data.Options()\n    if not ordered:\n        options.experimental_deterministic = False\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=tf.data.AUTOTUNE)\n    dataset = dataset.with_options(options)\n    if labeled:\n        dataset = dataset.map(read_labeled_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)\n    else:\n        dataset = dataset.map(read_test_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)\n\n    if augment_data:\n        dataset = dataset.map(augment, num_parallel_calls=tf.data.AUTOTUNE)\n\n    if preprocess_fn is not None:\n        # apply model-specific preprocess (expects [0,255] float inputs)\n        def _apply_preproc(image, label_or_id):\n            img = preprocess_fn(image)\n            return img, label_or_id\n        dataset = dataset.map(_apply_preproc, num_parallel_calls=tf.data.AUTOTUNE)\n\n    return dataset\n\n# ---------- Find TFRecord files ----------\nFILENAMES_TRAIN = tf.io.gfile.glob(os.path.join(DS_PATH, 'tfrecords-jpeg-224x224/train/*.tfrec'))\nFILENAMES_VAL   = tf.io.gfile.glob(os.path.join(DS_PATH, 'tfrecords-jpeg-224x224/val/*.tfrec'))\nFILENAMES_TEST  = tf.io.gfile.glob(os.path.join(DS_PATH, 'tfrecords-jpeg-224x224/test/*.tfrec'))\n\nif len(FILENAMES_TRAIN) == 0:\n    print(\"⚠️ No train tfrec files found at\", os.path.join(DS_PATH, 'tfrecords-jpeg-224x224/train/*.tfrec'))\nelse:\n    print(f\"Found {len(FILENAMES_TRAIN)} train tfrec files.\")\n\n# ---------- Choose base model and its preprocess function ----------\nbase_model = None\npreprocess_fn = None\n\nwith strategy.scope():\n    # Try VGG16 first, fallback to MobileNetV2, else simple CNN\n    try:\n        print(\"Trying VGG16 (imagenet weights)...\")\n        base_model = tf.keras.applications.VGG16(weights='imagenet', include_top=False,\n                                                 input_shape=(*IMAGE_SIZE, 3))\n        preprocess_fn = tf.keras.applications.vgg16.preprocess_input\n        print(\"✅ VGG16 available.\")\n    except Exception as e:\n        print(\"VGG16 not available:\", e)\n        try:\n            print(\"Trying MobileNetV2...\")\n            base_model = tf.keras.applications.MobileNetV2(weights='imagenet', include_top=False,\n                                                           input_shape=(*IMAGE_SIZE, 3))\n            preprocess_fn = tf.keras.applications.mobilenet_v2.preprocess_input\n            print(\"✅ MobileNetV2 available.\")\n        except Exception as e2:\n            print(\"Pretrained models failed. Using custom CNN backbone.\")\n            base_model = None\n            preprocess_fn = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:58:50.498498Z","iopub.execute_input":"2025-12-03T13:58:50.49876Z","iopub.status.idle":"2025-12-03T13:58:52.483784Z","shell.execute_reply.started":"2025-12-03T13:58:50.498731Z","shell.execute_reply":"2025-12-03T13:58:52.483156Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Builds Datasets & Models","metadata":{}},{"cell_type":"code","source":"train_ds = load_dataset(FILENAMES_TRAIN, labeled=True, ordered=False,\n                        augment_data=True, preprocess_fn=preprocess_fn)\ntrain_ds = train_ds.shuffle(2048).repeat().batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\nval_ds = load_dataset(FILENAMES_VAL, labeled=True, ordered=True,\n                      augment_data=False, preprocess_fn=preprocess_fn)\nval_ds = val_ds.batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\nprint(\"Datasets prepared.\")\n\nwith strategy.scope():\n    if base_model is None:\n        # custom small CNN backbone (pretrained not available)\n        inputs = tf.keras.Input(shape=(*IMAGE_SIZE, 3))\n        x = tf.keras.layers.Conv2D(64, 3, activation='relu', padding='same')(inputs)\n        x = tf.keras.layers.MaxPooling2D(2)(x)\n        x = tf.keras.layers.Conv2D(128, 3, activation='relu', padding='same')(x)\n        x = tf.keras.layers.MaxPooling2D(2)(x)\n        x = tf.keras.layers.Conv2D(256, 3, activation='relu', padding='same')(x)\n        x = tf.keras.layers.GlobalAveragePooling2D()(x)\n        # MLP head (rubric: deep MLP + dropout + BN)\n        x = tf.keras.layers.Dense(1024, kernel_regularizer=tf.keras.regularizers.l2(1e-3))(x)\n        x = tf.keras.layers.BatchNormalization()(x)\n        x = tf.keras.layers.Activation('relu')(x)\n        x = tf.keras.layers.Dropout(0.5)(x)\n        x = tf.keras.layers.Dense(512, kernel_regularizer=tf.keras.regularizers.l2(1e-3))(x)\n        x = tf.keras.layers.BatchNormalization()(x)\n        x = tf.keras.layers.Activation('relu')(x)\n        x = tf.keras.layers.Dropout(0.3)(x)\n        outputs = tf.keras.layers.Dense(NUM_CLASSES, activation='softmax')(x)\n        model = tf.keras.Model(inputs, outputs)\n    else:\n        # transfer learning backbone\n        base_model.trainable = False\n        inputs = tf.keras.Input(shape=(*IMAGE_SIZE, 3))\n        x = base_model(inputs, training=False)\n        x = tf.keras.layers.GlobalAveragePooling2D()(x)\n        x = tf.keras.layers.Dense(2048, kernel_regularizer=tf.keras.regularizers.l2(1e-3))(x)\n        x = tf.keras.layers.BatchNormalization()(x)\n        x = tf.keras.layers.Activation('relu')(x)\n        x = tf.keras.layers.Dropout(0.5)(x)\n        x = tf.keras.layers.Dense(1024, kernel_regularizer=tf.keras.regularizers.l2(1e-3))(x)\n        x = tf.keras.layers.BatchNormalization()(x)\n        x = tf.keras.layers.Activation('relu')(x)\n        x = tf.keras.layers.Dropout(0.4)(x)\n        x = tf.keras.layers.Dense(512, kernel_regularizer=tf.keras.regularizers.l2(1e-3))(x)\n        x = tf.keras.layers.BatchNormalization()(x)\n        x = tf.keras.layers.Activation('relu')(x)\n        x = tf.keras.layers.Dropout(0.3)(x)\n        outputs = tf.keras.layers.Dense(NUM_CLASSES, activation='softmax')(x)\n        model = tf.keras.Model(inputs, outputs)\n\n    optimizer = tf.keras.optimizers.Adam(learning_rate=1e-4)\n    model.compile(optimizer=optimizer,\n                  loss='sparse_categorical_crossentropy',\n                  metrics=['sparse_categorical_accuracy'])\n\n    model.summary()\n\n# ---------- Callbacks ----------\ncheckpoint_path = 'best_model.keras'\ncallbacks = [\n    tf.keras.callbacks.ModelCheckpoint(checkpoint_path, save_best_only=True,\n                                       monitor='val_sparse_categorical_accuracy', mode='max', verbose=1),\n    tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=3, min_lr=1e-7, verbose=1),\n    tf.keras.callbacks.EarlyStopping(monitor='val_sparse_categorical_accuracy', patience=7,\n                                     restore_best_weights=True, verbose=1)\n]\n\n# ---------- Train ----------\nprint(\"Starting training...\")\nhistory = model.fit(train_ds,\n                    steps_per_epoch=STEPS_PER_EPOCH,\n                    epochs=EPOCHS,\n                    validation_data=val_ds,\n                    callbacks=callbacks)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:58:52.484508Z","iopub.execute_input":"2025-12-03T13:58:52.484768Z","execution_failed":"2025-12-03T14:13:12.492Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Convergence analysis","metadata":{}},{"cell_type":"code","source":"def plot_history(history):\n    acc = history.history['sparse_categorical_accuracy']\n    val_acc = history.history['val_sparse_categorical_accuracy']\n    loss = history.history['loss']\n    val_loss = history.history['val_loss']\n    epochs = range(1, len(acc) + 1)\n\n    plt.figure(figsize=(16, 5))\n    # Accuracy\n    plt.subplot(1, 3, 1)\n    plt.plot(epochs, acc, label='train', marker='o')\n    plt.plot(epochs, val_acc, label='val', marker='s')\n    plt.title('Accuracy vs Epoch (Convergence)')\n    plt.xlabel('Epoch'); plt.ylabel('Accuracy')\n    plt.legend(); plt.grid(True)\n\n    # Loss\n    plt.subplot(1, 3, 2)\n    plt.plot(epochs, loss, label='train', marker='o')\n    plt.plot(epochs, val_loss, label='val', marker='s')\n    plt.title('Loss vs Epoch (Overfitting detection)')\n    plt.xlabel('Epoch'); plt.ylabel('Loss')\n    plt.legend(); plt.grid(True)\n\n    # Gap\n    plt.subplot(1, 3, 3)\n    gap = np.array(acc) - np.array(val_acc)\n    plt.plot(epochs, gap, marker='D')\n    plt.axhline(0, linestyle='--', color='k')\n    plt.title('Train-Val Gap (Accuracy)')\n    plt.xlabel('Epoch'); plt.ylabel('Gap (train - val)')\n    plt.grid(True)\n\n    plt.tight_layout()\n    plt.show()\n\n    # numeric summary\n    final_train_acc = acc[-1]\n    final_val_acc = val_acc[-1]\n    acc_gap = final_train_acc - final_val_acc\n\n    print(\"\\nConvergence summary:\")\n    print(f\" Final train acc: {final_train_acc:.4f}, val acc: {final_val_acc:.4f}, gap: {acc_gap:.4f}\")\n    best_epoch = int(np.argmax(val_acc) + 1)\n    print(f\" Best val acc: {max(val_acc):.4f} at epoch {best_epoch}\")\n    min_val_loss = min(val_loss)\n    min_val_loss_epoch = int(np.argmin(val_loss) + 1)\n    print(f\" Min val loss: {min_val_loss:.4f} at epoch {min_val_loss_epoch}\")\n    if acc_gap > 0.10:\n        print(\" ⚠️ OVERFITTING DETECTED (gap > 0.10). Consider stronger regularization.\")\n    elif acc_gap < -0.05:\n        print(\" ⚠️ UNDERFITTING DETECTED (train < val by >5%). Consider larger model / more training.\")\n    else:\n        print(\" ✅ Good fit according to gap heuristic.\")\n\nplot_history(history)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-12-03T14:13:12.493Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. Evaluation & rubric metrics","metadata":{}},{"cell_type":"code","source":"print(\"\\nLoading best weights from checkpoint for evaluation...\")\nmodel.load_weights(checkpoint_path)\n\n# Collect predictions on the entire validation set\ny_true = []\ny_pred = []\ny_pred_proba = []\n\nprint(\"Predicting validation set...\")\nfor batch in val_ds:\n    images, labels = batch\n    probs = model.predict(images, verbose=0)\n    preds = np.argmax(probs, axis=-1)\n    y_true.extend(labels.numpy())\n    y_pred.extend(preds)\n    y_pred_proba.extend(probs)\n\ny_true = np.array(y_true)\ny_pred = np.array(y_pred)\ny_pred_proba = np.array(y_pred_proba)\n\n# Metrics\nprint(\"\\n=== Evaluation Metrics (Rubric) ===\")\nacc = accuracy_score(y_true, y_pred)\nprec = precision_score(y_true, y_pred, average='macro', zero_division=0)\nrec = recall_score(y_true, y_pred, average='macro', zero_division=0)\nf1 = f1_score(y_true, y_pred, average='macro', zero_division=0)\nprint(f\" Accuracy: {acc*100:.2f}%\")\nprint(f\" Precision (macro): {prec*100:.2f}%\")\nprint(f\" Recall (macro): {rec*100:.2f}%\")\nprint(f\" F1-score (macro): {f1*100:.2f}%\")\n\n# Multi-class ROC-AUC (OvR) if possible\ntry:\n    from sklearn.preprocessing import label_binarize\n    y_true_bin = label_binarize(y_true, classes=range(NUM_CLASSES))\n    auc = roc_auc_score(y_true_bin, y_pred_proba, multi_class='ovr', average='macro')\n    print(f\" ROC-AUC (OvR macro): {auc*100:.2f}%\")\nexcept Exception as e:\n    print(\" ROC-AUC skipped (possible memory or shape issue):\", e)\n\nprint(\"\\nDetailed classification report:\")\nprint(classification_report(y_true, y_pred, zero_division=0))\n\n# Confusion matrix (show subset for readability)\ncm = confusion_matrix(y_true, y_pred)\nprint(\"Confusion matrix shape:\", cm.shape)\nprint(\"Total samples:\", np.sum(cm))\nprint(\"Correct predictions:\", np.trace(cm))\nprint(\"Wrong predictions:\", np.sum(cm) - np.trace(cm))\n\n# plot confusion matrix subset\nplt.figure(figsize=(10, 8))\nn_show = min(20, cm.shape[0])\ncm_sub = cm[:n_show, :n_show]\nplt.imshow(cm_sub, interpolation='nearest', cmap=plt.cm.Blues)\nplt.title('Confusion matrix (first {} classes)'.format(n_show))\nplt.colorbar()\nplt.xlabel('Predicted'); plt.ylabel('True')\nplt.tight_layout(); plt.show()\n\n# ---------- Create submission ----------\nprint(\"\\nCreating submission.csv from test set (if test TFRecord available)...\")\nif len(FILENAMES_TEST) == 0:\n    print(\"No test files found; skipping submission generation.\")\nelse:\n    test_ds = tf.data.TFRecordDataset(FILENAMES_TEST, num_parallel_reads=tf.data.AUTOTUNE)\n    # re-use read_test_tfrecord and preprocess_fn\n    def _parse_and_preproc(x):\n        image, idnum = read_test_tfrecord(x)\n        if preprocess_fn is not None:\n            image = preprocess_fn(image)\n        return image, idnum\n\n    test_ds = test_ds.map(_parse_and_preproc, num_parallel_calls=tf.data.AUTOTUNE)\n    test_ds = test_ds.batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\n    test_ids = []\n    test_preds = []\n    for images, idnums in test_ds:\n        probs = model.predict(images, verbose=0)\n        preds = np.argmax(probs, axis=-1)\n        test_preds.extend(preds.tolist())\n        # idnums are bytes strings\n        test_ids.extend([i.decode('utf-8') for i in idnums.numpy()])\n\n    import csv\n    with open('submission.csv', 'w', newline='') as f:\n        writer = csv.writer(f)\n        writer.writerow(['id', 'label'])\n        for i, lab in zip(test_ids, test_preds):\n            writer.writerow([i, int(lab)])\n\n    print(\"submission.csv saved. Total test samples:\", len(test_ids))\n\nprint(\"\\nAll done. See convergence plots and metrics above — they map directly to the rubric items:\\n\")\n\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-12-03T14:13:12.493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('\\n📝 Creating submission file...')\n\ndef read_test_tfrecord(example):\n    TEST_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, TEST_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\nFILENAMES_TEST = tf.io.gfile.glob(DS_PATH + '/tfrecords-jpeg-224x224/test/*.tfrec')\ntest_dataset = (\n    tf.data.TFRecordDataset(FILENAMES_TEST, num_parallel_reads=tf.data.AUTOTUNE)\n    .map(read_test_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)\n    .batch(BATCH_SIZE)\n    .prefetch(tf.data.AUTOTUNE)\n)\n\ntest_ids = []\ntest_preds = []\n\nfor image, idnum in test_dataset:\n    test_ids.extend([x.decode('utf-8') for x in idnum.numpy()])\n    probs = model.predict(image, verbose=0)\n    test_preds.extend(np.argmax(probs, axis=-1))\n\n# Write to CSV\nimport csv\nwith open('submission.csv', 'w', newline='') as f:\n    writer = csv.writer(f)\n    writer.writerow([\"id\", \"label\"])\n    for i in range(len(test_ids)):\n        writer.writerow([test_ids[i], test_preds[i]])\n\nprint(\"✅ SUCCESS! File 'submission.csv' is ready for submission.\")\nprint(f\"📊 Total test samples: {len(test_ids)}\")","metadata":{"trusted":true,"execution":{"execution_failed":"2025-12-03T14:13:12.493Z"}},"outputs":[],"execution_count":null}]}