{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import library utama\nimport tensorflow as tf\nimport tensorflow.keras.layers as L\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.applications import EfficientNetB3\nimport numpy as np\n\n# Deteksi TPU / GPU\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    strategy = tf.distribute.get_strategy()\nprint('Number of replicas:', strategy.num_replicas_in_sync)\n\n# Parameter\nIMAGE_SIZE = [512, 512]\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nEPOCHS = 5\nNUM_CLASSES = 104\n\n# Jumlah data tetap untuk dataset ini\nTRAIN_IMAGES = 12752\nVAL_IMAGES = 3712\nTEST_IMAGES = 7382\n\n# Hitung steps\nsteps_per_epoch = TRAIN_IMAGES // BATCH_SIZE\nvalidation_steps = VAL_IMAGES // BATCH_SIZE\n\nprint(f\"Steps per epoch: {steps_per_epoch}\")\nprint(f\"Validation steps: {validation_steps}\")\nprint(f\"Batch size: {BATCH_SIZE}\")\n\n# Path dataset\nGCS_PATH = '/kaggle/input/tpu-getting-started'\nTRAIN_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/tfrecords-jpeg-512x512/train/*.tfrec')\nVAL_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/tfrecords-jpeg-512x512/val/*.tfrec')\nTEST_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/tfrecords-jpeg-512x512/test/*.tfrec')\n\n# Decode image\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0\n    image = tf.image.resize(image, IMAGE_SIZE)\n    return image\n\ndef read_labeled_tfrecord(example):\n    tfrecord_format = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, tfrecord_format)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    tfrecord_format = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, tfrecord_format)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\n# Augmentation\ndef data_augment(image, label):\n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_brightness(image, 0.1)\n    image = tf.image.random_contrast(image, 0.9, 1.1)\n    image = tf.image.random_saturation(image, 0.8, 1.2)\n    return image, label\n\n# Load dataset\ndef load_dataset(filenames, labeled=True):\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=tf.data.AUTOTUNE)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord,\n                          num_parallel_calls=tf.data.AUTOTUNE)\n    if labeled:\n        dataset = dataset.map(data_augment, num_parallel_calls=tf.data.AUTOTUNE)\n    return dataset\n\ntrain_ds = load_dataset(TRAIN_FILENAMES, labeled=True).shuffle(2048).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\nval_ds = load_dataset(VAL_FILENAMES, labeled=True).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\ntest_ds = load_dataset(TEST_FILENAMES, labeled=False).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\n# Bangun model\nwith strategy.scope():\n    base_model = EfficientNetB3(input_shape=(*IMAGE_SIZE, 3),\n                                include_top=False,\n                                weights='imagenet')\n    base_model.trainable = False  # Head-only training\n\n    model = Sequential([\n        base_model,\n        L.GlobalAveragePooling2D(),\n        L.Dropout(0.3),\n        L.Dense(NUM_CLASSES, activation='softmax')\n    ])\n\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4),\n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\nmodel.summary()\n\n# Training dengan steps yang eksplisit → epoch konsisten!\nhistory = model.fit(\n    train_ds,\n    steps_per_epoch=steps_per_epoch,\n    validation_data=val_ds,\n    validation_steps=validation_steps,\n    epochs=EPOCHS\n)\n\n# Prediksi test set\nprint('Computing predictions...')\ntest_images_ds = test_ds.map(lambda image, idnum: image)\nprobabilities = model.predict(test_images_ds, verbose=1)\npredictions = np.argmax(probabilities, axis=1)\n\n# Ambil ID test secara dinamis (paling aman)\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\ntest_ids = np.array([id.numpy().decode('utf-8') for id in test_ids_ds])\n\n# Buat submission\nnp.savetxt(\n    'submission.csv',\n    np.rec.fromarrays([test_ids, predictions]),\n    fmt=['%s', '%d'],\n    delimiter=',',\n    header='id,label',\n    comments='',\n)\n\nprint(f'Submission generated! Total predictions: {len(predictions)}')\n!head submission.csv","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T16:10:38.678191Z","iopub.execute_input":"2025-12-03T16:10:38.678884Z","execution_failed":"2025-12-03T16:17:39.304Z"}},"outputs":[],"execution_count":null}]}