{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T12:59:22.099619Z","iopub.execute_input":"2025-12-03T12:59:22.099871Z","iopub.status.idle":"2025-12-03T12:59:24.651312Z","shell.execute_reply.started":"2025-12-03T12:59:22.099851Z","shell.execute_reply":"2025-12-03T12:59:24.650351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install protobuf==3.20.*\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:01:28.689172Z","iopub.execute_input":"2025-12-03T13:01:28.689490Z","iopub.status.idle":"2025-12-03T13:01:32.697115Z","shell.execute_reply.started":"2025-12-03T13:01:28.689459Z","shell.execute_reply":"2025-12-03T13:01:32.696051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport math\nimport re\nimport matplotlib.pyplot as plt\n\n# Tambahan untuk mengatasi Overfitting\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dropout\nfrom tensorflow.keras.callbacks import EarlyStopping\nimport tensorflow.keras.layers as L\nimport tensorflow.keras.models as M\n\n# List files untuk konfirmasi (sama seperti di notebook Anda)\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:02:45.301146Z","iopub.execute_input":"2025-12-03T13:02:45.302723Z","iopub.status.idle":"2025-12-03T13:02:45.326501Z","shell.execute_reply.started":"2025-12-03T13:02:45.302683Z","shell.execute_reply":"2025-12-03T13:02:45.325418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ================================================================\n# 1. IMPORT LIBRARIES & TPU SETUP\n# ================================================================\nAUTO = tf.data.AUTOTUNE\n\ntry:\n    # Mendeteksi dan menginisialisasi TPU\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\nexcept ValueError:\n    print('Not running on TPU. Falling back to default strategy.')\n    strategy = tf.distribute.get_strategy() # Strategi default untuk CPU/GPU\n\nREPLICAS = strategy.num_replicas_in_sync\nprint(f\"REPLICAS: {REPLICAS}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:03:26.923336Z","iopub.execute_input":"2025-12-03T13:03:26.923723Z","iopub.status.idle":"2025-12-03T13:03:26.931602Z","shell.execute_reply.started":"2025-12-03T13:03:26.923695Z","shell.execute_reply":"2025-12-03T13:03:26.930351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ================================================================\n# 2. SET GLOBAL VARIABLES \n# ================================================================\nIMAGE_SIZE = [224, 224]\n# BATCH_SIZE yang optimal untuk TPU: 16 per core (replika)\nBATCH_SIZE = 16 * REPLICAS\nDATA_DIR = \"/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224\"\n\nTRAINING_FILENAMES = tf.io.gfile.glob(DATA_DIR + \"/train/*.tfrec\")\nVALIDATION_FILENAMES = tf.io.gfile.glob(DATA_DIR + \"/val/*.tfrec\")\nTEST_FILENAMES = tf.io.gfile.glob(DATA_DIR + \"/test/*.tfrec\")\n\n# Helper function untuk menghitung jumlah data dari nama file TFRecord\ndef count_data_items(filenames):\n    # Nama file TFRecord biasanya menyertakan jumlah item: e.g., '...-12753.tfrec'\n    n = [int(re.compile(r\"-(\\d+)\\.\").search(filename).group(1)) for filename in filenames]\n    return np.sum(n)\n\nNUM_TRAINING_IMAGES = count_data_items(TRAINING_FILENAMES)\nNUM_VALIDATION_IMAGES = count_data_items(VALIDATION_FILENAMES)\nNUM_TEST_IMAGES = count_data_items(TEST_FILENAMES)\nNUM_CLASSES = 104 # Berdasarkan konteks kompetisi flower classification\n\nprint(\"Train Images :\", NUM_TRAINING_IMAGES)\nprint(\"Val Images   :\", NUM_VALIDATION_IMAGES)\nprint(\"Test Images  :\", NUM_TEST_IMAGES)\n\n# Hitungan langkah untuk pelatihan\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\nVALIDATION_STEPS = NUM_VALIDATION_IMAGES // BATCH_SIZE","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:03:30.453134Z","iopub.execute_input":"2025-12-03T13:03:30.453435Z","iopub.status.idle":"2025-12-03T13:03:30.483180Z","shell.execute_reply.started":"2025-12-03T13:03:30.453413Z","shell.execute_reply":"2025-12-03T13:03:30.482239Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ================================================================\n# 3. TFRecord PARSER\n# ================================================================\n\ndef read_tfrecord(example):\n    features = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64)\n    }\n    example = tf.io.parse_single_example(example, features)\n    \n    image = tf.image.decode_jpeg(example['image'], channels=3)\n    # Resize, normalisasi, dan casting ke float32 (0-1)\n    image = tf.image.resize(image, IMAGE_SIZE) \n    image = tf.cast(image, tf.float32) / 255.0\n    \n    label = tf.cast(example['class'], tf.int32)\n    return image, label\n\ndef read_tfrecord_test(example):\n    features = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string)\n    }\n    example = tf.io.parse_single_example(example, features)\n    \n    image = tf.image.decode_jpeg(example['image'], channels=3)\n    image = tf.image.resize(image, IMAGE_SIZE)\n    image = tf.cast(image, tf.float32) / 255.0\n    \n    image_id = example[\"id\"]\n    return image, image_id","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:03:39.810126Z","iopub.execute_input":"2025-12-03T13:03:39.810428Z","iopub.status.idle":"2025-12-03T13:03:39.818694Z","shell.execute_reply.started":"2025-12-03T13:03:39.810405Z","shell.execute_reply":"2025-12-03T13:03:39.817703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ================================================================\n# 4. DATASET BUILDERS\n# ================================================================\n\ndef load_dataset(files, labeled=True):\n    ignore_order = tf.data.Options()\n    ignore_order.experimental_deterministic = False\n    \n    dataset = tf.data.TFRecordDataset(files, num_parallel_reads=AUTO)\n    dataset = dataset.with_options(ignore_order)\n    \n    if labeled:\n        dataset = dataset.map(read_tfrecord, num_parallel_calls=AUTO)\n    else:\n        dataset = dataset.map(read_tfrecord_test, num_parallel_calls=AUTO)\n    \n    return dataset\n\ndef get_dataset(files, labeled=True, shuffle=False):\n    dataset = load_dataset(files, labeled)\n    \n    if shuffle:\n        # Buffer size yang lebih besar untuk pengacakan yang lebih baik\n        dataset = dataset.shuffle(buffer_size=5000, reshuffle_each_iteration=True)\n    \n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\n# Buat dataset\ntrain_ds = get_dataset(TRAINING_FILENAMES, labeled=True, shuffle=True)\nval_ds   = get_dataset(VALIDATION_FILENAMES, labeled=True, shuffle=False)\ntest_ds  = get_dataset(TEST_FILENAMES, labeled=False, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:04:03.986738Z","iopub.execute_input":"2025-12-03T13:04:03.987082Z","iopub.status.idle":"2025-12-03T13:04:04.133710Z","shell.execute_reply.started":"2025-12-03T13:04:03.987058Z","shell.execute_reply":"2025-12-03T13:04:04.132628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ================================================================\n# 5. BUILD MODEL (EfficientNetB0 dengan Dropout) 🚀\n# ================================================================\n\nwith strategy.scope():\n    base = tf.keras.applications.EfficientNetB0(\n        input_shape=(*IMAGE_SIZE, 3),\n        weights=\"imagenet\",\n        include_top=False\n    )\n    # Kita pertahankan lapisan dasar DIBEKUKAN (Frozen) \n    # karena masalahnya adalah overfitting pada lapisan dense.\n    base.trainable = False \n    \n    model = M.Sequential([\n        base,\n        L.GlobalAveragePooling2D(),\n        # Tambahkan lapisan Dense ekstra untuk pemisahan fitur sebelum output\n        L.Dense(512, activation='relu'), \n        # === PENTING: TAMBAHKAN DROPOUT UNTUK REGULARISASI ===\n        Dropout(0.3), # Rate 0.3 (30%) adalah titik awal yang baik\n        L.Dense(NUM_CLASSES, activation=\"softmax\")\n    ])\n\n    model.compile(\n        optimizer=\"adam\",\n        loss=\"sparse_categorical_crossentropy\",\n        metrics=[\"accuracy\"]\n    )\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:04:07.719306Z","iopub.execute_input":"2025-12-03T13:04:07.719686Z","iopub.status.idle":"2025-12-03T13:04:09.588720Z","shell.execute_reply.started":"2025-12-03T13:04:07.719659Z","shell.execute_reply":"2025-12-03T13:04:09.587869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ================================================================\n# 6. TRAIN MODEL (Dengan Early Stopping) \n# ================================================================\n\nimport tensorflow as tf\nfrom tensorflow.keras.callbacks import EarlyStopping\n\n# Catatan: Pastikan Anda telah menjalankan bagian Dataset Builder (Bagian 4) yang sudah diperbaiki\n# dengan .repeat() sebelum menjalankan bagian ini.\n\n# === 1. DEFINISIKAN CALLBACK EARLY STOPPING ===\n# Pantau Validation Loss. Karena total epoch hanya 6, kita set patience sedikit lebih tinggi \n# agar model memiliki kesempatan penuh untuk berlatih, namun tetap memulihkan bobot terbaik.\nearly_stopping_callback = EarlyStopping(\n    monitor='val_loss', \n    patience=6, # Tunggu 6 epoch tanpa perbaikan (sama dengan total epoch)\n    restore_best_weights=True,\n    verbose=1 \n)\n\n# === 2. SET EPOCH MAKSIMUM ===\n# Diatur menjadi 6 sesuai permintaan Anda.\nEPOCHS = 6 \n\nprint(f\"\\nMemulai Pelatihan dengan Early Stopping (max {EPOCHS} epochs)...\")\n\nhistory = model.fit(\n    train_ds,\n    steps_per_epoch=STEPS_PER_EPOCH,\n    validation_data=val_ds,\n    validation_steps=VALIDATION_STEPS,\n    epochs=EPOCHS,\n    # === TAMBAHKAN CALLBACK ===\n    callbacks=[early_stopping_callback]\n)\n\nprint(\"-\" * 50)\nprint(\"Pelatihan selesai. Model telah dikembalikan ke bobot terbaik (val_loss terendah).\")\nprint(\"-\" * 50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:40:11.112900Z","iopub.status.idle":"2025-12-03T13:40:11.113217Z","shell.execute_reply.started":"2025-12-03T13:40:11.113078Z","shell.execute_reply":"2025-12-03T13:40:11.113093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ================================================================\n# 7. PREDICT TEST & GENERATE SUBMISSION CSV\n# ================================================================\n\nprint(\"\\nMembuat Prediksi...\")\n\n# 1. Ambil gambar saja untuk prediksi\ntest_images_ds = test_ds.map(lambda image, idnum: image)\n\n# 2. Lakukan prediksi\nprobabilities = model.predict(test_images_ds, verbose=1)\npredictions = np.argmax(probabilities, axis=-1)\nprint(f\"Jumlah Prediksi: {len(predictions)}\")\n\n# 3. Ambil ID secara terpisah dan pastikan urutan terjaga\nprint(\"Mengambil ID...\")\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\n\n# Tentukan batch size yang cukup besar untuk mengambil semua ID\nTEST_BATCH_SIZE = NUM_TEST_IMAGES + 10 \ntest_ids = next(iter(test_ids_ds.batch(TEST_BATCH_SIZE))).numpy().astype('U') \nprint(f\"Jumlah ID: {len(test_ids)}\")\n\n# 4. Cek dan Simpan Submission\nif len(test_ids) != len(predictions):\n    print(f\"ERROR: Mismatch! ID: {len(test_ids)}, Prediksi: {len(predictions)}\")\nelse:\n    submission = pd.DataFrame({'id': test_ids, 'label': predictions})\n    submission.to_csv('submission.csv', index=False)\n    print(\"\\nSukses! File submission.csv berhasil disimpan.\")\n    print(\"5 baris pertama submission:\")\n    print(submission.head())\n\nprint(\"-\" * 50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:39:30.192900Z","iopub.execute_input":"2025-12-03T13:39:30.193669Z","iopub.status.idle":"2025-12-03T13:40:11.111881Z","shell.execute_reply.started":"2025-12-03T13:39:30.193629Z","shell.execute_reply":"2025-12-03T13:40:11.109942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ================================================================\n# 8. Plot Grafik (Perlu import matplotlib.pyplot sebagai plt di awal)\n# ================================================================\nif 'history' in locals():\n    # Grafik Loss (Semakin turun semakin baik)\n    plt.figure(figsize=(10, 5))\n    plt.plot(history.history['loss'], label='Training Loss')\n    plt.plot(history.history['val_loss'], label='Validation Loss')\n    plt.title('Loss Curve')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend()\n    plt.show() # Tampilkan jika di Jupyter/Kaggle Notebook\n    plt.savefig('loss_curve_new.png') \n    plt.close()\n\n    # Grafik Akurasi (Semakin naik semakin baik)\n    plt.figure(figsize=(10, 5))\n    plt.plot(history.history['accuracy'], label='Training Acc')\n    plt.plot(history.history['val_accuracy'], label='Validation Acc')\n    plt.title('Accuracy Curve')\n    plt.xlabel('Epoch')\n    plt.ylabel('Accuracy')\n    plt.legend()\n    plt.show() # Tampilkan jika di Jupyter/Kaggle Notebook\n    plt.savefig('accuracy_curve_new.png') \n    plt.close()\n\n    print(\"\\nGrafik Loss dan Akurasi telah dihasilkan.\")\nelse:\n    print(\"\\nVariabel history tidak ditemukan. Pastikan pelatihan (Cell 6) telah dijalankan.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}