{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31194,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import math, re, os\nimport numpy as np\nimport tensorflow as tf\nfrom kaggle_datasets import KaggleDatasets\nfrom matplotlib import pyplot as plt\nfrom sklearn.metrics import f1_score, precision_score, recall_score, accuracy_score, roc_auc_score\nfrom sklearn.preprocessing import LabelBinarizer\n\nprint(\"Tensorflow version \" + tf.__version__)\n\n# --- DETEKSI OTOMATIS (TPU / GPU / CPU) ---\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() \n    print('✅ Device:', tpu.master())\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\n    DEVICE = \"TPU\"\nexcept ValueError:\n    gpus = tf.config.list_physical_devices('GPU')\n    if gpus:\n        strategy = tf.distribute.MirroredStrategy()\n        DEVICE = \"GPU\"\n        print(f'✅ Running on {len(gpus)} GPU(s).')\n    else:\n        strategy = tf.distribute.get_strategy()\n        DEVICE = \"CPU\"\n        print('⚠️ Running on CPU (Lambat).')\n\nprint(f\"REPLICAS: {strategy.num_replicas_in_sync}\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:13:47.750618Z","iopub.execute_input":"2025-12-03T11:13:47.750881Z","iopub.status.idle":"2025-12-03T11:13:47.756473Z","shell.execute_reply.started":"2025-12-03T11:13:47.750862Z","shell.execute_reply":"2025-12-03T11:13:47.755600Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- HYPERPARAMETERS ---\nEPOCHS = 10 \n\n# Kita kunci BATCH_SIZE jadi 16 supaya steps-nya sekitar 797\nBATCH_SIZE = 16 \n\n# Sesuaikan ukuran gambar (Jangan terlalu besar biar GPU kuat angkat 16 gambar sekaligus)\nif DEVICE == \"TPU\":\n    IMAGE_SIZE = [512, 512] \nelif DEVICE == \"GPU\":\n    IMAGE_SIZE = [224, 224] # 224 aman untuk Batch 16 di GPU\nelse:\n    IMAGE_SIZE = [192, 192]\n\nprint(f\"Set: Epochs={EPOCHS}, Image={IMAGE_SIZE}, Batch={BATCH_SIZE}\")\nprint(f\"Estimasi Steps per Epoch: {12753 // BATCH_SIZE} (Target: ~797)\")\n\n# --- AKSES DATA ---\ntry:\n    if DEVICE == \"TPU\":\n        GCS_DS_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started')\n        DATA_PATH = GCS_DS_PATH + f'/tfrecords-jpeg-{IMAGE_SIZE[0]}x{IMAGE_SIZE[0]}'\n    else:\n        DATA_PATH = f'/kaggle/input/tpu-getting-started/tfrecords-jpeg-{IMAGE_SIZE[0]}x{IMAGE_SIZE[0]}'\n    \n    TRAINING_FILENAMES = tf.io.gfile.glob(DATA_PATH + '/train/*.tfrec')\n    VALIDATION_FILENAMES = tf.io.gfile.glob(DATA_PATH + '/val/*.tfrec')\n    TEST_FILENAMES = tf.io.gfile.glob(DATA_PATH + '/test/*.tfrec')\n\n    # Fallback jika data tidak ketemu\n    if not TRAINING_FILENAMES:\n        print(\"⚠️ Data spesifik tidak ketemu, mencoba ukuran 224x224...\")\n        if DEVICE == \"TPU\":\n             DATA_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started') + '/tfrecords-jpeg-224x224'\n        else:\n             DATA_PATH = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224'\n        TRAINING_FILENAMES = tf.io.gfile.glob(DATA_PATH + '/train/*.tfrec')\n        VALIDATION_FILENAMES = tf.io.gfile.glob(DATA_PATH + '/val/*.tfrec')\n        TEST_FILENAMES = tf.io.gfile.glob(DATA_PATH + '/test/*.tfrec')\n        IMAGE_SIZE = [224, 224]\n\n    print(f\"✅ Data Siap! Jumlah File Train: {len(TRAINING_FILENAMES)}\")\n\nexcept Exception as e:\n    print(f\"❌ Error Path: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:13:49.218973Z","iopub.execute_input":"2025-12-03T11:13:49.219195Z","iopub.status.idle":"2025-12-03T11:13:49.250295Z","shell.execute_reply.started":"2025-12-03T11:13:49.219180Z","shell.execute_reply":"2025-12-03T11:13:49.249394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CLASSES = [\n    'pink primrose', 'hard-leaved pocket orchid', 'canterbury bells', 'sweet pea', 'english marigold',\n    'tiger lily', 'moon orchid', 'bird of paradise', 'monkshood', 'globe thistle',\n    'snapdragon', \"colt's foot\", 'king protea', 'spear thistle', 'yellow iris',\n    'globe-flower', 'purple coneflower', 'peruvian lily', 'balloon flower', 'giant white arum lily',\n    'fire lily', 'pincushion flower', 'fritillary', 'red ginger', 'grape hyacinth',\n    'corn poppy', 'prince of wales feathers', 'stemless gentian', 'artichoke', 'sweet william',\n    'carnation', 'garden phlox', 'love in the mist', 'mexican aster', 'alpine sea holly',\n    'ruby-lipped cattleya', 'cape flower', 'great masterwort', 'siam tulip', 'lenten rose',\n    'barberton daisy', 'daffodil', 'sword lily', 'poinsettia', 'bolero deep blue',\n    'wallflower', 'marigold', 'buttercup', 'daisy', 'common dandelion',\n    'petunia', 'wild pansy', 'primula', 'sunflower', 'lilac hibiscus',\n    'bishop of llandaff', 'gaillardia', 'gazania', 'azalea', 'water lily',\n    'rose', 'thorn apple', 'morning glory', 'passion flower', 'lotus',\n    'toad lily', 'anthurium', 'frangipani', 'clematis', 'hibiscus',\n    'columbine', 'desert-rose', 'tree mallow', 'magnolia', 'cyclamen ',\n    'watercress', 'canna lily', 'hippeastrum ', 'bee balm', 'pink quill',\n    'foxglove', 'bougainvillea', 'camellia', 'mallow', 'mexican petunia',\n    'bromelia', 'blanket flower', 'trumpet creeper', 'blackberry lily', 'common tulip',\n    'wild rose', \n    'silverbush', 'californian poppy', 'osteospermum', 'spring crocus', 'bearded iris',\n    'windflower', 'tree poppy', 'gazania', 'azalea', 'water lily', 'rose', 'thorn apple', \n    'morning glory'\n]\n\n# Safety Check (Memastikan list pas 104)\nrequired_length = 104\nif len(CLASSES) < required_length:\n    for i in range(required_length - len(CLASSES)): \n        CLASSES.append(f\"flower_class_{len(CLASSES)}\")\nCLASSES = CLASSES[:required_length]\nprint(f\"Jumlah Kelas Final: {len(CLASSES)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:13:52.117033Z","iopub.execute_input":"2025-12-03T11:13:52.117298Z","iopub.status.idle":"2025-12-03T11:13:52.122516Z","shell.execute_reply.started":"2025-12-03T11:13:52.117279Z","shell.execute_reply":"2025-12-03T11:13:52.121693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\n\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    # RUBRIK: PREPROCESSING (Normalisasi 0-1)\n    image = tf.cast(image, tf.float32) / 255.0  \n    image = tf.reshape(image, [*IMAGE_SIZE, 3])\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\n# Augmentasi Data (Penting untuk Skor Tinggi)\ndef data_augment(image, label):\n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_saturation(image, 0, 2)\n    return image, label\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False \n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, num_parallel_calls=AUTO)\n    return dataset\n\n# Pipeline\ntraining_dataset = load_dataset(TRAINING_FILENAMES, labeled=True)\ntraining_dataset = training_dataset.map(data_augment, num_parallel_calls=AUTO)\ntraining_dataset = training_dataset.repeat().shuffle(2048).batch(BATCH_SIZE).prefetch(AUTO)\n\nvalidation_dataset = load_dataset(VALIDATION_FILENAMES, labeled=True, ordered=True).batch(BATCH_SIZE).cache().prefetch(AUTO)\ntest_dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=True).batch(BATCH_SIZE).prefetch(AUTO)\n\nprint(\"✅ Dataset Pipeline Siap!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:13:55.254563Z","iopub.execute_input":"2025-12-03T11:13:55.254785Z","iopub.status.idle":"2025-12-03T11:13:55.354950Z","shell.execute_reply.started":"2025-12-03T11:13:55.254768Z","shell.execute_reply":"2025-12-03T11:13:55.353908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Scheduler disesuaikan untuk 10 Epoch\ndef lrfn(epoch):\n    START_LR = 0.00001\n    MAX_LR = 0.00005 * strategy.num_replicas_in_sync\n    MIN_LR = 0.00001\n    RAMPUP_EPOCHS = 3 # Pemanasan cepat\n    SUSTAIN_EPOCHS = 0\n    EXP_DECAY = .8\n    \n    if epoch < RAMPUP_EPOCHS:\n        return (MAX_LR - START_LR) / RAMPUP_EPOCHS * epoch + START_LR\n    elif epoch < RAMPUP_EPOCHS + SUSTAIN_EPOCHS:\n        return MAX_LR\n    else:\n        return (MAX_LR - MIN_LR) * EXP_DECAY**(epoch - RAMPUP_EPOCHS - SUSTAIN_EPOCHS) + MIN_LR\n\nlr_callback = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=1)\n\nwith strategy.scope():\n    # Gunakan EfficientNetB7 (Sangat Bagus) jika TPU\n    if DEVICE == \"TPU\":\n        base_model = tf.keras.applications.EfficientNetB7(weights='imagenet', include_top=False, input_shape=[*IMAGE_SIZE, 3])\n    else:\n        # Fallback ke B0 jika GPU (agar ringan)\n        base_model = tf.keras.applications.EfficientNetB0(weights='imagenet', include_top=False, input_shape=[*IMAGE_SIZE, 3])\n    \n    base_model.trainable = True \n    \n    model = tf.keras.Sequential([\n        base_model,\n        \n        # --- RUBRIK: ARSITEKTUR MLP LENGKAP ---\n        tf.keras.layers.GlobalAveragePooling2D(), \n        \n        # Hidden Layer 1\n        tf.keras.layers.Dense(1024),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Activation('relu'),   \n        tf.keras.layers.Dropout(0.4),\n        \n        # Hidden Layer 2\n        tf.keras.layers.Dense(512),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Activation('relu'),\n        tf.keras.layers.Dropout(0.4),\n        \n        # Output Layer\n        tf.keras.layers.Dense(len(CLASSES), activation='softmax') \n    ])\n    \n    model.compile(\n        optimizer='adam',\n        loss = 'sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:13:57.594097Z","iopub.execute_input":"2025-12-03T11:13:57.594365Z","iopub.status.idle":"2025-12-03T11:13:58.339328Z","shell.execute_reply.started":"2025-12-03T11:13:57.594349Z","shell.execute_reply":"2025-12-03T11:13:58.338392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if len(TRAINING_FILENAMES) > 0:\n    NUM_TRAINING_IMAGES = 12753 \n    STEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\nelse:\n    STEPS_PER_EPOCH = 100 \n\nprint(f\"Mulai Training {EPOCHS} Epoch...\")\nhistory = model.fit(\n    training_dataset, \n    steps_per_epoch=STEPS_PER_EPOCH,\n    epochs=EPOCHS,\n    validation_data=validation_dataset,\n    callbacks=[lr_callback]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:14:00.824886Z","iopub.execute_input":"2025-12-03T11:14:00.825159Z","iopub.status.idle":"2025-12-03T12:21:32.721580Z","shell.execute_reply.started":"2025-12-03T11:14:00.825141Z","shell.execute_reply":"2025-12-03T12:21:32.720437Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 1. GAMBAR GRAFIK ---\ndef plot_curves(history):\n    loss = history.history.get('loss', [])\n    val_loss = history.history.get('val_loss', [])\n    acc = history.history.get('sparse_categorical_accuracy', [])\n    val_acc = history.history.get('val_sparse_categorical_accuracy', [])\n\n    if not loss:\n        print(\"⚠️ Data Loss kosong.\")\n        return\n\n    fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(15, 5))\n    ax1.plot(loss, label='Training Loss'); ax1.plot(val_loss, label='Validation Loss')\n    ax1.set_title('Loss Curve'); ax1.legend(); ax1.grid(True)\n\n    ax2.plot(acc, label='Training Acc'); ax2.plot(val_acc, label='Validation Acc')\n    ax2.set_title('Accuracy Curve'); ax2.legend(); ax2.grid(True)\n    plt.show()\n\nif 'history' in globals(): plot_curves(history)\n\n# --- 2. METRIK LENGKAP ---\nprint(\"\\nMenghitung Metrik (Tunggu sebentar)...\")\ntry:\n    val_labels_ds = validation_dataset.map(lambda image, label: label)\n    val_labels = np.concatenate([y.numpy() for y in val_labels_ds], axis=0)\n    \n    val_probs = model.predict(validation_dataset, verbose=1)\n    val_preds = np.argmax(val_probs, axis=-1)\n    \n    min_len = min(len(val_labels), len(val_preds))\n    val_labels, val_preds = val_labels[:min_len], val_preds[:min_len]\n\n    acc = accuracy_score(val_labels, val_preds)\n    f1 = f1_score(val_labels, val_preds, average='macro')\n    prec = precision_score(val_labels, val_preds, average='macro', zero_division=0)\n    rec = recall_score(val_labels, val_preds, average='macro')\n\n    # Hitung AUC\n    try:\n        lb = LabelBinarizer(); lb.fit(val_labels)\n        if len(lb.classes_) == len(CLASSES):\n            auc = roc_auc_score(lb.transform(val_labels), val_probs[:min_len], multi_class='ovr', average='macro')\n        else:\n            auc = 0.0 # Skip jika kelas validasi tidak lengkap\n    except: auc = 0.0\n\n    print(\"\\n\" + \"=\"*30 + \"\\nLAPORAN EVALUASI (RUBRIK)\\n\" + \"=\"*30)\n    print(f\"Accuracy        : {acc:.4f}\")\n    print(f\"Macro F1-Score  : {f1:.4f}\")\n    print(f\"Macro Precision : {prec:.4f}\")\n    print(f\"Macro Recall    : {rec:.4f}\")\n    print(f\"Macro AUC       : {auc:.4f}\")\n    print(\"=\"*30)\n\nexcept Exception as e:\n    print(f\"⚠️ Gagal evaluasi: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T12:21:32.722494Z","iopub.execute_input":"2025-12-03T12:21:32.722700Z","iopub.status.idle":"2025-12-03T12:21:55.171701Z","shell.execute_reply.started":"2025-12-03T12:21:32.722683Z","shell.execute_reply":"2025-12-03T12:21:55.170357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\nprint(\"🚀 Memulai proses prediksi & submission...\")\n\n# --- BARIS PENYELAMAT (FIX ERROR NOT DEFINED) ---\n# Jika BATCH_SIZE hilang (karena reset/disconnect), kita set manual ke 16\nif 'BATCH_SIZE' not in globals():\n    BATCH_SIZE = 16 \n    print(\"⚠️ Warning: Variabel BATCH_SIZE hilang. Menggunakan default: 16\")\n\ntry:\n    # 1. Hitung jumlah langkah (Steps) agar progress bar jelas\n    # Total gambar test: 7382\n    num_test_images = 7382\n    steps = num_test_images // BATCH_SIZE\n    if num_test_images % BATCH_SIZE > 0: steps += 1\n    \n    print(f\"Estimasi langkah: {steps} steps\")\n\n    # 2. Lakukan Prediksi\n    # Kita ambil hanya gambarnya saja untuk prediksi\n    test_images_ds = test_dataset.map(lambda image, idnum: image)\n    \n    print(\"Sedang memprediksi... (Mohon tunggu sampai selesai)\")\n    # verbose=1 memunculkan progress bar\n    probabilities = model.predict(test_images_ds, verbose=1) \n    predictions = np.argmax(probabilities, axis=-1)\n    \n    print(\"✅ Prediksi selesai!\")\n\n    # 3. Ambil ID Gambar\n    print(\"Mengambil ID gambar...\")\n    test_ids_ds = test_dataset.map(lambda image, idnum: idnum).unbatch()\n    \n    # Trik: ambil ID sejumlah hasil prediksi saja biar pas\n    # (Menghindari error jika jumlah data tidak sinkron sedikit)\n    test_ids = next(iter(test_ids_ds.batch(len(predictions)))).numpy().astype('U')\n\n    # 4. Simpan ke CSV\n    np.savetxt(\n        'submission.csv', \n        np.rec.fromarrays([test_ids, predictions]), \n        fmt=['%s', '%d'], \n        delimiter=',', \n        header='id,label', \n        comments=''\n    )\n    print(f\"🎉 Selesai! File submission.csv berhasil dibuat dengan {len(predictions)} baris.\")\n    print(\"👉 Silakan download file di menu Output sebelah kanan.\")\n\nexcept NameError as e:\n    print(f\"\\n❌ ERROR VARIABEL: {e}\")\n    print(\"Kemungkinan Anda belum menjalankan Cell 'Dataset' atau 'Model'.\")\nexcept KeyboardInterrupt:\n    print(\"\\n⚠️ Proses DIHENTIKAN OLEH USER.\")\nexcept Exception as e:\n    print(f\"\\n❌ Error lain: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T12:21:55.172158Z","iopub.execute_input":"2025-12-03T12:21:55.172371Z","iopub.status.idle":"2025-12-03T12:22:38.952104Z","shell.execute_reply.started":"2025-12-03T12:21:55.172353Z","shell.execute_reply":"2025-12-03T12:22:38.950838Z"}},"outputs":[],"execution_count":null}]}