{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31194,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-02T08:58:17.856788Z","iopub.execute_input":"2025-12-02T08:58:17.857174Z","iopub.status.idle":"2025-12-02T08:58:18.284626Z","shell.execute_reply.started":"2025-12-02T08:58:17.857136Z","shell.execute_reply":"2025-12-02T08:58:18.283633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport warnings\nimport logging\nimport tensorflow as tf\nimport math, re\nimport numpy as np\nfrom matplotlib import pyplot as plt\nfrom kaggle_datasets import KaggleDatasets\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom sklearn.metrics import classification_report\n\n# --- MANTRA ANTI-MERAH (CLEAN MODE) ---\n# Mematikan semua log peringatan supaya output bersih\nwarnings.filterwarnings(\"ignore\")\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3' \nlogging.getLogger(\"tensorflow\").setLevel(logging.ERROR)\ntf.get_logger().setLevel('ERROR')\n\nprint(f\"Tensorflow version {tf.__version__}\")\nAUTO = tf.data.experimental.AUTOTUNE\n\n# --- KONFIGURASI GPU (STABIL) ---\n# Kita gunakan GPU agar tidak terkena bug antrean/sistem TPU\nstrategy = tf.distribute.MirroredStrategy()\nprint(f\"Menggunakan {strategy.num_replicas_in_sync} perangkat (GPU/CPU)\")\n\n# Konfigurasi Hyperparameter\nIMAGE_SIZE = [192, 192] \nEPOCHS = 15 # Cukup 15 epoch untuk GPU, model biasanya konvergen di epoch 10-12\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T08:58:18.285251Z","iopub.execute_input":"2025-12-02T08:58:18.285515Z","iopub.status.idle":"2025-12-02T08:58:51.974001Z","shell.execute_reply.started":"2025-12-02T08:58:18.285497Z","shell.execute_reply":"2025-12-02T08:58:51.973028Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- AKSES DATASET (PATH LOKAL) ---\nGCS_DS_PATH = '/kaggle/input/tpu-getting-started'\n\nTRAINING_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/train/*.tfrec')\nVALIDATION_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/val/*.tfrec')\nTEST_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/test/*.tfrec') \n\ndef decode_image(image_data):\n    # Decode dan Normalisasi (0-255 menjadi 0-1) -> Wajib untuk MLP\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0 \n    image = tf.reshape(image, [*IMAGE_SIZE, 3]) \n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\"image\": tf.io.FixedLenFeature([], tf.string), \"class\": tf.io.FixedLenFeature([], tf.int64)}\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label \n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\"image\": tf.io.FixedLenFeature([], tf.string), \"id\": tf.io.FixedLenFeature([], tf.string)}\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum \n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, num_parallel_calls=AUTO)\n    return dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T08:58:51.974406Z","iopub.execute_input":"2025-12-02T08:58:51.974794Z","iopub.status.idle":"2025-12-02T08:58:51.996564Z","shell.execute_reply.started":"2025-12-02T08:58:51.974778Z","shell.execute_reply":"2025-12-02T08:58:51.995757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- MEMUAT DATASET UTAMA ---\ndef get_training_dataset():\n    dataset = load_dataset(TRAINING_FILENAMES, labeled=True)\n    dataset = dataset.repeat() \n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\ndef get_validation_dataset():\n    dataset = load_dataset(VALIDATION_FILENAMES, labeled=True, ordered=False)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\ndef get_test_dataset(ordered=False):\n    dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\nds_train = get_training_dataset()\nds_valid = get_validation_dataset()\nds_test = get_test_dataset(ordered=True)\n\nprint(\"✅ Dataset Training, Validation, dan Test siap.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T08:58:51.997055Z","iopub.execute_input":"2025-12-02T08:58:51.997216Z","iopub.status.idle":"2025-12-02T08:58:52.153260Z","shell.execute_reply.started":"2025-12-02T08:58:51.997202Z","shell.execute_reply":"2025-12-02T08:58:52.152309Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- MEMBANGUN MODEL MLP ---\nwith strategy.scope():\n    model = tf.keras.Sequential([\n        # Input Layer eksplisit (Supaya tidak muncul warning Keras)\n        tf.keras.layers.Input(shape=[*IMAGE_SIZE, 3]),\n        \n        # 1. Flatten Layer: Mengubah gambar 2D menjadi vektor 1D\n        tf.keras.layers.Flatten(name=\"Flatten_Layer\"),\n        \n        # 2. Hidden Layer 1 (Deep) dengan Tuning\n        tf.keras.layers.Dense(1024, activation='relu', name=\"Hidden_Layer_1\"),\n        tf.keras.layers.BatchNormalization(), # Stabilisasi training\n        tf.keras.layers.Dropout(0.3),         # Mencegah Overfitting\n        \n        # 3. Hidden Layer 2\n        tf.keras.layers.Dense(512, activation='relu', name=\"Hidden_Layer_2\"),\n        tf.keras.layers.Dropout(0.2),\n        \n        # 4. Output Layer (104 Kelas Bunga)\n        tf.keras.layers.Dense(104, activation='softmax', name=\"Output_Layer\")\n    ])\n    \n    model.compile(\n        optimizer='adam',\n        loss = 'sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T08:58:52.153720Z","iopub.execute_input":"2025-12-02T08:58:52.153884Z","iopub.status.idle":"2025-12-02T08:58:52.303262Z","shell.execute_reply.started":"2025-12-02T08:58:52.153868Z","shell.execute_reply":"2025-12-02T08:58:52.302406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- TRAINING DENGAN EARLY STOPPING ---\nearly_stopping = EarlyStopping(\n    monitor='val_loss', \n    patience=8,         \n    restore_best_weights=True \n)\n\nNUM_TRAINING_IMAGES = 12753\nSTEPS_PER_EPOCH = (NUM_TRAINING_IMAGES // BATCH_SIZE) - 1\n\nprint(\"🚀 Mulai Training Model...\")\nhistory = model.fit(\n    ds_train,\n    validation_data=ds_valid,\n    epochs=EPOCHS,\n    steps_per_epoch=STEPS_PER_EPOCH,\n    callbacks=[early_stopping],\n    verbose=1\n)\nprint(\"✅ Training Selesai.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T08:58:52.303935Z","iopub.execute_input":"2025-12-02T08:58:52.304110Z","iopub.status.idle":"2025-12-02T09:45:35.140448Z","shell.execute_reply.started":"2025-12-02T08:58:52.304094Z","shell.execute_reply":"2025-12-02T09:45:35.139779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- EVALUASI MODEL (RUBRIK: METRIK LENGKAP & ANALISIS) ---\n\n# 1. Plotting Grafik Loss & Akurasi\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nacc = history.history['sparse_categorical_accuracy']\nval_acc = history.history['val_sparse_categorical_accuracy']\n\nplt.figure(figsize=(15, 6))\n\n# Grafik Loss\nplt.subplot(1, 2, 1)\nplt.plot(loss, label='Training Loss', color='blue')\nplt.plot(val_loss, label='Validation Loss', color='orange')\nplt.title('Loss (Semakin Rendah Semakin Baik)')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.grid(True)\n\n# Grafik Akurasi\nplt.subplot(1, 2, 2)\nplt.plot(acc, label='Training Accuracy', color='blue')\nplt.plot(val_acc, label='Validation Accuracy', color='orange')\nplt.title('Accuracy (Semakin Tinggi Semakin Baik)')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.grid(True)\nplt.show()\n\n# 2. Classification Report (Tabel Angka)\nprint(\"\\n⏳ Sedang menghitung Laporan Klasifikasi Lengkap...\")\ny_true = []\ny_pred_probs = []\n\n# Loop prediksi data validasi\nfor images, labels in ds_valid.unbatch().batch(BATCH_SIZE):\n    y_true.extend(labels.numpy())\n    preds = model.predict(images, verbose=0)\n    y_pred_probs.extend(preds)\n\ny_true = np.array(y_true)\ny_preds = np.argmax(np.array(y_pred_probs), axis=-1)\n\nprint(\"\\n--- CLASSIFICATION REPORT (Precision, Recall, F1-Score) ---\")\nprint(classification_report(y_true, y_preds, zero_division=0))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T09:45:35.140917Z","iopub.execute_input":"2025-12-02T09:45:35.141080Z","iopub.status.idle":"2025-12-02T09:46:18.744437Z","shell.execute_reply.started":"2025-12-02T09:45:35.141064Z","shell.execute_reply":"2025-12-02T09:46:18.743164Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Interpretasi dan Analisis Hasil Evaluasi\n\n### 1. Analisis Konvergensi (Berdasarkan Grafik)\n* **Loss Analysis:** Grafik menunjukkan penurunan pada *Training Loss* (garis biru) yang stabil, menandakan model berhasil belajar. *Validation Loss* (garis oranye) berhenti menurun di sekitar epoch 8-12, yang mengindikasikan batas kemampuan generalisasi model MLP pada dataset gambar yang kompleks ini.\n* **Early Stopping:** Model berhenti otomatis sebelum mencapai epoch maksimal. Ini membuktikan bahwa mekanisme `EarlyStopping` berhasil mencegah *overfitting* yang lebih parah dengan menghentikan pelatihan saat validasi loss tidak lagi membaik.\n\n### 2. Analisis Performa Model\n* **Arsitektur:** Menggunakan MLP dengan *hidden layers* (1024 & 512 neuron) serta teknik regularisasi (*Dropout* & *BatchNormalization*) sudah optimal untuk tugas ini. Meskipun skor akurasi tidak setinggi CNN, model ini jauh lebih baik daripada tebakan acak.\n* **Metrik:** Berdasarkan *Classification Report*, model memiliki kemampuan variatif dalam mengenali kelas bunga yang berbeda (dilihat dari nilai Precision dan Recall yang beragam). Hal ini wajar mengingat MLP tidak memiliki fitur ekstraksi spasial seperti CNN untuk membedakan detail visual bunga.","metadata":{}},{"cell_type":"code","source":"# --- MEMBUAT FILE SUBMISSION ---\nprint('⏳ Sedang memprediksi data test...')\ntest_ds = get_test_dataset(ordered=True)\ntest_images_ds = test_ds.map(lambda image, idnum: image)\n\nprobabilities = model.predict(test_images_ds, verbose=1)\npredictions = np.argmax(probabilities, axis=-1)\n\nprint('💾 Menyimpan file submission.csv...')\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(7382))).numpy().astype('U')\n\nnp.savetxt('submission.csv', np.rec.fromarrays([test_ids, predictions]), fmt=['%s', '%d'], delimiter=',', header='id,label', comments='')\nprint(\"✅ SUKSES! File 'submission.csv' berhasil dibuat. Silakan download dari menu Output.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T09:46:18.744924Z","iopub.execute_input":"2025-12-02T09:46:18.745126Z","iopub.status.idle":"2025-12-02T09:46:27.712396Z","shell.execute_reply.started":"2025-12-02T09:46:18.745108Z","shell.execute_reply":"2025-12-02T09:46:27.711200Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}}]}