{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31194,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-02T02:05:29.831952Z","iopub.execute_input":"2025-12-02T02:05:29.832203Z","iopub.status.idle":"2025-12-02T02:05:32.680079Z","shell.execute_reply.started":"2025-12-02T02:05:29.832181Z","shell.execute_reply":"2025-12-02T02:05:32.679173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport logging\nimport warnings\n\n# --- MANTRA ANTI-MERAH (CLEAN MODE) ---\n# 1. Matikan warning Python biasa\nwarnings.filterwarnings(\"ignore\")\n# 2. Matikan log sistem Tensorflow (hanya tampilkan Error fatal)\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3' \n# 3. Matikan logger internal Tensorflow\nlogging.getLogger(\"tensorflow\").setLevel(logging.ERROR)\nimport tensorflow as tf\ntf.get_logger().setLevel('ERROR')\n\nimport math, re\nimport numpy as np\nfrom matplotlib import pyplot as plt\nfrom kaggle_datasets import KaggleDatasets\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom sklearn.metrics import classification_report\n\nprint(\"Tensorflow version \" + tf.__version__)\nAUTO = tf.data.experimental.AUTOTUNE\n\n# --- KONFIGURASI TPU (Dengan Penanganan Error Diam-diam) ---\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() \n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)\n\n# Konfigurasi Hyperparameter\nIMAGE_SIZE = [192, 192] \nEPOCHS = 20\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T02:05:32.680753Z","iopub.execute_input":"2025-12-02T02:05:32.680988Z","iopub.status.idle":"2025-12-02T02:06:03.305927Z","shell.execute_reply.started":"2025-12-02T02:05:32.680972Z","shell.execute_reply":"2025-12-02T02:06:03.304864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- AKSES DATASET (SOLUSI LOKAL) ---\n# Kita pakai alamat lokal supaya tidak muncul warning \"GCS Path\"\nGCS_DS_PATH = '/kaggle/input/tpu-getting-started'\n\nTRAINING_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/train/*.tfrec')\nVALIDATION_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/val/*.tfrec')\nTEST_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/test/*.tfrec') \n\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0  # Normalisasi\n    image = tf.reshape(image, [*IMAGE_SIZE, 3]) \n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\"image\": tf.io.FixedLenFeature([], tf.string), \"class\": tf.io.FixedLenFeature([], tf.int64)}\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label \n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\"image\": tf.io.FixedLenFeature([], tf.string), \"id\": tf.io.FixedLenFeature([], tf.string)}\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum \n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, num_parallel_calls=AUTO)\n    return dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T02:06:03.306701Z","iopub.execute_input":"2025-12-02T02:06:03.307118Z","iopub.status.idle":"2025-12-02T02:06:03.331182Z","shell.execute_reply.started":"2025-12-02T02:06:03.307093Z","shell.execute_reply":"2025-12-02T02:06:03.330273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- SIAPKAN DATA ---\ndef get_training_dataset():\n    dataset = load_dataset(TRAINING_FILENAMES, labeled=True)\n    dataset = dataset.repeat() \n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\ndef get_validation_dataset():\n    dataset = load_dataset(VALIDATION_FILENAMES, labeled=True, ordered=False)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\ndef get_test_dataset(ordered=False):\n    dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\nds_train = get_training_dataset()\nds_valid = get_validation_dataset()\nds_test = get_test_dataset(ordered=True)\n\nprint(\"Dataset siap!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T02:06:03.331803Z","iopub.execute_input":"2025-12-02T02:06:03.331988Z","iopub.status.idle":"2025-12-02T02:06:03.499303Z","shell.execute_reply.started":"2025-12-02T02:06:03.331973Z","shell.execute_reply":"2025-12-02T02:06:03.498225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- MEMBANGUN MODEL MLP ---\nwith strategy.scope():\n    model = tf.keras.Sequential([\n        # Pakai Input Layer eksplisit supaya Keras tidak ngomel (muncul merah)\n        tf.keras.layers.Input(shape=[*IMAGE_SIZE, 3]),\n        \n        tf.keras.layers.Flatten(name=\"Flatten\"),\n        \n        tf.keras.layers.Dense(1024, activation='relu', name=\"Hidden1\"),\n        tf.keras.layers.BatchNormalization(), \n        tf.keras.layers.Dropout(0.3),         \n        \n        tf.keras.layers.Dense(512, activation='relu', name=\"Hidden2\"),\n        tf.keras.layers.Dropout(0.2),\n        \n        tf.keras.layers.Dense(104, activation='softmax', name=\"Output\")\n    ])\n    \n    model.compile(\n        optimizer='adam',\n        loss = 'sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T02:06:03.500046Z","iopub.execute_input":"2025-12-02T02:06:03.500226Z","iopub.status.idle":"2025-12-02T02:06:03.648655Z","shell.execute_reply.started":"2025-12-02T02:06:03.500210Z","shell.execute_reply":"2025-12-02T02:06:03.647768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- TRAINING MODEL ---\nearly_stopping = EarlyStopping(\n    monitor='val_loss', \n    patience=8,         \n    restore_best_weights=True \n)\n\nNUM_TRAINING_IMAGES = 12753\n# Kurangi 1 langkah supaya tidak muncul error 'Ran out of data'\nSTEPS_PER_EPOCH = (NUM_TRAINING_IMAGES // BATCH_SIZE) - 1\n\nprint(\"Mulai Training (Mode Hening)...\")\n# verbose=1 tetap menampilkan progress bar hijau, tapi tanpa warning merah\nhistory = model.fit(\n    ds_train,\n    validation_data=ds_valid,\n    epochs=EPOCHS,\n    steps_per_epoch=STEPS_PER_EPOCH,\n    callbacks=[early_stopping],\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T02:06:03.649501Z","iopub.execute_input":"2025-12-02T02:06:03.649691Z","iopub.status.idle":"2025-12-02T02:50:59.029387Z","shell.execute_reply.started":"2025-12-02T02:06:03.649674Z","shell.execute_reply":"2025-12-02T02:50:59.027940Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- EVALUASI (GRAFIK & TABEL LAPORAN) ---\nprint(\"Membuat Grafik & Laporan...\")\n\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nacc = history.history['sparse_categorical_accuracy']\nval_acc = history.history['val_sparse_categorical_accuracy']\n\nplt.figure(figsize=(15, 6))\n\n# Grafik 1\nplt.subplot(1, 2, 1)\nplt.plot(loss, label='Training Loss', color='blue')\nplt.plot(val_loss, label='Validation Loss', color='orange')\nplt.title('Loss')\nplt.legend()\nplt.grid(True)\n\n# Grafik 2\nplt.subplot(1, 2, 2)\nplt.plot(acc, label='Training Acc', color='blue')\nplt.plot(val_acc, label='Validation Acc', color='orange')\nplt.title('Accuracy')\nplt.legend()\nplt.grid(True)\nplt.show()\n\n# Tabel Laporan\ny_true = []\ny_pred_probs = []\nfor images, labels in ds_valid.unbatch().batch(BATCH_SIZE):\n    y_true.extend(labels.numpy())\n    preds = model.predict(images, verbose=0)\n    y_pred_probs.extend(preds)\n\ny_true = np.array(y_true)\ny_preds = np.argmax(np.array(y_pred_probs), axis=-1)\n\nprint(\"\\n--- CLASSIFICATION REPORT ---\")\nprint(classification_report(y_true, y_preds, zero_division=0))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T02:50:59.029952Z","iopub.execute_input":"2025-12-02T02:50:59.030127Z","iopub.status.idle":"2025-12-02T02:51:19.116609Z","shell.execute_reply.started":"2025-12-02T02:50:59.030111Z","shell.execute_reply":"2025-12-02T02:51:19.115372Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Interpretasi Hasil Evaluasi Model\n\nBerdasarkan grafik konvergensi di atas, berikut adalah analisis performa model MLP yang dibangun:\n\n1.  **Analisis Loss (Loss vs Epoch):**\n    * Grafik menunjukkan tren penurunan pada *Training Loss* (garis biru) seiring bertambahnya epoch, yang menandakan model berhasil mempelajari pola dari data latih.\n    * *Validation Loss* (garis oranye) juga menurun di awal, namun kemudian cenderung stabil. Titik di mana *Validation Loss* berhenti menurun sementara *Training Loss* terus turun mengindikasikan batas kemampuan generalisasi model.\n    * Penggunaan **Dropout** dan **BatchNormalization** pada *Hidden Layer* terbukti efektif menjaga jarak antara *Training* dan *Validation* agar tidak terlalu jauh (mengurangi dampak *overfitting*).\n\n2.  **Keputusan Penghentian (Early Stopping):**\n    * Training berhenti secara otomatis sebelum mencapai batas maksimum epoch karena mekanisme **EarlyStopping**. Hal ini memastikan kita mendapatkan model dengan bobot terbaik (saat *validation loss* paling rendah) dan menghindari *overfitting* yang lebih parah jika training dipaksakan terus berlanjut.\n\n3.  **Kesimpulan:**\n    * Mengingat kompleksitas dataset gambar bunga yang tinggi, arsitektur MLP (yang tidak memiliki fitur ekstraksi spasial seperti CNN) sudah bekerja cukup baik dengan strategi normalisasi data dan tuning hyperparameter yang diterapkan.","metadata":{}},{"cell_type":"code","source":"# --- SUBMISSION ---\nprint('Membuat file submission...')\ntest_ds = get_test_dataset(ordered=True)\ntest_images_ds = test_ds.map(lambda image, idnum: image)\n\nprobabilities = model.predict(test_images_ds, verbose=0)\npredictions = np.argmax(probabilities, axis=-1)\n\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(7382))).numpy().astype('U')\n\nnp.savetxt('submission.csv', np.rec.fromarrays([test_ids, predictions]), fmt=['%s', '%d'], delimiter=',', header='id,label', comments='')\nprint(\"Selesai! Cek Output.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T02:51:19.117470Z","iopub.execute_input":"2025-12-02T02:51:19.117680Z","iopub.status.idle":"2025-12-02T02:51:27.114672Z","shell.execute_reply.started":"2025-12-02T02:51:19.117650Z","shell.execute_reply":"2025-12-02T02:51:27.113344Z"}},"outputs":[],"execution_count":null}]}