{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import math, re, os\nimport tensorflow as tf\nimport numpy as np\nimport pandas as pd\nfrom matplotlib import pyplot as plt\nfrom sklearn.metrics import f1_score, precision_score, recall_score, classification_report\n# Baris di bawah ini yang sebelumnya kurang:\nfrom kaggle_datasets import KaggleDatasets \n\nprint(\"Tensorflow version \" + tf.__version__)\n\n# Deteksi Hardware (TPU atau GPU)\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() \n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy() \n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)\n\n# Akses Data\ntry:\n    # GCS Path diperlukan jika menggunakan TPU\n    GCS_DS_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started')\nexcept:\n    # Fallback jika dijalankan di lokal/tanpa internet akses ke GCS\n    GCS_DS_PATH = '/kaggle/input/tpu-getting-started'\n\nprint(\"Dataset Path:\", GCS_DS_PATH)\n\nIMAGE_SIZE = [192, 192] \nEPOCHS = 10 \nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\n\nNUM_TRAINING_IMAGES = 12753\nNUM_TEST_IMAGES = 7382\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T03:27:59.396096Z","iopub.execute_input":"2025-12-03T03:27:59.396376Z","iopub.status.idle":"2025-12-03T03:27:59.702428Z","shell.execute_reply.started":"2025-12-03T03:27:59.396357Z","shell.execute_reply":"2025-12-03T03:27:59.701706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    # -- PENTING UNTUK RUBRIK: NORMALISASI --\n    image = tf.cast(image, tf.float32) / 255.0  # Mengubah range pixel ke 0-1\n    image = tf.reshape(image, [*IMAGE_SIZE, 3])\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=tf.data.experimental.AUTOTUNE)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, num_parallel_calls=tf.data.experimental.AUTOTUNE)\n    return dataset\n\ndef get_training_dataset():\n    dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/train/*.tfrec'), labeled=True)\n    dataset = dataset.repeat()\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(tf.data.experimental.AUTOTUNE)\n    return dataset\n\ndef get_validation_dataset():\n    dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/val/*.tfrec'), labeled=True, ordered=False)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(tf.data.experimental.AUTOTUNE)\n    return dataset\n\ndef get_test_dataset(ordered=False):\n    dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/test/*.tfrec'), labeled=False, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(tf.data.experimental.AUTOTUNE)\n    return dataset\n\n# Load Data\nprint(\"Loading Training Data...\")\ntraining_dataset = get_training_dataset()\nprint(\"Loading Validation Data...\")\nvalidation_dataset = get_validation_dataset()\nprint(\"Loading Test Data...\")\ntest_dataset = get_test_dataset()\nprint(\"Data Preparation Complete!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T03:28:05.052581Z","iopub.execute_input":"2025-12-03T03:28:05.053041Z","iopub.status.idle":"2025-12-03T03:28:07.229575Z","shell.execute_reply.started":"2025-12-03T03:28:05.053008Z","shell.execute_reply":"2025-12-03T03:28:07.228591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 3: MEMBANGUN MODEL MLP ---\n# Sesuai rubrik: Arsitektur MLP Lengkap (Input, Hidden, Output)\n# Kita menggunakan 'strategy.scope()' agar kode ini kompatibel jika nanti Anda menyalakan TPU/GPU\n\nwith strategy.scope():    \n    model = tf.keras.Sequential([\n        # 1. INPUT LAYER & PREPROCESSING\n        # Flattening: Mengubah gambar matriks (192x192x3) menjadi vektor lurus (flat)\n        # Ini WAJIB untuk MLP. Jika tidak di-flatten, akan error.\n        tf.keras.layers.Flatten(input_shape=[*IMAGE_SIZE, 3]),\n        \n        # 2. HIDDEN LAYERS\n        # Layer 1: 512 neuron dengan aktivasi ReLU\n        tf.keras.layers.Dense(512, activation='relu'),\n        # Dropout: Mematikan 20% neuron secara acak saat training untuk mencegah Overfitting\n        tf.keras.layers.Dropout(0.2), \n        \n        # Layer 2: 256 neuron dengan aktivasi ReLU\n        tf.keras.layers.Dense(256, activation='relu'),\n        \n        # 3. OUTPUT LAYER\n        # Layer Output: 104 neuron (sesuai jumlah kelas bunga)\n        # Aktivasi Softmax WAJIB untuk klasifikasi banyak kelas\n        tf.keras.layers.Dense(104, activation='softmax')\n    ])\n    \n    # Compile Model\n    # Optimizer 'adam' dan Loss 'sparse_categorical_crossentropy' adalah standar terbaik untuk kasus ini\n    model.compile(\n        optimizer='adam',\n        loss = 'sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\n# Menampilkan ringkasan struktur model\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T03:28:17.064373Z","iopub.execute_input":"2025-12-03T03:28:17.065246Z","iopub.status.idle":"2025-12-03T03:28:17.519168Z","shell.execute_reply.started":"2025-12-03T03:28:17.06522Z","shell.execute_reply":"2025-12-03T03:28:17.518546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Mulai Training... Bersabarlah, ini butuh waktu :)\")\nhistory = model.fit(\n    training_dataset, \n    steps_per_epoch=STEPS_PER_EPOCH, \n    epochs=10,  # <--- Langsung tulis angkanya di sini\n    validation_data=validation_dataset\n)\nprint(\"Training Selesai!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T03:28:21.979386Z","iopub.execute_input":"2025-12-03T03:28:21.980093Z","iopub.status.idle":"2025-12-03T04:36:11.864383Z","shell.execute_reply.started":"2025-12-03T03:28:21.980067Z","shell.execute_reply":"2025-12-03T04:36:11.863581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 5: EVALUASI & VISUALISASI ---\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom sklearn.metrics import classification_report\n\n# 1. Plot Grafik Loss & Accuracy\ndef plot_learning_curves(history):\n    fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(15, 5))\n    \n    # Grafik Loss\n    ax1.plot(history.history['loss'], label='Training Loss')\n    ax1.plot(history.history['val_loss'], label='Validation Loss')\n    ax1.set_title('Grafik Loss (Semakin rendah semakin baik)')\n    ax1.set_xlabel('Epoch')\n    ax1.set_ylabel('Loss')\n    ax1.legend()\n    ax1.grid(True)\n    \n    # Grafik Accuracy\n    ax2.plot(history.history['sparse_categorical_accuracy'], label='Training Accuracy')\n    ax2.plot(history.history['val_sparse_categorical_accuracy'], label='Validation Accuracy')\n    ax2.set_title('Grafik Akurasi (Semakin tinggi semakin baik)')\n    ax2.set_xlabel('Epoch')\n    ax2.set_ylabel('Accuracy')\n    ax2.legend()\n    ax2.grid(True)\n    \n    plt.show()\n\n# Tampilkan Grafik\nif 'history' in locals():\n    plot_learning_curves(history)\nelse:\n    print(\"Variabel 'history' tidak ditemukan. Pastikan proses Training (Cell 4) sudah selesai.\")\n\n# 2. Classification Report\nprint(\"Sedang menghitung Classification Report pada data validasi...\")\n\n# Ambil data validasi dan label aslinya\nval_images_ds = validation_dataset.map(lambda image, label: image)\nval_labels_ds = validation_dataset.map(lambda image, label: label).unbatch()\nval_labels = next(iter(val_labels_ds.batch(NUM_TEST_IMAGES))).numpy() # Ambil semua label\n\n# Prediksi model\nval_probs = model.predict(val_images_ds)\nval_preds = np.argmax(val_probs, axis=-1)\n\n# Tampilkan report\nprint(\"\\n--- CLASSIFICATION REPORT ---\")\nprint(classification_report(val_labels, val_preds))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T04:36:57.265311Z","iopub.execute_input":"2025-12-03T04:36:57.265709Z","iopub.status.idle":"2025-12-03T04:37:04.582308Z","shell.execute_reply.started":"2025-12-03T04:36:57.265686Z","shell.execute_reply":"2025-12-03T04:37:04.581436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 6: PREDIKSI TEST SET & SUBMISSION ---\nimport pandas as pd\n\n# Pastikan urutan data test benar (ordered=True)\ntest_ds = get_test_dataset(ordered=True) \n\nprint('Computing predictions...')\n# Ambil hanya gambarnya saja\ntest_images_ds = test_ds.map(lambda image, idnum: image)\n\n# Lakukan prediksi\nprobabilities = model.predict(test_images_ds)\npredictions = np.argmax(probabilities, axis=-1)\nprint(f\"Prediksi selesai. Contoh hasil: {predictions[:10]}\")\n\nprint('Generating submission.csv file...')\n# Ambil ID gambar untuk dipasangkan dengan hasil prediksi\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(NUM_TEST_IMAGES))).numpy().astype('U')\n\n# Buat DataFrame dan simpan ke CSV\nsubmission = pd.DataFrame({'id': test_ids, 'label': predictions})\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission saved! Cek folder Output di sidebar kanan.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T04:37:29.035595Z","iopub.execute_input":"2025-12-03T04:37:29.035893Z","iopub.status.idle":"2025-12-03T04:37:54.407771Z","shell.execute_reply.started":"2025-12-03T04:37:29.035867Z","shell.execute_reply":"2025-12-03T04:37:54.406686Z"}},"outputs":[],"execution_count":null}]}