{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# PRAKTIKUM KECERDASAN BUATAN - NEURAL NETWORK\n# Multi-Layer Perceptron untuk Klasifikasi 104 Jenis Bunga\n# Institut Teknologi Del - Semester Ganjil 2025/2026\n# NIM : 12S23013\n# Nama: Andika Immanuel Nadapdap","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:47:42.130148Z","iopub.execute_input":"2025-12-03T13:47:42.130464Z","iopub.status.idle":"2025-12-03T13:47:42.134971Z","shell.execute_reply.started":"2025-12-03T13:47:42.130433Z","shell.execute_reply":"2025-12-03T13:47:42.134023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =================================================================================\n# PRAKTIKUM KECERDASAN BUATAN - NEURAL NETWORK\n# Multi-Layer Perceptron untuk Klasifikasi 104 Jenis Bunga\n# =================================================================================\n\n# 1. PERBAIKAN DEPENDENCIES (WAJIB DI BARIS PALING ATAS)\n# Mengatasi error \"MessageFactory\" yang menyebabkan Save Version gagal di Kaggle\n!pip install -q protobuf==3.20.3\n\nimport os\nimport warnings\nimport tensorflow as tf\nfrom kaggle_datasets import KaggleDatasets\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport seaborn as sns\n\n# Konfigurasi Environment (biar log TensorFlow tidak berisik)\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'\nwarnings.filterwarnings(\"ignore\")\n\nprint(f\"TensorFlow Version: {tf.__version__}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:47:42.136227Z","iopub.execute_input":"2025-12-03T13:47:42.136463Z","iopub.status.idle":"2025-12-03T13:47:45.346807Z","shell.execute_reply.started":"2025-12-03T13:47:42.136442Z","shell.execute_reply":"2025-12-03T13:47:45.346021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# 2. DETEKSI HARDWARE (TPU / GPU)\n# ==========================================\ntry:\n    # Coba inisialisasi TPU (kalau notebook pakai TPU)\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\n    print(\">> HARDWARE: TPU DETECTED\")\nexcept ValueError:\n    # Jika TPU tidak tersedia → fallback ke MirroredStrategy (GPU/CPU multi-device)\n    strategy = tf.distribute.MirroredStrategy()\n    print(\">> HARDWARE: GPU/CPU (MirroredStrategy) DETECTED\")\n\nprint(f\">> Number of accelerators: {strategy.num_replicas_in_sync}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:47:45.347998Z","iopub.execute_input":"2025-12-03T13:47:45.348295Z","iopub.status.idle":"2025-12-03T13:47:45.358348Z","shell.execute_reply.started":"2025-12-03T13:47:45.348268Z","shell.execute_reply":"2025-12-03T13:47:45.357634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# 3. HYPERPARAMETERS & PATHS\n# ==========================================\n# Ukuran gambar kecil (64x64) agar MLP tidak kehabisan memori (OOM)\nIMAGE_SIZE = [64, 64]\nEPOCHS = 25\n\n# Batch size disesuaikan dengan jumlah device (TPU/GPU)\nBATCH_SIZE = 128 * strategy.num_replicas_in_sync\nLEARNING_RATE = 0.001\n\n# Mendapatkan Path GCS Data dari KaggleDatasets\ntry:\n    GCS_DS_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started')\nexcept:\n    print(\"GCS Path not found, using local path.\")\n    GCS_DS_PATH = \"/kaggle/input/tpu-getting-started\"\n\n# List file TFRecord untuk train / val / test\nTRAIN_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/train/*.tfrec')\nVAL_FILENAMES   = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/val/*.tfrec')\nTEST_FILENAMES  = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/test/*.tfrec')\n\nprint(f\"Train files  : {len(TRAIN_FILENAMES)}\")\nprint(f\"Val files    : {len(VAL_FILENAMES)}\")\nprint(f\"Test files   : {len(TEST_FILENAMES)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:47:45.359134Z","iopub.execute_input":"2025-12-03T13:47:45.359369Z","iopub.status.idle":"2025-12-03T13:47:45.760510Z","shell.execute_reply.started":"2025-12-03T13:47:45.359348Z","shell.execute_reply":"2025-12-03T13:47:45.759924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# 4. PREPROCESSING DATA PIPELINE\n# ==========================================\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)   # decode dari bytes ke tensor RGB\n    image = tf.image.convert_image_dtype(image, tf.float32)  # ubah ke float32 [0,1]\n    image = tf.image.resize(image, IMAGE_SIZE)             # resize ke IMAGE_SIZE\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    # Opsi untuk mengabaikan urutan agar bisa di-parallel dan lebih cepat\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False\n\n    dataset = tf.data.TFRecordDataset(\n        filenames,\n        num_parallel_reads=tf.data.AUTOTUNE\n    )\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(\n        read_labeled_tfrecord if labeled else read_unlabeled_tfrecord,\n        num_parallel_calls=tf.data.AUTOTUNE\n    )\n    return dataset\n\n# Setup Dataset\n# PERHATIKAN: .repeat() ditambahkan agar data tidak habis saat distributed training\ntrain_dataset = load_dataset(TRAIN_FILENAMES, labeled=True)\ntrain_dataset = (\n    train_dataset\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(tf.data.AUTOTUNE)\n)\n\nvalid_dataset = load_dataset(VAL_FILENAMES, labeled=True, ordered=True)\nvalid_dataset = (\n    valid_dataset\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(tf.data.AUTOTUNE)\n)\n\ntest_dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=True)\ntest_dataset = (\n    test_dataset\n    .batch(BATCH_SIZE)\n    .prefetch(tf.data.AUTOTUNE)\n)\n\n# Hitung Steps per Epoch (jumlah batch per epoch)\nNUM_TRAINING_IMAGES = 12753  # dari deskripsi dataset\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\nprint(f\"Steps per epoch: {STEPS_PER_EPOCH}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:47:45.762042Z","iopub.execute_input":"2025-12-03T13:47:45.762243Z","iopub.status.idle":"2025-12-03T13:47:45.954861Z","shell.execute_reply.started":"2025-12-03T13:47:45.762227Z","shell.execute_reply":"2025-12-03T13:47:45.954221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# 5. MEMBANGUN MODEL MLP (RUBRIK: ARSITEKTUR LENGKAP)\n# ==========================================\nwith strategy.scope():\n    model = tf.keras.Sequential([\n        # Input Layer Eksplisit (menghindari warning shape)\n        tf.keras.layers.InputLayer(shape=(IMAGE_SIZE[0], IMAGE_SIZE[1], 3)),\n\n        # Flattening: Mengubah gambar 2D (64x64x3) -> vector 1D\n        tf.keras.layers.Flatten(),\n\n        # Hidden Layer 1: ReLU + BatchNorm + Dropout\n        tf.keras.layers.Dense(2048, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.3),\n\n        # Hidden Layer 2\n        tf.keras.layers.Dense(1024, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.3),\n\n        # Hidden Layer 3\n        tf.keras.layers.Dense(512, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n\n        # Output Layer: 104 kelas bunga dengan Softmax\n        tf.keras.layers.Dense(104, activation='softmax')\n    ])\n\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=LEARNING_RATE),\n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\nprint(\"\\n>>> MODEL SUMMARY:\")\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:47:45.955519Z","iopub.execute_input":"2025-12-03T13:47:45.955770Z","iopub.status.idle":"2025-12-03T13:47:46.084402Z","shell.execute_reply.started":"2025-12-03T13:47:45.955754Z","shell.execute_reply":"2025-12-03T13:47:46.083787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# 6. TRAINING PROCESS\n# ==========================================\nprint(\"\\n>>> STARTING TRAINING...\")\n\nearly_stopping = tf.keras.callbacks.EarlyStopping(\n    monitor='val_loss',\n    patience=10,\n    restore_best_weights=True\n)\n\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor='val_loss',\n    factor=0.2,\n    patience=3,\n    min_lr=1e-6\n)\n\nhistory = model.fit(\n    train_dataset,\n    steps_per_epoch=STEPS_PER_EPOCH,\n    epochs=EPOCHS,\n    validation_data=valid_dataset,\n    callbacks=[early_stopping, reduce_lr],\n    verbose=1\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:47:46.085694Z","iopub.execute_input":"2025-12-03T13:47:46.086329Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# 7. EVALUASI & VISUALISASI (RUBRIK: ANALISIS)\n# ==========================================\ndef plot_training(history):\n    acc      = history.history['sparse_categorical_accuracy']\n    val_acc  = history.history['val_sparse_categorical_accuracy']\n    loss     = history.history['loss']\n    val_loss = history.history['val_loss']\n    epochs_range = range(1, len(acc) + 1)\n\n    plt.figure(figsize=(15, 6))\n\n    # Grafik Akurasi\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs_range, acc, label='Training Accuracy')\n    plt.plot(epochs_range, val_acc, label='Validation Accuracy')\n    plt.title('Training and Validation Accuracy')\n    plt.legend(loc='lower right')\n\n    # Grafik Loss\n    plt.subplot(1, 2, 2)\n    plt.plot(epochs_range, loss, label='Training Loss')\n    plt.plot(epochs_range, val_loss, label='Validation Loss')\n    plt.title('Training and Validation Loss')\n    plt.legend(loc='upper right')\n\n    plt.show()\n\nprint(\"\\n>>> MENAMPILKAN GRAFIK EVALUASI:\")\nplot_training(history)\n\n# Metrik Lengkap (Classification Report)\nprint(\"\\n>>> MENGHITUNG CLASSIFICATION REPORT...\")\n\n# Ambil gambar dari valid_dataset untuk prediksi\nval_images_ds = valid_dataset.map(lambda image, label: image)\nval_labels_ds = valid_dataset.map(lambda image, label: label).unbatch()\ny_true = next(iter(val_labels_ds.batch(3712))).numpy()\n\nprobs_val = model.predict(val_images_ds, verbose=1)\ny_pred = np.argmax(probs_val, axis=-1)\n\nprint(classification_report(y_true, y_pred))\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# 8. SUBMISSION KE KAGGLE\n# ==========================================\nprint(\"\\n>>> MEMBUAT FILE SUBMISSION...\")\n\n# Ambil hanya gambar dari test_dataset\ntest_images_ds = test_dataset.map(lambda image, idnum: image)\nprobabilities = model.predict(test_images_ds, verbose=1)\npredictions = np.argmax(probabilities, axis=-1)\n\n# Ambil ID gambar dari test_dataset\ntest_ids_ds = test_dataset.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(7382))).numpy().astype('U')\n\n# Susun DataFrame submission\nsubmission = pd.DataFrame({\n    'id': test_ids,\n    'label': predictions\n})\n\nsubmission.to_csv('submission.csv', index=False)\nprint(\">>> SELESAI! File 'submission.csv' berhasil dibuat dan siap di-submit.\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}