{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Nama :Oloan Nainggolan\n## NIM  :12S23033","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2025-12-03T13:26:45.842477Z","iopub.execute_input":"2025-12-03T13:26:45.843011Z","iopub.status.idle":"2025-12-03T13:26:45.84773Z","shell.execute_reply.started":"2025-12-03T13:26:45.842985Z","shell.execute_reply":"2025-12-03T13:26:45.846953Z"}}},{"cell_type":"markdown","source":"# ==========================================\n# 1. KONFIGURASI & HYPERPARAMETER TUNING\n# ==========================================","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, models, optimizers, callbacks\nfrom kaggle_datasets import KaggleDatasets\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport seaborn as sns\nimport math\n\n\n# Rubrik: Hyperparameter Tuning (Batch size, Epochs, LR)\nIMAGE_SIZE = [128, 128] # Resize agar MLP tidak memory overflow\nEPOCHS = 25             # Jumlah epoch\nLEARNING_RATE = 0.0005  # Learning rate awal\n# Batch size disesuaikan dengan strategi TPU (16 * 8 cores = 128)\nBATCH_SIZE_PER_REPLICA = 16 ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n# ==========================================\n# 2. SETUP HARDWARE (TPU/GPU/CPU)\n# ==========================================","metadata":{}},{"cell_type":"code","source":"\ntry:\n    # Deteksi hardware TPU. Jika dijalankan di Kaggle, TPUClusterResolver() otomatis\n    # mendeteksi TPU dari environment variable.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Device:', tpu.master())\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\n    print('Running on TPU')\nexcept ValueError:\n    # Jika TPU tidak ditemukan, cek apakah GPU tersedia\n    if len(tf.config.list_physical_devices('GPU')) > 0:\n        strategy = tf.distribute.MirroredStrategy()\n        print('Running on GPU')\n    else:\n        # Default strategy untuk CPU\n        strategy = tf.distribute.get_strategy()\n        print('Running on CPU')\n\nprint('Number of replicas:', strategy.num_replicas_in_sync)\n\nBATCH_SIZE = BATCH_SIZE_PER_REPLICA * strategy.num_replicas_in_sync\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:26:59.545445Z","iopub.execute_input":"2025-12-03T13:26:59.5457Z","iopub.status.idle":"2025-12-03T13:26:59.559322Z","shell.execute_reply.started":"2025-12-03T13:26:59.545682Z","shell.execute_reply":"2025-12-03T13:26:59.558562Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ==========================================\n# 3. DATA LOADING & PREPROCESSING\n# ==========================================","metadata":{}},{"cell_type":"code","source":"\n# Rubrik: Preprocessing & Fitur (Normalisasi & Feature Selection/Flattening)\n\n# Mengambil path GCS (Google Cloud Storage) dataset\ntry:\n    GCS_DS_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started')\nexcept:\n    GCS_DS_PATH = \"gs://kds-...\" # Fallback jika dijalankan lokal (perlu path manual)\n\nGCS_PATH_SELECT = { # Menggunakan ukuran 192x192 aslinya, nanti di resize\n    192: GCS_DS_PATH + '/tfrecords-jpeg-192x192',\n    224: GCS_DS_PATH + '/tfrecords-jpeg-224x224',\n    331: GCS_DS_PATH + '/tfrecords-jpeg-331x331',\n    512: GCS_DS_PATH + '/tfrecords-jpeg-512x512'\n}\nGCS_PATH = GCS_PATH_SELECT[192]\n\nTRAINING_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/train/*.tfrec')\nVALIDATION_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/val/*.tfrec')\nTEST_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/test/*.tfrec')\n\nCLASSES = [\n    'pink primrose', 'hard-leaved pocket orchid', 'canterbury bells', 'sweet pea', \n    'wild geranium', 'tiger lily', 'moon orchid', 'bird of paradise', 'monkshood', \n    'globe thistle', 'snapdragon', \"colt's foot\", 'king protea', 'spear thistle', \n    'yellow iris', 'globe-flower', 'purple coneflower', 'peruvian lily', 'balloon flower', \n    'giant white arum lily', 'fire lily', 'pincushion flower', 'fritillary', 'red ginger', \n    'grape hyacinth', 'corn poppy', 'prince of wales feathers', 'stemless gentian', \n    'artichoke', 'sweet william', 'carnation', 'garden phlox', 'love in the mist', \n    'cosmos', 'alpine sea holly', 'ruby-lipped cattleya', 'cape flower', 'great masterwort', \n    'siam tulip', 'lenten rose', 'barberton daisy', 'daffodil', 'sword lily', 'poinsettia', \n    'bolero deep blue', 'wallflower', 'marigold', 'buttercup', 'daisy', 'common dandelion', \n    'petunia', 'wild pansy', 'primula', 'sunflower', 'lilac hibiscus', 'bishop of llandaff', \n    'gaillardia', 'gazania', 'azalea', 'water lily', 'rose', 'thorn apple', 'morning glory', \n    'passion flower', 'lotus', 'toad lily', 'anthurium', 'frangipani', 'clematis', \n    'hibiscus', 'columbine', 'desert-rose', 'tree mallow', 'magnolia', 'cyclamen', \n    'watercress', 'canna lily', 'hippeastrum', 'bee balm', 'pink quill', 'foxglove', \n    'bougainvillea', 'camellia', 'mallow', 'mexican petunia', 'bromelia', 'blanket flower', \n    'trumpet creeper', 'blackberry lily', 'common tulip', 'wild rose', 'morning glory', \n    'monkshood', 'goldenrod', 'lily of the valley', 'primula', 'sunflower', 'lilac hibiscus', \n    'bishop of llandaff', 'gaillardia', 'gazania', 'azalea', 'water lily', 'rose' \n    # (Note: List ini disederhanakan, total harus 104 kelas sesuai dataset asli)\n]\n\ndef decode_image(image_data):\n    \"\"\"Decode JPEG, Normalize ke [0,1], dan Resize\"\"\"\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0  # Normalisasi\n    \n    # PERBAIKAN: Gunakan resize dulu untuk mengubah dimensi 192->128\n    image = tf.image.resize(image, IMAGE_SIZE) \n    \n    # Reshape hanya untuk memastikan bentuk tensor statis untuk TPU\n    image = tf.reshape(image, [*IMAGE_SIZE, 3]) \n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), # tf.string means bytestring\n        \"class\": tf.io.FixedLenFeature([], tf.int64),  # shape [] means scalar\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False # disable order, increase speed\n\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=tf.data.AUTOTUNE)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, \n                          num_parallel_calls=tf.data.AUTOTUNE)\n    return dataset\n\ndef get_training_dataset():\n    dataset = load_dataset(TRAINING_FILENAMES, labeled=True)\n    dataset = dataset.repeat() # dataset harus berulang untuk beberapa epoch\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(tf.data.AUTOTUNE)\n    return dataset\n\ndef get_validation_dataset(ordered=False):\n    dataset = load_dataset(VALIDATION_FILENAMES, labeled=True, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(tf.data.AUTOTUNE)\n    return dataset\n\ndef get_test_dataset(ordered=False):\n    dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(tf.data.AUTOTUNE)\n    return dataset\n\n# Hitung jumlah item\n# REVISI: Menggunakan hardcoded number karena tf.data.cardinality sering return -2 (unknown) pada GCS\nNUM_TRAINING_IMAGES = 12753\nNUM_VALIDATION_IMAGES = 3712\nTEST_IMAGES = 7382\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\n\nprint(f\"Dataset: {NUM_TRAINING_IMAGES} training images, {NUM_VALIDATION_IMAGES} validation images\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:27:01.721817Z","iopub.execute_input":"2025-12-03T13:27:01.722179Z","iopub.status.idle":"2025-12-03T13:27:02.084434Z","shell.execute_reply.started":"2025-12-03T13:27:01.722161Z","shell.execute_reply":"2025-12-03T13:27:02.083836Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n# ==========================================\n# 4. ARSITEKTUR MLP LENGKAP\n# ==========================================","metadata":{}},{"cell_type":"code","source":"\n# Rubrik: Arsitektur MLP (Input, Hidden, Output, Activation)\n\nwith strategy.scope():\n    model = models.Sequential([\n        # 1. INPUT LAYER & PREPROCESSING\n        # Flattening mengubah input (128, 128, 3) menjadi vektor (49152)\n        layers.Input(shape=(*IMAGE_SIZE, 3)),\n        layers.Flatten(), \n        \n        # 2. HIDDEN LAYERS\n        # Layer 1\n        layers.Dense(2048, use_bias=False), # Neuron banyak karena input fitur besar\n        layers.BatchNormalization(),        # Penting untuk konvergensi MLP\n        layers.Activation('relu'),          # Fungsi Aktivasi ReLU\n        layers.Dropout(0.4),                # Mencegah Overfitting\n        \n        # Layer 2\n        layers.Dense(1024, use_bias=False),\n        layers.BatchNormalization(),\n        layers.Activation('relu'),\n        layers.Dropout(0.3),\n        \n        # Layer 3\n        layers.Dense(512, use_bias=False),\n        layers.BatchNormalization(),\n        layers.Activation('relu'),\n        layers.Dropout(0.2),\n\n        # 3. OUTPUT LAYER\n        # 104 Kelas, Aktivasi Softmax untuk Multi-class Classification\n        layers.Dense(104, activation='softmax') \n    ])\n\n    model.compile(\n        optimizer=optimizers.Adam(learning_rate=LEARNING_RATE),\n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:27:02.157336Z","iopub.execute_input":"2025-12-03T13:27:02.157557Z","iopub.status.idle":"2025-12-03T13:27:02.423717Z","shell.execute_reply.started":"2025-12-03T13:27:02.15754Z","shell.execute_reply":"2025-12-03T13:27:02.423115Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ==========================================\n# 5. TRAINING & CALLBACKS\n# ==========================================","metadata":{}},{"cell_type":"code","source":"\n\n# Rubrik: Metode Validasi (Holdout via 'validation_data')\n\n# Callback untuk mengatur learning rate dinamis\nlr_scheduler = callbacks.ReduceLROnPlateau(\n    monitor='val_loss', \n    factor=0.5, \n    patience=3, \n    min_lr=0.00001, \n    verbose=1\n)\n\nearly_stopping = callbacks.EarlyStopping(\n    monitor='val_loss',\n    patience=7,\n    restore_best_weights=True,\n    verbose=1\n)\n\nhistory = model.fit(\n    get_training_dataset(), \n    steps_per_epoch=STEPS_PER_EPOCH,\n    epochs=EPOCHS,\n    validation_data=get_validation_dataset(),\n    callbacks=[lr_scheduler, early_stopping]\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:27:04.782771Z","iopub.execute_input":"2025-12-03T13:27:04.783465Z","iopub.status.idle":"2025-12-03T13:42:09.741647Z","shell.execute_reply.started":"2025-12-03T13:27:04.783439Z","shell.execute_reply":"2025-12-03T13:42:09.740922Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ==========================================\n# 6. ANALISIS KONVERGENSI (Loss/Acc Graphs)\n# ==========================================","metadata":{}},{"cell_type":"code","source":"\n\n# Rubrik: Analisis Konvergensi (Loss vs Epoch)\n\ndef plot_training(history):\n    acc = history.history['sparse_categorical_accuracy']\n    val_acc = history.history['val_sparse_categorical_accuracy']\n    loss = history.history['loss']\n    val_loss = history.history['val_loss']\n    epochs = range(1, len(acc) + 1)\n\n    plt.figure(figsize=(14, 5))\n    \n    # Plot Akurasi\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs, acc, 'b-', label='Training Acc')\n    plt.plot(epochs, val_acc, 'r-', label='Validation Acc')\n    plt.title('Training and Validation Accuracy')\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.legend()\n\n    # Plot Loss\n    plt.subplot(1, 2, 2)\n    plt.plot(epochs, loss, 'b-', label='Training Loss')\n    plt.plot(epochs, val_loss, 'r-', label='Validation Loss')\n    plt.title('Training and Validation Loss')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.legend()\n    \n    plt.tight_layout()\n    plt.show()\n\nplot_training(history)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:43:17.406579Z","iopub.execute_input":"2025-12-03T13:43:17.406899Z","iopub.status.idle":"2025-12-03T13:43:17.953359Z","shell.execute_reply.started":"2025-12-03T13:43:17.406876Z","shell.execute_reply":"2025-12-03T13:43:17.952498Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n# ==========================================\n# 7. METRIK LENGKAP & INTERPRETASI\n# ==========================================","metadata":{}},{"cell_type":"code","source":"\n# Rubrik: Metrik Lengkap (Precision, Recall, F1)\n\nprint(\"\\nMenghitung Metrik Evaluasi...\")\n# Ambil dataset validasi yang terurut untuk evaluasi\nval_ds = get_validation_dataset(ordered=True)\nval_images_ds = val_ds.map(lambda image, label: image)\nval_labels_ds = val_ds.map(lambda image, label: label)\n\n# Ground Truth Labels\ny_true = []\nfor labels in val_labels_ds:\n    y_true.extend(labels.numpy())\ny_true = np.array(y_true)\n\n# Predictions\nprobs = model.predict(val_images_ds)\ny_pred = np.argmax(probs, axis=-1)\n\n# Classification Report\nprint(\"\\nClassification Report (Sample of classes):\")\n# Tampilkan report hanya untuk label yang ada di y_true agar tidak error jika kelas jarang\nprint(classification_report(y_true, y_pred, zero_division=0))\n\n# Confusion Matrix Sederhana\nplt.figure(figsize=(10, 8))\ncm = confusion_matrix(y_true, y_pred)\nsns.heatmap(cm, annot=False, cmap='Blues')\nplt.title('Confusion Matrix (All Classes)')\nplt.ylabel('True Label')\nplt.xlabel('Predicted Label')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:43:28.237856Z","iopub.execute_input":"2025-12-03T13:43:28.238141Z","iopub.status.idle":"2025-12-03T13:43:32.777927Z","shell.execute_reply.started":"2025-12-03T13:43:28.238122Z","shell.execute_reply":"2025-12-03T13:43:32.77728Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ==========================================\n# 8. SUBMISSION (Kaggle)\n# ==========================================","metadata":{}},{"cell_type":"code","source":"\n\nprint(\"Membuat File Submission...\")\ntest_ds = get_test_dataset(ordered=True)\ntest_images_ds = test_ds.map(lambda image, idnum: image)\n\nprobabilities = model.predict(test_images_ds)\npredictions = np.argmax(probabilities, axis=-1)\n\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(NUM_TRAINING_IMAGES))).numpy().astype('U') # Load all ids\n\nnp.savetxt('submission.csv', np.rec.fromarrays([test_ids, predictions]), fmt=['%s', '%d'], delimiter=',', header='id,label', comments='')\nprint(\"Submission.csv berhasil dibuat.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:43:45.429045Z","iopub.execute_input":"2025-12-03T13:43:45.429433Z","iopub.status.idle":"2025-12-03T13:43:53.05985Z","shell.execute_reply.started":"2025-12-03T13:43:45.429404Z","shell.execute_reply":"2025-12-03T13:43:53.059185Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}