{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# CELL 1 \nimport os\nos.environ[\"TF_CPP_MIN_LOG_LEVEL\"] = \"3\"\n\n# Cuma force protobuf aja, TF 2.16.1 biasanya sudah ada di Kaggle image terbaru\n!pip install -q protobuf==4.25.3 --no-cache-dir\n\nprint(\"SELESAI!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:00:11.154156Z","iopub.execute_input":"2025-12-03T11:00:11.154462Z","iopub.status.idle":"2025-12-03T11:00:14.375551Z","shell.execute_reply.started":"2025-12-03T11:00:11.154435Z","shell.execute_reply":"2025-12-03T11:00:14.374646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport tensorflow as tf\nprint(\"TensorFlow version :\", tf.__version__)   # Harus 2.16.1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:55:30.738303Z","iopub.execute_input":"2025-12-03T09:55:30.738624Z","iopub.status.idle":"2025-12-03T09:55:40.034100Z","shell.execute_reply.started":"2025-12-03T09:55:30.738574Z","shell.execute_reply":"2025-12-03T09:55:40.033307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Inisialisasi TPU\ntry:\n    resolver = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(resolver)\n    tf.tpu.experimental.initialize_tpu_system(resolver)\n    strategy = tf.distribute.TPUStrategy(resolver)\n    print(\"TPU SIAP – 8 cores\")\nexcept:\n    strategy = tf.distribute.get_strategy()\n    print(\"No TPU → GPU/CPU\")\n\nprint(\"Replicas :\", strategy.num_replicas_in_sync)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:55:40.034988Z","iopub.execute_input":"2025-12-03T09:55:40.035533Z","iopub.status.idle":"2025-12-03T09:55:40.041674Z","shell.execute_reply.started":"2025-12-03T09:55:40.035509Z","shell.execute_reply":"2025-12-03T09:55:40.040788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Library lain\nimport numpy as np, pandas as pd, matplotlib.pyplot as plt\nfrom sklearn.metrics import classification_report, confusion_matrix, roc_auc_score\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:55:40.042613Z","iopub.execute_input":"2025-12-03T09:55:40.043391Z","iopub.status.idle":"2025-12-03T09:55:40.487683Z","shell.execute_reply.started":"2025-12-03T09:55:40.043370Z","shell.execute_reply":"2025-12-03T09:55:40.486922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 2: Path Dataset + Hyperparameter\n# Path otomatis di Kaggle (Petals to the Metal - Flower Classification on TPU)\nGCS_PATH = '/kaggle/input/tpu-getting-started'\n\n# Pilih resolusi gambar (untuk MLP kita pakai yang kecil supaya cepat)\nSIZE = '192x192'        # paling cepat & cukup untuk MLP\n# SIZE = '224x224'      # alternatif lebih bagus sedikit\n# SIZE = '331x331'     # lebih lambat\n# SIZE = '512x512'     # terlalu besar untuk MLP\n\nTRAIN_FILES = tf.io.gfile.glob(f'{GCS_PATH}/tfrecords-jpeg-{SIZE}/train/*.tfrec')\nVAL_FILES   = tf.io.gfile.glob(f'{GCS_PATH}/tfrecords-jpeg-{SIZE}/val/*.tfrec')\nTEST_FILES  = tf.io.gfile.glob(f'{GCS_PATH}/tfrecords-jpeg-{SIZE}/test/*.tfrec')\n\nprint(f\"Train files : {len(TRAIN_FILES)} files\")\nprint(f\"Val   files : {len(VAL_FILES)} files\")\nprint(f\"Test  files : {len(TEST_FILES)} files\")\n\n# Hyperparameter (akan kita tuning nanti)\nIMG_SIZE = 128          # kita resize lagi ke 128×128 supaya fitur tidak terlalu banyak\nBATCH_SIZE = 64 * strategy.num_replicas_in_sync   # besar karena TPU suka batch besar\nEPOCHS = 25\nCLASSES = 104","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:55:40.488587Z","iopub.execute_input":"2025-12-03T09:55:40.489127Z","iopub.status.idle":"2025-12-03T09:55:40.548775Z","shell.execute_reply.started":"2025-12-03T09:55:40.489107Z","shell.execute_reply":"2025-12-03T09:55:40.548061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 3: Fungsi Parsing + Augmentasi\nAUTO = tf.data.AUTOTUNE\n\ndef parse_tfrecord(example_proto, labeled=True):\n    feature_description = {\n        'image': tf.io.FixedLenFeature([], tf.string),\n        'class': tf.io.FixedLenFeature([], tf.int64),\n        'id'   : tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example_proto, feature_description)\n    image = tf.image.decode_jpeg(example['image'], channels=3)\n    image = tf.image.resize(image, [IMG_SIZE, IMG_SIZE])\n    image = tf.cast(image, tf.float32) / 255.0                    # Normalisasi [0,1] → KRUSIAL untuk MLP\n    \n    if labeled:\n        label = tf.cast(example['class'], tf.int32)\n        return image, label\n    return image, example['id']\n\ndef augment(image, label):\n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_brightness(image, 0.1)\n    image = tf.image.random_contrast(image, 0.9, 1.1)\n    return image, label\n\ndef load_dataset(filenames, labeled=True, ordered=False, augment_data=False):\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    if not ordered:\n        dataset = dataset.shuffle(2048)\n    dataset = dataset.map(lambda x: parse_tfrecord(x, labeled), num_parallel_calls=AUTO)\n    if augment_data and labeled:\n        dataset = dataset.map(augment, num_parallel_calls=AUTO)\n    return dataset\n\n# Dataset siap pakai\ntrain_ds = load_dataset(TRAIN_FILES, labeled=True, augment_data=True).repeat().batch(BATCH_SIZE).prefetch(AUTO)\nval_ds   = load_dataset(VAL_FILES,   labeled=True).batch(BATCH_SIZE).cache().prefetch(AUTO)\ntest_ds  = load_dataset(TEST_FILES,  labeled=False, ordered=True).batch(BATCH_SIZE).prefetch(AUTO)\n\nSTEPS_PER_EPOCH = 12753 // BATCH_SIZE","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:55:40.550525Z","iopub.execute_input":"2025-12-03T09:55:40.550765Z","iopub.status.idle":"2025-12-03T09:55:41.706517Z","shell.execute_reply.started":"2025-12-03T09:55:40.550748Z","shell.execute_reply":"2025-12-03T09:55:41.705944Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 4: Arsitektur MLP Lengkap\nwith strategy.scope():\n    model = tf.keras.Sequential([\n        tf.keras.layers.Input(shape=(IMG_SIZE, IMG_SIZE, 3)),\n        tf.keras.layers.Flatten(),                                      # Input → vektor 128×128×3 = 49,152 fitur\n        \n        tf.keras.layers.Dense(1024, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.4),\n        \n        tf.keras.layers.Dense(512, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.3),\n        \n        tf.keras.layers.Dense(256, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        \n        tf.keras.layers.Dense(CLASSES, activation='softmax')            # Output 104 kelas\n    ])\n    \n    model.compile(\n        optimizer = tf.keras.optimizers.Adam(learning_rate=0.001),\n        loss      = 'sparse_categorical_crossentropy',\n        metrics   = ['sparse_categorical_accuracy']\n    )\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:55:41.707223Z","iopub.execute_input":"2025-12-03T09:55:41.707441Z","iopub.status.idle":"2025-12-03T09:55:42.140363Z","shell.execute_reply.started":"2025-12-03T09:55:41.707424Z","shell.execute_reply":"2025-12-03T09:55:42.139817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 5: Training\ncallbacks = [\n    tf.keras.callbacks.EarlyStopping(patience=6, restore_best_weights=True),\n    tf.keras.callbacks.ReduceLROnPlateau(patience=3, factor=0.5)\n]\n\nhistory = model.fit(\n    train_ds,\n    epochs=EPOCHS,\n    steps_per_epoch=STEPS_PER_EPOCH,\n    validation_data=val_ds,\n    callbacks=callbacks,\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:55:42.140990Z","iopub.execute_input":"2025-12-03T09:55:42.141200Z","iopub.status.idle":"2025-12-03T10:41:53.539196Z","shell.execute_reply.started":"2025-12-03T09:55:42.141183Z","shell.execute_reply":"2025-12-03T10:41:53.538290Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 6: Plot Loss & Accuracy + Analisis Konvergensi (SUDAH DIPERBAIKI)\nplt.figure(figsize=(14, 5))\n\n# Plot Loss\nplt.subplot(1, 2, 1)\nplt.plot(history.history['loss'], label='Train Loss', linewidth=2)\nplt.plot(history.history['val_loss'], label='Validation Loss', linewidth=2)\nplt.title('Loss vs Epoch', fontsize=14)\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.grid(True, alpha=0.3)\n\n# Plot Accuracy\nplt.subplot(1, 2, 2)\nplt.plot(history.history['sparse_categorical_accuracy'], label='Train Accuracy', linewidth=2)\nplt.plot(history.history['val_sparse_categorical_accuracy'], label='Validation Accuracy', linewidth=2)\nplt.title('Accuracy vs Epoch', fontsize=14)\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.grid(True, alpha=0.3)\n\nplt.tight_layout()\nplt.show()\n\n# Analisis konvergensi otomatis\nprint(\"\\n\" + \"=\"*50)\nprint(\"           ANALISIS KONVERGENSI\")\nprint(\"=\"*50)\nbest_val_acc = max(history.history['val_sparse_categorical_accuracy'])\nbest_epoch = history.history['val_sparse_categorical_accuracy'].index(best_val_acc) + 1\n\nprint(f\"Best Validation Accuracy : {best_val_acc:.4f} (pada epoch {best_epoch})\")\nprint(f\"Final Validation Accuracy: {history.history['val_sparse_categorical_accuracy'][-1]:.4f}\")\n\nif history.history['val_loss'][-1] > min(history.history['val_loss']) + 0.05:\n    print(\"Terdeteksi Overfitting → val_loss naik di akhir training\")\n    print(\"   Solusi: Dropout & BatchNorm sudah membantu, bisa ditambah augmentasi lebih kuat\")\nelif best_val_acc < 0.50:\n    print(\"Terdeteksi Underfitting → model belum cukup kuat\")\n    print(\"   Solusi: tambah layer/neuron atau epochs\")\nelse:\n    print(\"Konvergensi BAGUS → model stabil dan tidak overfit/underfit berat\")\n\nprint(\"=\"*50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T10:41:53.540371Z","iopub.execute_input":"2025-12-03T10:41:53.540682Z","iopub.status.idle":"2025-12-03T10:41:54.027735Z","shell.execute_reply.started":"2025-12-03T10:41:53.540656Z","shell.execute_reply":"2025-12-03T10:41:54.026873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 7: Metrik Lengkap pada Validation Set\nval_images = []\nval_labels = []\nfor img, lbl in val_ds.unbatch():\n    val_images.append(img.numpy())\n    val_labels.append(lbl.numpy())\nval_images = np.array(val_images)\nval_labels = np.array(val_labels)\n\nval_pred_prob = model.predict(val_images)\nval_pred = np.argmax(val_pred_prob, axis=1)\n\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\n\nprint(\"=== METRIK VALIDATION SET (macro average) ===\")\nprint(f\"Accuracy  : {accuracy_score(val_labels, val_pred):.4f}\")\nprint(f\"Precision : {precision_score(val_labels, val_pred, average='macro'):.4f}\")\nprint(f\"Recall    : {recall_score(val_labels, val_pred, average='macro'):.4f}\")\nprint(f\"F1-Score  : {f1_score(val_labels, val_pred, average='macro'):.4f}\")\n\n# AUC multi-class\nval_labels_onehot = tf.keras.utils.to_categorical(val_labels, CLASSES)\nauc = roc_auc_score(val_labels_onehot, val_pred_prob, average='macro')\nprint(f\"AUC (macro) : {auc:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T10:41:54.028490Z","iopub.execute_input":"2025-12-03T10:41:54.028734Z","iopub.status.idle":"2025-12-03T10:42:01.647771Z","shell.execute_reply.started":"2025-12-03T10:41:54.028710Z","shell.execute_reply":"2025-12-03T10:42:01.646844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 8: Prediksi Test Set & Buat submission.csv (SUDAH DIPERBAIKI)\nprint(\"Memproses test set...\")\n\ntest_images = []\ntest_ids = []\n\n# Perbaikan: buat ulang test dataset dengan parse yang benar (tanpa label)\ndef parse_test_tfrecord(example_proto):\n    feature_description = {\n        'image': tf.io.FixedLenFeature([], tf.string),\n        'id'   : tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example_proto, feature_description)\n    image = tf.image.decode_jpeg(example['image'], channels=3)\n    image = tf.image.resize(image, [IMG_SIZE, IMG_SIZE])\n    image = tf.cast(image, tf.float32) / 255.0\n    id_str = example['id']\n    return image, id_str\n\n# Buat ulang test dataset khusus untuk submission\ntest_dataset_fixed = tf.data.TFRecordDataset(TEST_FILES, num_parallel_reads=AUTO)\ntest_dataset_fixed = test_dataset_fixed.map(parse_test_tfrecord, num_parallel_calls=AUTO)\ntest_dataset_fixed = test_dataset_fixed.batch(BATCH_SIZE).prefetch(AUTO)\n\n# Sekarang loop aman\nfor img_batch, id_batch in test_dataset_fixed:\n    test_images.append(img_batch.numpy())\n    # id_batch adalah bytes → decode satu per satu\n    for id_bytes in id_batch.numpy():\n        test_ids.append(id_bytes.decode('utf-8'))\n\n# Gabungkan semua batch gambar\ntest_images = np.concatenate(test_images, axis=0)\n\nprint(f\"Total test images : {len(test_images)}\")\nprint(f\"Total test ids    : {len(test_ids)}\")\n\n# Prediksi\nprint(\"Sedang prediksi test set...\")\ntest_predictions = model.predict(test_images, verbose=1)\ntest_pred_labels = np.argmax(test_predictions, axis=1)\n\n# Buat submission\nsubmission = pd.DataFrame({\n    'id': test_ids,\n    'label': test_pred_labels\n})\n\nsubmission.to_csv('submission.csv', index=False)\nprint(\"submission.csv berhasil dibuat dan siap di-submit!\")\n\n# Tampilkan 10 baris pertama\nsubmission.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T10:42:01.648656Z","iopub.execute_input":"2025-12-03T10:42:01.649195Z","iopub.status.idle":"2025-12-03T10:42:16.150436Z","shell.execute_reply.started":"2025-12-03T10:42:01.649165Z","shell.execute_reply":"2025-12-03T10:42:16.149849Z"}},"outputs":[],"execution_count":null}]}