{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31194,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T10:13:07.231523Z","iopub.execute_input":"2025-12-03T10:13:07.231838Z","iopub.status.idle":"2025-12-03T10:13:07.276902Z","shell.execute_reply.started":"2025-12-03T10:13:07.231819Z","shell.execute_reply":"2025-12-03T10:13:07.275932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport re\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nfrom sklearn.metrics import classification_report\n\n# Konfigurasi Environment\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'\nwarnings.filterwarnings(\"ignore\")\nprint(f\"TensorFlow Version: {tf.__version__}\")\n\n# Deteksi Hardware\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\n    print(\">> HARDWARE: TPU DETECTED\")\nexcept ValueError:\n    # Fallback ke GPU (MirroredStrategy untuk T4 x2)\n    strategy = tf.distribute.MirroredStrategy()\n    print(\">> HARDWARE: GPU DETECTED\")\n\nREPLICAS = strategy.num_replicas_in_sync\nprint(f\">> Number of accelerators: {REPLICAS}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T10:14:44.134036Z","iopub.execute_input":"2025-12-03T10:14:44.134292Z","iopub.status.idle":"2025-12-03T10:14:44.141212Z","shell.execute_reply.started":"2025-12-03T10:14:44.134272Z","shell.execute_reply":"2025-12-03T10:14:44.140161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 2: CONFIG & CLASSES ---\n\n# Path Handling (Prioritas Lokal untuk GPU agar lebih cepat & stabil)\nBASE_PATH = \"/kaggle/input/tpu-getting-started\"\nif not os.path.exists(BASE_PATH):\n    from kaggle_datasets import KaggleDatasets\n    try:\n        BASE_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started')\n    except:\n        print(\"Warning: GCS path failed, checking local...\")\n\nprint(f\">> Dataset Path: {BASE_PATH}\")\n\n# Hyperparameters\nIMAGE_SIZE = [64, 64]  # Ukuran kecil agar MLP aman dari Out of Memory\nEPOCHS = 20         # Epoch cukup panjang untuk konvergensi\nBATCH_SIZE = 64 * REPLICAS \nLEARNING_RATE = 0.001\n\n# File Patterns\nTRAIN_FILENAMES = tf.io.gfile.glob(BASE_PATH + '/tfrecords-jpeg-192x192/train/*.tfrec')\nVAL_FILENAMES = tf.io.gfile.glob(BASE_PATH + '/tfrecords-jpeg-192x192/val/*.tfrec')\nTEST_FILENAMES = tf.io.gfile.glob(BASE_PATH + '/tfrecords-jpeg-192x192/test/*.tfrec')\n\n# DAFTAR KELAS LENGKAP (104 ITEM)\n# Wajib lengkap agar output layer sesuai dengan label dataset\nCLASSES = [\n    'pink primrose', 'hard-leaved pocket orchid', 'canterbury bells', 'sweet pea', 'wild geranium', 'tiger lily', 'moon orchid', 'bird of paradise', 'monkshood', 'globe thistle',\n    'snapdragon', \"colt's foot\", 'king protea', 'spear thistle', 'yellow iris', 'globe-flower', 'purple coneflower', 'peruvian lily', 'balloon flower', 'giant white arum lily',\n    'fire lily', 'pincushion flower', 'fritillary', 'red ginger', 'grape hyacinth', 'corn poppy', 'prince of wales feathers', 'stemless gentian', 'artichoke', 'sweet william',\n    'carnation', 'garden phlox', 'love in the mist', 'mexican aster', 'alpine sea holly', 'ruby-lipped cattleya', 'cape flower', 'great masterwort', 'siam tulip', 'lenten rose',\n    'barbeton daisy', 'daffodil', 'sword lily', 'poinsettia', 'bolero deep blue', 'wallflower', 'marigold', 'buttercup', 'daisy', 'common dandelion',\n    'petunia', 'wild pansy', 'primula', 'sunflower', 'lilac hibiscus', 'bishop of llandaff', 'gaillardia', 'gazania', 'azalea', 'water lily',\n    'rose', 'thorn apple', 'morning glory', 'passion flower', 'lotus', 'toad lily', 'anthurium', 'frangipani', 'clematis', 'hibiscus',\n    'columbine', 'desert-rose', 'tree mallow', 'magnolia', 'cyclamen ', 'watercress', 'canna lily', 'hippeastrum', 'bee balm', 'pink quill',\n    'foxglove', 'bougainvillea', 'camellia', 'mallow', 'mexican petunia', 'bromelia', 'blanket flower', 'trumpet creeper', 'blackberry lily', 'common tulip', 'wild rose'\n]\n\nprint(f\"Jumlah Kelas: {len(CLASSES)} (Harus 104)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T10:37:39.011710Z","iopub.execute_input":"2025-12-03T10:37:39.011975Z","iopub.status.idle":"2025-12-03T10:37:39.060547Z","shell.execute_reply.started":"2025-12-03T10:37:39.011956Z","shell.execute_reply":"2025-12-03T10:37:39.059624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 3: DATA PIPELINE ---\n\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    # [Rubrik: Preprocessing] Normalisasi [0,1] & Resize\n    image = tf.cast(image, tf.float32) / 255.0\n    image = tf.image.resize(image, IMAGE_SIZE)\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=tf.data.AUTOTUNE)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, \n                          num_parallel_calls=tf.data.AUTOTUNE)\n    return dataset\n\n# Setup Dataset Loading\n# FIX: drop_remainder=True mencegah error shape mismatch pada Multi-GPU\ntrain_dataset = load_dataset(TRAIN_FILENAMES, labeled=True)\ntrain_dataset = train_dataset.repeat().shuffle(2048)\ntrain_dataset = train_dataset.batch(BATCH_SIZE, drop_remainder=True)\ntrain_dataset = train_dataset.prefetch(tf.data.AUTOTUNE)\n\nvalid_dataset = load_dataset(VAL_FILENAMES, labeled=True, ordered=True)\nvalid_dataset = valid_dataset.batch(BATCH_SIZE, drop_remainder=True)\nvalid_dataset = valid_dataset.cache().prefetch(tf.data.AUTOTUNE)\n\ntest_dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=True)\ntest_dataset = test_dataset.batch(BATCH_SIZE)\ntest_dataset = test_dataset.prefetch(tf.data.AUTOTUNE)\n\n# Hitung jumlah gambar untuk steps training\ndef count_data_items(filenames):\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) for filename in filenames]\n    return np.sum(n)\n\nNUM_TRAINING_IMAGES = count_data_items(TRAIN_FILENAMES)\nSTEPS_PER_EPOCH = (NUM_TRAINING_IMAGES // BATCH_SIZE) - 1\n\nprint(f\"Steps per Epoch: {STEPS_PER_EPOCH}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T10:14:44.648582Z","iopub.execute_input":"2025-12-03T10:14:44.648815Z","iopub.status.idle":"2025-12-03T10:14:44.757841Z","shell.execute_reply.started":"2025-12-03T10:14:44.648798Z","shell.execute_reply":"2025-12-03T10:14:44.756751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with strategy.scope():\n    model = tf.keras.Sequential([\n        # Input Layer\n        tf.keras.layers.InputLayer(input_shape=(IMAGE_SIZE[0], IMAGE_SIZE[1], 3)),\n        tf.keras.layers.Flatten(),\n        \n        # Hidden Layers\n        tf.keras.layers.Dense(2048, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.3),\n        \n        tf.keras.layers.Dense(1024, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.3),\n        \n        tf.keras.layers.Dense(512, activation='relu'),\n        \n        # --- PERBAIKAN UTAMA DI SINI ---\n        # Pastikan angka 104 tertulis jelas, atau gunakan len(CLASSES) yang sudah diperbaiki\n        tf.keras.layers.Dense(104, activation='softmax') \n    ])\n\n    model.compile(\n        optimizer='adam',\n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\nmodel.summary()\n# Pastikan di baris paling bawah output summary tertulis: (None, 104)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T10:14:44.907801Z","iopub.execute_input":"2025-12-03T10:14:44.908008Z","iopub.status.idle":"2025-12-03T10:14:44.991498Z","shell.execute_reply.started":"2025-12-03T10:14:44.907991Z","shell.execute_reply":"2025-12-03T10:14:44.990365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with strategy.scope():\n    model = tf.keras.Sequential([\n        # Input Layer\n        tf.keras.layers.InputLayer(input_shape=(IMAGE_SIZE[0], IMAGE_SIZE[1], 3)),\n        tf.keras.layers.Flatten(),\n        \n        # Hidden Layers\n        tf.keras.layers.Dense(2048, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.3),\n        \n        tf.keras.layers.Dense(1024, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.3),\n        \n        tf.keras.layers.Dense(512, activation='relu'),\n        \n        # --- PERBAIKAN UTAMA: ANGKA 104 ---\n        tf.keras.layers.Dense(104, activation='softmax') \n    ])\n\n    model.compile(\n        optimizer='adam',\n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\nmodel.summary()\n# Pastikan Output Shape paling bawah adalah (None, 104)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T10:14:45.196025Z","iopub.execute_input":"2025-12-03T10:14:45.196227Z","iopub.status.idle":"2025-12-03T10:14:45.279333Z","shell.execute_reply.started":"2025-12-03T10:14:45.196210Z","shell.execute_reply":"2025-12-03T10:14:45.278048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 5: TRAINING ---\n\n# Callbacks\nearly_stopping = tf.keras.callbacks.EarlyStopping(\n    monitor='val_loss', patience=10, restore_best_weights=True, verbose=1\n)\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor='val_loss', factor=0.2, patience=3, min_lr=1e-5, verbose=1\n)\n\nprint(\"\\n>>> STARTING TRAINING...\")\nhistory = model.fit(\n    train_dataset,\n    steps_per_epoch=STEPS_PER_EPOCH,\n    epochs=EPOCHS,\n    validation_data=valid_dataset,\n    callbacks=[early_stopping, reduce_lr],\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T10:14:45.707814Z","iopub.execute_input":"2025-12-03T10:14:45.708046Z","iopub.status.idle":"2025-12-03T10:21:55.610933Z","shell.execute_reply.started":"2025-12-03T10:14:45.708029Z","shell.execute_reply":"2025-12-03T10:21:55.609511Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_training(history):\n    # Mengambil metrics dengan aman (handle perbedaan nama key di versi TF berbeda)\n    acc = history.history.get('sparse_categorical_accuracy', history.history.get('accuracy'))\n    val_acc = history.history.get('val_sparse_categorical_accuracy', history.history.get('val_accuracy'))\n    loss = history.history['loss']\n    val_loss = history.history['val_loss']\n    \n    if acc is None:\n        print(\"Metric accuracy tidak ditemukan di history object.\")\n        return\n\n    epochs_range = range(1, len(acc) + 1)\n\n    plt.figure(figsize=(15, 6))\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs_range, acc, label='Training Accuracy')\n    plt.plot(epochs_range, val_acc, label='Validation Accuracy')\n    plt.title('Training and Validation Accuracy')\n    plt.legend(loc='lower right')\n\n    plt.subplot(1, 2, 2)\n    plt.plot(epochs_range, loss, label='Training Loss')\n    plt.plot(epochs_range, val_loss, label='Validation Loss')\n    plt.title('Training and Validation Loss')\n    plt.legend(loc='upper right')\n    plt.show()\n\nprint(\"\\n>>> MENAMPILKAN GRAFIK EVALUASI:\")\nif 'history' in locals():\n    plot_training(history)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T10:39:35.892188Z","iopub.execute_input":"2025-12-03T10:39:35.892461Z","iopub.status.idle":"2025-12-03T10:39:36.077740Z","shell.execute_reply.started":"2025-12-03T10:39:35.892442Z","shell.execute_reply":"2025-12-03T10:39:36.076685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 7: SUBMISSION ---\n\nprint(\"\\n>>> MEMBUAT SUBMISSION FILE...\")\n# Hitung ulang jumlah test images\nNUM_TEST_IMAGES = count_data_items(TEST_FILENAMES)\n\ntest_ds = test_dataset.map(lambda image, idnum: image)\nprobabilities = model.predict(test_ds, verbose=1)\npredictions = np.argmax(probabilities, axis=-1)\n\ntest_ids_ds = test_dataset.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(NUM_TEST_IMAGES))).numpy().astype('U')\n\nsubmission = pd.DataFrame({'id': test_ids, 'label': predictions})\nsubmission.to_csv('submission.csv', index=False)\nprint(\">>> SUKSES! File 'submission.csv' telah dibuat dan siap disubmit.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T10:39:39.475702Z","iopub.execute_input":"2025-12-03T10:39:39.475961Z","iopub.status.idle":"2025-12-03T10:39:41.488303Z","shell.execute_reply.started":"2025-12-03T10:39:39.475943Z","shell.execute_reply":"2025-12-03T10:39:41.487103Z"}},"outputs":[],"execution_count":null}]}