{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31194,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T06:08:06.703065Z","iopub.execute_input":"2025-12-03T06:08:06.703297Z","iopub.status.idle":"2025-12-03T06:08:07.112449Z","shell.execute_reply.started":"2025-12-03T06:08:06.703275Z","shell.execute_reply":"2025-12-03T06:08:07.111639Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1. Inisialisasi Lingkungan & TPU\nLangkah ini bertujuan untuk mendeteksi dan menginisialisasi sistem TPU pada lingkungan Kaggle.","metadata":{}},{"cell_type":"code","source":"import math, re, os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.metrics import classification_report, confusion_matrix\nfrom tensorflow.keras import layers, models, callbacks, optimizers\n\n# Deteksi Hardware TPU\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T06:08:11.885554Z","iopub.execute_input":"2025-12-03T06:08:11.885950Z","iopub.status.idle":"2025-12-03T06:08:42.868291Z","shell.execute_reply.started":"2025-12-03T06:08:11.885930Z","shell.execute_reply":"2025-12-03T06:08:42.867440Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2.Konfigurasi data\nMenurunkan resolusi dari data asli, sehingga menjadi 128x128 dengan tujuan mengurangi beban komputasi arsitektur MLP","metadata":{}},{"cell_type":"code","source":"# Ukuran Input ke Model (Tetap 128x128)\nIMAGE_SIZE = [128, 128] \nEPOCHS = 25\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\n\n# DATA LOADING \nGCS_DS_PATH = '/kaggle/input/tpu-getting-started'\n\n# Ganti ke folder 512x512\nTRAINING_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-512x512/train/*.tfrec')\nVALIDATION_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-512x512/val/*.tfrec')\nTEST_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-512x512/test/*.tfrec')\n\n# Daftar CLASSES:\nCLASSES = ['pink primrose', 'hard-leaved pocket orchid', 'canterbury bells', 'sweet pea', 'wild geranium', 'tiger lily', 'moon orchid', 'bird of paradise', 'monkshood', 'globe thistle', 'snapdragon', \"colt's foot\", 'king protea', 'spear thistle', 'yellow iris', 'globe-flower', 'purple coneflower', 'peruvian lily', 'balloon flower', 'giant white arum lily', 'fire lily', 'pincushion flower', 'fritillary', 'red ginger', 'grape hyacinth', 'corn poppy', 'prince of wales feathers', 'stemless gentian', 'artichoke', 'sweet william', 'carnation', 'garden phlox', 'love in the mist', 'mexican aster', 'alpine sea holly', 'ruby-lipped cattleya', 'cape flower', 'great masterwort', 'siam tulip', 'lenten rose', 'barbeton daisy', 'daffodil', 'sword lily', 'poinsettia', 'bolero deep blue', 'wallflower', 'marigold', 'buttercup', 'daisy', 'common dandelion', 'petunia', 'wild pansy', 'primula', 'sunflower', 'lilac hibiscus', 'bishop of llandaff', 'gaillardia', 'gazania', 'azalea', 'water lily', 'rose', 'thorn apple', 'morning glory', 'passion flower', 'lotus', 'toad lily', 'anthurium', 'frangipani', 'clematis', 'hibiscus', 'columbine', 'desert-rose', 'tree mallow', 'magnolia', 'cyclamen ', 'watercress', 'canna lily', 'hippeastrum', 'bee balm', 'pink-yellow dahlia', 'ball moss', 'foxglove', 'bougainvillea', 'camellia', 'mallow', 'mexican petunia', 'bromelia', 'blanket flower', 'trumpet creeper', 'blackberry lily', 'common tulip', 'wild rose'] \n\nprint(f\"Target Image Size: {IMAGE_SIZE}\")\nprint(f\"Jumlah Kelas Didefinisikan: {len(CLASSES)}\") # Target: 104","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T06:09:13.632572Z","iopub.execute_input":"2025-12-03T06:09:13.633062Z","iopub.status.idle":"2025-12-03T06:09:13.656412Z","shell.execute_reply.started":"2025-12-03T06:09:13.633045Z","shell.execute_reply":"2025-12-03T06:09:13.655500Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Preprocessing Pipeline (Normalisasi)\nMengubah data mentah menjadi format numerik yang siap diproses oleh jaringan saraf.","metadata":{}},{"cell_type":"code","source":"def decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    # 1. Normalisasi: Ubah range pixel dari 0-255 ke 0-1 (Float32)\n    image = tf.cast(image, tf.float32) / 255.0  \n    # 2. Resizing: Pakai method 'reshape' karena source dan target sudah konsisten\n    # Tapi agar aman, kita paksa bentuknya ke [128, 128, 3]\n    image = tf.image.resize(image, IMAGE_SIZE)\n    image = tf.reshape(image, [*IMAGE_SIZE, 3]) \n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False \n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=tf.data.experimental.AUTOTUNE)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, num_parallel_calls=tf.data.experimental.AUTOTUNE)\n    return dataset\n\n# Setup Pipeline\nds_train = load_dataset(TRAINING_FILENAMES, labeled=True).repeat().shuffle(2048).batch(BATCH_SIZE).prefetch(tf.data.experimental.AUTOTUNE)\nds_valid = load_dataset(VALIDATION_FILENAMES, labeled=True, ordered=False).batch(BATCH_SIZE).cache().prefetch(tf.data.experimental.AUTOTUNE)\nds_test = load_dataset(TEST_FILENAMES, labeled=False, ordered=True).batch(BATCH_SIZE).prefetch(tf.data.experimental.AUTOTUNE)\n\nprint(\"Pipeline Data (Resized) Siap.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T06:09:17.212335Z","iopub.execute_input":"2025-12-03T06:09:17.212607Z","iopub.status.idle":"2025-12-03T06:09:17.388258Z","shell.execute_reply.started":"2025-12-03T06:09:17.212587Z","shell.execute_reply":"2025-12-03T06:09:17.387363Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Arsitektur Model Multi-Layer Perceptron\nMembangun struktur jaringan saraf tiruan yang mampu mengklasifikasikan pola gambar menjadi 104 kelas","metadata":{}},{"cell_type":"code","source":"def create_optimized_mlp():\n    model = models.Sequential([\n        # Input Layer (Explicit Input)\n        layers.Input(shape=[*IMAGE_SIZE, 3]),\n        layers.Flatten(name='Input_Flatten'),\n        \n        # Hidden Layer 1\n        layers.Dense(2048, name='Hidden_Layer_1'),\n        layers.BatchNormalization(), \n        layers.Activation('relu'),\n        layers.Dropout(0.3), \n        \n        # Hidden Layer 2\n        layers.Dense(1024, name='Hidden_Layer_2'),\n        layers.BatchNormalization(),\n        layers.Activation('relu'),\n        layers.Dropout(0.3),\n        \n        # Hidden Layer 3\n        layers.Dense(512, name='Hidden_Layer_3'),\n        layers.BatchNormalization(),\n        layers.Activation('relu'),\n        layers.Dropout(0.2),\n        \n        # Output Layer: WAJIB 104 (Sesuai jumlah kelas di dataset)\n        # Jangan gunakan len(CLASSES) jika list namanya tidak lengkap\n        layers.Dense(104, activation='softmax', name='Output_Layer')\n    ])\n    return model\n\n# Inisialisasi Ulang Model\nwith strategy.scope():\n    model = create_optimized_mlp()\n    model.compile(\n        optimizer=optimizers.Adam(learning_rate=0.0001), \n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T06:09:29.027800Z","iopub.execute_input":"2025-12-03T06:09:29.028041Z","iopub.status.idle":"2025-12-03T06:09:29.190747Z","shell.execute_reply.started":"2025-12-03T06:09:29.028020Z","shell.execute_reply":"2025-12-03T06:09:29.189808Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5.Training & Hyperparameter Tuning\nMelatih model menggunakan data latih","metadata":{}},{"cell_type":"code","source":"# Callbacks\n# Reduce LR: Jika validasi loss tidak turun selama 3 epoch, kurangi LR\nlr_scheduler = callbacks.ReduceLROnPlateau(\n    monitor='val_loss', factor=0.5, patience=3, min_lr=1e-6, verbose=1\n)\n# Early Stopping: Stop jika tidak ada perbaikan selama 10 epoch\nearly_stopping = callbacks.EarlyStopping(\n    monitor='val_loss', patience=10, restore_best_weights=True, verbose=1\n)\n\n# Hitung steps per epoch\nNUM_TRAINING_IMAGES = 12753\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\n\nprint(\"Mulai Training (Resized Data)...\")\nhistory = model.fit(\n    ds_train,\n    validation_data=ds_valid,\n    epochs=EPOCHS,\n    steps_per_epoch=STEPS_PER_EPOCH,\n    callbacks=[lr_scheduler, early_stopping]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T06:09:34.154131Z","iopub.execute_input":"2025-12-03T06:09:34.154391Z","iopub.status.idle":"2025-12-03T07:14:28.222384Z","shell.execute_reply.started":"2025-12-03T06:09:34.154371Z","shell.execute_reply":"2025-12-03T07:14:28.220982Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6.Analisis Konvergensi & Evaluasi Metrik\nMengukur kinerja model secara objektif dan mendeteksi kelemahan pada kelas tertentu.","metadata":{}},{"cell_type":"code","source":"# 1. Plot Grafik Konvergensi\ndef plot_convergence(history):\n    acc = history.history['sparse_categorical_accuracy']\n    val_acc = history.history['val_sparse_categorical_accuracy']\n    loss = history.history['loss']\n    val_loss = history.history['val_loss']\n    epochs_range = range(len(acc))\n\n    plt.figure(figsize=(15, 5))\n    \n    # Grafik Akurasi\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs_range, acc, label='Training Accuracy')\n    plt.plot(epochs_range, val_acc, label='Validation Accuracy')\n    plt.title('Accuracy vs Epoch')\n    plt.legend(loc='lower right'); plt.grid(True)\n\n    # Grafik Loss\n    plt.subplot(1, 2, 2)\n    plt.plot(epochs_range, loss, label='Training Loss')\n    plt.plot(epochs_range, val_loss, label='Validation Loss')\n    plt.title('Loss vs Epoch')\n    plt.legend(loc='upper right'); plt.grid(True)\n    \n    plt.show()\n\nplot_convergence(history)\n\n# 2. Laporan Metrik Lengkap (Classification Report)\nprint(\"Menghitung Metrik Lengkap...\")\nds_valid_eval = load_dataset(VALIDATION_FILENAMES, labeled=True, ordered=True).batch(BATCH_SIZE)\ny_true = []\ny_pred = []\n\n# Iterasi dataset validasi\nfor images, labels in ds_valid_eval:\n    preds = model.predict(images, verbose=0)\n    y_true.extend(labels.numpy())\n    y_pred.extend(np.argmax(preds, axis=-1))\n\nprint(\"\\n--- CLASSIFICATION REPORT ---\")\n# Menampilkan Precision, Recall, F1-Score untuk setiap kelas bunga\nprint(classification_report(y_true, y_pred, zero_division=0))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T07:26:38.228857Z","iopub.execute_input":"2025-12-03T07:26:38.229196Z","iopub.status.idle":"2025-12-03T07:26:57.977705Z","shell.execute_reply.started":"2025-12-03T07:26:38.229169Z","shell.execute_reply":"2025-12-03T07:26:57.976355Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Berdasarkan output diatas, nilai Precision dan Recall = 0.00 pada kelas tertentu yang menandakan bahwa model MLP gagal total mengenali varietas bunga tersebut (True Positive = 0). Kegagalan ini terjadi karena arsitektur MLP mengharuskan input gambar diratakan (flatten) menjadi vektor 1 dimensi, yang secara fatal menghilangkan informasi spasial (bentuk dan tekstur) yang krusial untuk membedakan 104 jenis bunga yang sangat mirip. Akibatnya, model mengalami underfitting parah pada kelas-kelas yang sulit dibedakan tersebut dan cenderung hanya memprediksi kelas mayoritas yang polanya lebih sederhana.","metadata":{}},{"cell_type":"markdown","source":"## 7. Prediksi & Submission\nMengimplementasikan model yang telah dilatih pada data uji (Test Set) untuk kebutuhan kompetisi.","metadata":{}},{"cell_type":"code","source":"print('Memproses Prediksi Data Test...')\ntest_ds = load_dataset(TEST_FILENAMES, labeled=False, ordered=True).batch(BATCH_SIZE)\ntest_images_ds = test_ds.map(lambda image, idnum: image)\n\nprobabilities = model.predict(test_images_ds)\npredictions = np.argmax(probabilities, axis=-1)\n\nprint('Membuat file submission.csv...')\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(7382))).numpy().astype('U')\n\nsubmission = pd.DataFrame({'id': test_ids, 'label': predictions})\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Selesai! File siap disubmit.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T07:36:47.118243Z","iopub.execute_input":"2025-12-03T07:36:47.118554Z","iopub.status.idle":"2025-12-03T07:36:55.249247Z","shell.execute_reply.started":"2025-12-03T07:36:47.118536Z","shell.execute_reply":"2025-12-03T07:36:55.248038Z"}},"outputs":[],"execution_count":null}]}