{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31194,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"NIM = \"12S23035\"\nNama = \"Julius Sinaga\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Flower Classicatior - Neural Network","metadata":{}},{"cell_type":"markdown","source":"## Import Library & Deteksi Hardware","metadata":{}},{"cell_type":"code","source":"import math, re, os\nimport tensorflow as tf\nimport numpy as np\nfrom matplotlib import pyplot as plt\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.metrics import f1_score, precision_score, recall_score, classification_report\n\n# Mendeteksi dan menginisialisasi akselerator TPU untuk mengaktifkan pelatihan paralel pada 8 core secara otomatis.\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Konfigurasi & Path Data","metadata":{}},{"cell_type":"code","source":"# --- KONFIGURASI ---\n# Menentukan resolusi gambar (192x192); harus konsisten dengan path dataset yang dipilih di bawah.\nIMAGE_SIZE = [192, 192]\n\n# Menentukan jumlah putaran training; angka ini bisa diubah untuk Hyperparameter Tuning (poin rubrik).\nEPOCHS = 20\n\n# Mengatur ukuran batch otomatis dikali 8 agar optimal dijalankan pada seluruh core TPU.\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\n\n# --- DATA PATH ---\n# Mengambil jalur Google Cloud Storage karena TPU membutuhkan akses data cloud, bukan lokal.\nGCS_DS_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started')\n\n# Mendapatkan daftar nama file TFRecord untuk Train, Validation, dan Test dari server GCS.\nTRAINING_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-{}x{}/train/*.tfrec'.format(IMAGE_SIZE[0], IMAGE_SIZE[1]))\nVALIDATION_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-{}x{}/val/*.tfrec'.format(IMAGE_SIZE[0], IMAGE_SIZE[1]))\nTEST_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-{}x{}/test/*.tfrec'.format(IMAGE_SIZE[0], IMAGE_SIZE[1]))\n\n# --- DAFTAR KELAS ---\n# Daftar nama 104 jenis bunga yang unik; digunakan untuk menentukan jumlah neuron di Output Layer.\nCLASSES = [\n    'pink primrose', 'hard-leaved pocket orchid', 'canterbury bells', 'sweet pea', 'wild geranium', 'tiger lily', 'moon orchid', 'bird of paradise', 'monkshood', 'globe thistle', 'snapdragon', \"colt's foot\", 'king protea', 'spear thistle', 'yellow iris', 'globe-flower', 'purple coneflower', 'peruvian lily', 'balloon flower', 'giant white arum lily', 'fire lily',\n    'pincushion flower', 'fritillary', 'red ginger', 'grape hyacinth', 'corn poppy', 'prince of wales feathers', 'stemless gentian', 'artichoke', 'sweet william', 'carnation', 'garden phlox', 'love in the mist', 'cosmos', 'alpine sea holly', 'ruby-lipped cattleya', 'cape flower', 'great masterwort', 'siam tulip', 'lenten rose', 'barberton daisy', 'daffodil',\n    'sword lily', 'poinsettia', 'bolero deep blue', 'wallflower', 'marigold', 'buttercup', 'daisy', 'common dandelion', 'petunia', 'wild pansy', 'primula', 'sunflower', 'lilac hibiscus', 'bishop of llandaff', 'gaura', 'geranium', 'orange dahlia', 'pink-yellow dahlia', 'cautleya spicata', 'japanese anemone', 'black-eyed susan',\n    'silverbush', 'californian poppy', 'osteospermum', 'spring crocus', 'iris', 'windflower', 'tree poppy', 'gazania', 'azalea', 'water lily', 'rose', 'thorn apple', 'morning glory', 'passion flower', 'lotus', 'toad lily', 'anthurium', 'frangipani', 'clematis', 'hibiscus', 'columbine',\n    'desert-rose', 'tree mallow', 'magnolia', 'cyclamen', 'watercress', 'canna lily', 'hippeastrum', 'bee balm', 'pink quill', 'foxglove', 'bougainvillea', 'camellia', 'mallow', 'mexican petunia', 'bromelia', 'blanket flower', 'trumpet creeper', 'blackberry lily', 'common tulip', 'wild rose'\n]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Fungsi Preprocessing","metadata":{}},{"cell_type":"code","source":"def decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    # Preprocessing & Fitur - Normalisasi pixel (0-255) menjadi (0-1) agar MLP bekerja optimal\n    image = tf.cast(image, tf.float32) / 255.0\n    image = tf.reshape(image, [*IMAGE_SIZE, 3]) # Pastikan dimensi gambar konsisten\n    return image\n\ndef read_labeled_tfrecord(example):\n    # Format data untuk Training/Validasi (punya label 'class')\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image']) # Dekode gambar\n    label = tf.cast(example['class'], tf.int32) # Ambil label kelas\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    # Format data untuk Test (hanya punya 'id', tidak ada label)\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id'] # Ambil ID untuk file submission\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False # Matikan urutan agar pembacaan data lebih cepat (untuk training)\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=tf.data.experimental.AUTOTUNE)\n    dataset = dataset.with_options(ignore_order)\n    # Mapping fungsi pembacaan data secara paralel\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, num_parallel_reads=tf.data.experimental.AUTOTUNE)\n    return dataset","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Inisialisasi Dataset","metadata":{}},{"cell_type":"code","source":"def get_training_dataset():\n    dataset = load_dataset(TRAINING_FILENAMES, labeled=True)\n    dataset = dataset.repeat() \n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(tf.data.experimental.AUTOTUNE)\n    return dataset\n\ndef get_validation_dataset(ordered=False):\n    dataset = load_dataset(VALIDATION_FILENAMES, labeled=True, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(tf.data.experimental.AUTOTUNE)\n    return dataset\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False \n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=tf.data.experimental.AUTOTUNE)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(\n        read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, \n        num_parallel_calls=tf.data.experimental.AUTOTUNE\n    )\n    return dataset\n\ndef get_test_dataset(ordered=False):\n    dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(tf.data.experimental.AUTOTUNE)\n    return dataset\n\n# Inisialisasi Dataset\ntraining_dataset = get_training_dataset()\nvalidation_dataset = get_validation_dataset()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Pembangunan Model MLP","metadata":{}},{"cell_type":"code","source":"# Menghitung berapa kali \"langkah\" training per epoch (Total Gambar / Ukuran Batch)\nNUM_TRAINING_IMAGES = 12753\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\n\n# Membuka cakupan strategi agar model diduplikasi ke seluruh core TPU\nwith strategy.scope():\n    model = tf.keras.Sequential([\n        # Mengubah gambar 3D (192x192x3) menjadi 1 garis lurus vektor (Flatten)\n        tf.keras.layers.Flatten(input_shape=[*IMAGE_SIZE, 3]),\n        \n        # Layer besar dengan 512 neuron dan aktivasi ReLU\n        tf.keras.layers.Dense(512, activation='relu'),\n        tf.keras.layers.BatchNormalization(), # Menstabilkan bobot agar training lebih cepat\n        tf.keras.layers.Dropout(0.3),         # Mematikan 30% neuron secara acak agar tidak overfitting\n        \n        # Layer menengah\n        tf.keras.layers.Dense(256, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.3),\n        \n        # Layer kecil sebelum output\n        tf.keras.layers.Dense(128, activation='relu'),\n        tf.keras.layers.Dropout(0.2),\n\n        # 104 Neuron (sesuai jumlah bunga) dengan Softmax untuk probabilitas\n        tf.keras.layers.Dense(len(CLASSES), activation='softmax')\n    ])\n    \n    # Kompilasi model dengan Optimizer Adam dan Loss function untuk label integer\n    model.compile(\n        optimizer='adam',\n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\n# Menampilkan struktur model untuk memastikan arsitektur sudah benar\nmodel.summary()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training Model","metadata":{}},{"cell_type":"code","source":"print(\"Mulai Training Model MLP...\")\nhistory = model.fit(\n    training_dataset, \n    steps_per_epoch=STEPS_PER_EPOCH, \n    epochs=EPOCHS, \n    validation_data=validation_dataset\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Analisis Konvergensi","metadata":{}},{"cell_type":"code","source":"def display_training_curves(training, validation, title, subplot):\n    if subplot%10==1: \n        plt.subplots(figsize=(10,10), facecolor='#F0F0F0')\n        plt.tight_layout()\n    ax = plt.subplot(subplot)\n    ax.set_facecolor('#F8F8F8')\n    ax.plot(training)\n    ax.plot(validation)\n    ax.set_title('Model '+ title)\n    ax.set_ylabel(title)\n    ax.set_xlabel('epoch')\n    ax.legend(['train', 'valid.'])\n\n# Menampilkan Grafik Loss dan Akurasi\ndisplay_training_curves(history.history['loss'], history.history['val_loss'], 'Loss', 211)\ndisplay_training_curves(history.history['sparse_categorical_accuracy'], history.history['val_sparse_categorical_accuracy'], 'Accuracy', 212)\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Evaluasi Metrik Lengkap","metadata":{}},{"cell_type":"code","source":"# --- EVALUASI METRIK LENGKAP ---\nprint(\"Melakukan evaluasi metrik lengkap pada Data Validasi...\")\n\n# Siapkan data validasi agar urut (ordered=True penting agar label cocok dengan prediksi)\nds_val = get_validation_dataset(ordered=True)\nimages_ds = ds_val.map(lambda image, label: image)\nlabels_ds = ds_val.map(lambda image, label: label).unbatch()\n\n# Ambil Label Asli (y_true)\n# Kita ambil semua data validasi (3712 gambar)\ny_true = next(iter(labels_ds.batch(3712))).numpy()\n\n# Lakukan Prediksi (y_pred)\nprint(\"Sedang memprediksi...\")\nprobabilities = model.predict(images_ds)\ny_pred = np.argmax(probabilities, axis=-1)\n\n# Tampilkan Laporan Klasifikasi (Precision, Recall, F1-Score)\n# Parameter 'zero_division=0' ditambahkan untuk mencegah warning jika ada kelas yang tidak tertebak sama sekali\nprint(classification_report(y_true, y_pred, target_names=CLASSES, zero_division=0))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"test_ds = get_test_dataset(ordered=True) # Wajib ordered agar ID sesuai urutan\n\n# Lakukan Prediksi dan Ambil Label Terbesar (Argmax)\npredictions = np.argmax(model.predict(test_ds.map(lambda x, i: x)), axis=-1)\n\n# Ambil semua ID gambar\nids = next(iter(test_ds.map(lambda x, i: i).unbatch().batch(7382))).numpy().astype('U')\n\n# Simpan ke CSV\nnp.savetxt('submission.csv', np.rec.fromarrays([ids, predictions]), fmt=['%s', '%d'], delimiter=',', header='id,label', comments='')\nprint('Selesai! submission.csv siap didownload.')","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}