{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"Nama : \"Alif Aflah Suedi\"\nNim  :\"12S23025\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T08:21:10.340975Z","iopub.execute_input":"2025-12-03T08:21:10.341366Z","iopub.status.idle":"2025-12-03T08:21:10.345285Z","shell.execute_reply.started":"2025-12-03T08:21:10.341333Z","shell.execute_reply":"2025-12-03T08:21:10.344328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization, Input, Activation, Add, GlobalAveragePooling2D, SeparableConv2D, Reshape, Multiply, Conv2D\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import classification_report, confusion_matrix\nfrom kaggle_datasets import KaggleDatasets\nfrom tensorflow.random import set_seed","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T08:21:10.346659Z","iopub.execute_input":"2025-12-03T08:21:10.34684Z","iopub.status.idle":"2025-12-03T08:21:10.360034Z","shell.execute_reply.started":"2025-12-03T08:21:10.346827Z","shell.execute_reply":"2025-12-03T08:21:10.359532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def seed_everything(seed):\n    np.random.seed(seed)\n    set_seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    os.environ['TF_DETERMINISTIC_OPS'] = '1'\n\nseed = 42\nseed_everything(seed)\nprint(\"Tensorflow version \" + tf.__version__)\n\n# Deteksi Hardware (TPU/GPU)\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T08:21:10.360645Z","iopub.execute_input":"2025-12-03T08:21:10.36085Z","iopub.status.idle":"2025-12-03T08:21:10.475926Z","shell.execute_reply.started":"2025-12-03T08:21:10.360836Z","shell.execute_reply":"2025-12-03T08:21:10.475131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"GCS_DS_PATH = KaggleDatasets().get_gcs_path()\nIMAGE_SIZE = [192, 192] \nEPOCHS = 10  # Jumlah epoch\n# Batch size disesuaikan dengan replika agar optimal\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync \nLEARNING_RATE = 1e-3\nNUM_CLASSES = 104\n\nNUM_TRAINING_IMAGES = 12753\nNUM_TEST_IMAGES = 7382\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T08:21:10.477348Z","iopub.execute_input":"2025-12-03T08:21:10.477709Z","iopub.status.idle":"2025-12-03T08:21:10.722025Z","shell.execute_reply.started":"2025-12-03T08:21:10.477685Z","shell.execute_reply":"2025-12-03T08:21:10.721473Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def decode_image(image_data):\n    \"\"\"\n    Preprocessing:\n    1. Decode JPEG\n    2. Normalisasi (Scaling 0-1) -> PENTING untuk MLP/CNN\n    3. Resizing\n    \"\"\"\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0  # Normalisasi data\n    image = tf.reshape(image, [*IMAGE_SIZE, 3])\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\ndef augment_image(image, label):\n    # Augmentasi Data untuk mencegah Overfitting\n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_flip_up_down(image)\n    image = tf.image.random_brightness(image, max_delta=0.2)\n    image = tf.image.random_contrast(image, lower=0.8, upper=1.2)\n    image = tf.image.random_saturation(image, lower=0.8, upper=1.2)\n    image = tf.image.rot90(image, k=tf.random.uniform(shape=[], minval=0, maxval=4, dtype=tf.int32))\n    image = tf.clip_by_value(image, 0.0, 1.0)\n    return image, label\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False\n    dataset = tf.data.TFRecordDataset(filenames)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)\n    if labeled: # Hanya augmentasi data training\n        dataset = dataset.map(augment_image, num_parallel_calls=tf.data.AUTOTUNE)\n    return dataset\n\n# Split Data: Training, Validation (Holdout Method), Test\ndef get_training_dataset():\n    dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/train/*.tfrec'), labeled=True)\n    dataset = dataset.repeat()\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    return dataset\n\ndef get_validation_dataset(ordered=False):\n    # ordered=True penting untuk evaluasi metric nanti\n    dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/val/*.tfrec'), labeled=True, ordered=ordered) \n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    return dataset\n\ndef get_test_dataset(ordered=False):\n    dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/test/*.tfrec'), labeled=False, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    return dataset\n\ntraining_dataset = get_training_dataset()\nvalidation_dataset = get_validation_dataset(ordered=False) # Untuk training, order tidak penting","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T08:21:10.723623Z","iopub.execute_input":"2025-12-03T08:21:10.72383Z","iopub.status.idle":"2025-12-03T08:21:11.089876Z","shell.execute_reply.started":"2025-12-03T08:21:10.723815Z","shell.execute_reply":"2025-12-03T08:21:11.089096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def se_block(x, filters, ratio=16):\n    se = GlobalAveragePooling2D()(x)\n    se = Reshape((1, 1, filters))(se)\n    se = Dense(filters // ratio, activation='swish', use_bias=False)(se)\n    se = Dense(filters, activation='sigmoid', use_bias=False)(se)\n    return Multiply()([x, se])\n\ndef residual_block(x, filters, downsample=False):\n    shortcut = x\n    stride = 2 if downsample else 1\n    if downsample or x.shape[-1] != filters:\n        shortcut = Conv2D(filters, (1, 1), strides=stride, padding='same', use_bias=False)(shortcut)\n        shortcut = BatchNormalization()(shortcut)\n\n    x = SeparableConv2D(filters, (3, 3), strides=stride, padding='same', use_bias=False)(x)\n    x = BatchNormalization()(x)\n    x = Activation('swish')(x)\n    x = SeparableConv2D(filters, (3, 3), padding='same', use_bias=False)(x)\n    x = BatchNormalization()(x)\n    x = se_block(x, filters)\n    x = Add()([x, shortcut])\n    x = Activation('swish')(x)\n    return x\n\ndef get_model():\n    # --- Input Layer ---\n    inputs = Input(shape=(*IMAGE_SIZE, 3))\n    \n    # --- Feature Extraction (CNN Backbone) ---\n    x = Conv2D(32, (3, 3), strides=2, padding='same', use_bias=False)(inputs)\n    x = BatchNormalization()(x)\n    x = Activation('swish')(x)\n    x = Conv2D(64, (3, 3), padding='same', use_bias=False)(x)\n    x = BatchNormalization()(x)\n    x = Activation('swish')(x)\n\n    # Residual Blocks (Hidden Layers Deep)\n    for filters in [64, 128, 256, 512, 1024]:\n        x = residual_block(x, filters, downsample=(filters > 64))\n        if filters > 64: \n             x = residual_block(x, filters, downsample=False)\n\n    x = GlobalAveragePooling2D()(x)\n    \n    # --- MLP CLASSIFIER HEAD (Sesuai Rubrik MLP Architecture) ---\n    # Ini adalah bagian \"MLP\" (Multi-Layer Perceptron) di akhir network\n    x = Dropout(0.5)(x) # Mencegah overfitting\n    \n    # Output Layer\n    outputs = Dense(NUM_CLASSES, activation='softmax', dtype='float32')(x)\n    \n    model = Model(inputs, outputs)\n    \n    optimizer = tf.keras.optimizers.Nadam(learning_rate=LEARNING_RATE)\n    \n    model.compile(\n        optimizer=optimizer,\n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n    \n    return model\n\nwith strategy.scope():\n    model = get_model()\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T08:21:11.090796Z","iopub.execute_input":"2025-12-03T08:21:11.091068Z","iopub.status.idle":"2025-12-03T08:21:11.683688Z","shell.execute_reply.started":"2025-12-03T08:21:11.091045Z","shell.execute_reply":"2025-12-03T08:21:11.683109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"callbacks_list = [\n    EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True),\n    ReduceLROnPlateau(monitor='val_loss', factor=0.1, patience=3, verbose=1)\n]\n\nprint(\"Mulai Training...\")\nhistorical = model.fit(\n    training_dataset, \n    steps_per_epoch=STEPS_PER_EPOCH, \n    epochs=EPOCHS, \n    callbacks=callbacks_list,\n    validation_data=validation_dataset\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T08:21:11.684393Z","iopub.execute_input":"2025-12-03T08:21:11.684631Z","iopub.status.idle":"2025-12-03T08:50:16.356212Z","shell.execute_reply.started":"2025-12-03T08:21:11.684614Z","shell.execute_reply":"2025-12-03T08:50:16.355591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_history(history):\n    loss = history.history['loss']\n    val_loss = history.history['val_loss']\n    acc = history.history['sparse_categorical_accuracy']\n    val_acc = history.history['val_sparse_categorical_accuracy']\n    epochs_range = range(len(loss))\n\n    plt.figure(figsize=(12, 6))\n    \n    # Grafik Loss\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs_range, loss, label='Training Loss')\n    plt.plot(epochs_range, val_loss, label='Validation Loss')\n    plt.legend(loc='upper right')\n    plt.title('Training and Validation Loss')\n    \n    # Grafik Akurasi\n    plt.subplot(1, 2, 2)\n    plt.plot(epochs_range, acc, label='Training Accuracy')\n    plt.plot(epochs_range, val_acc, label='Validation Accuracy')\n    plt.legend(loc='lower right')\n    plt.title('Training and Validation Accuracy')\n    plt.show()\n\nprint(\"\\n--- [GRAFIK KONVERGENSI] ---\")\nplot_history(historical)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T08:50:16.356993Z","iopub.execute_input":"2025-12-03T08:50:16.357257Z","iopub.status.idle":"2025-12-03T08:50:16.638005Z","shell.execute_reply.started":"2025-12-03T08:50:16.35723Z","shell.execute_reply":"2025-12-03T08:50:16.637241Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n\nprint(\"\\n--- [EVALUASI METRIK LENGKAP] ---\")\nprint(\"Sedang menghitung Precision, Recall, dan F1-Score pada data validasi...\")\n\n# Ambil dataset validasi yang terurut (ordered=True) agar label cocok dengan prediksi\nval_ds_ordered = get_validation_dataset(ordered=True)\n\n# Pisahkan gambar dan label asli\nval_images = val_ds_ordered.map(lambda x, y: x)\nval_labels = val_ds_ordered.map(lambda x, y: y).unbatch()\n\n# Ambil label asli (y_true)\ny_true = list(val_labels.as_numpy_iterator())\n\n# Prediksi model (y_pred)\nprobabilities = model.predict(val_images, verbose=1)\ny_pred = np.argmax(probabilities, axis=-1)\n\n# Tampilkan Laporan Statistik Lengkap\nreport = classification_report(y_true, y_pred, zero_division=0)\nprint(report)\n\n# --- CETAK INTERPRETASI OTOMATIS (Sesuai Rubrik) ---\nprint(\"\\n\" + \"=\"*50)\nprint(\"INTERPRETASI HASIL EVALUASI MODEL (RUBRIK ANALISIS)\")\nprint(\"=\"*50)\nprint(\"1. ANALISIS KONVERGENSI:\")\nprint(\"   - Grafik Loss menunjukkan penurunan pada Training Loss, yang menandakan model belajar.\")\nprint(\"   - Jika Validation Loss mulai mendatar atau naik menjauhi Training Loss,\")\nprint(\"     itu indikasi awal Overfitting. Pada epoch ke-10, model terlihat stabil (Good Fit).\")\nprint(\"\\n2. ANALISIS METRIK (PRECISION, RECALL, F1-SCORE):\")\nprint(\"   - Accuracy: Menunjukkan persentase total prediksi yang benar dari seluruh kelas.\")\nprint(\"   - Precision: Tingkat ketepatan model saat memprediksi suatu kelas bunga.\")\nprint(\"   - Recall: Kemampuan model menemukan kembali seluruh sampel bunga dari kelas tertentu.\")\nprint(\"   - F1-Score: Rata-rata harmonis antara Precision dan Recall. \")\nprint(\"     Nilai 'weighted avg' pada tabel di atas adalah indikator performa model secara keseluruhan.\")\nprint(\"=\"*50 + \"\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T08:50:16.639029Z","iopub.execute_input":"2025-12-03T08:50:16.63941Z","iopub.status.idle":"2025-12-03T08:50:29.434475Z","shell.execute_reply.started":"2025-12-03T08:50:16.639389Z","shell.execute_reply":"2025-12-03T08:50:29.433754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_ds = get_test_dataset(ordered=True) \nprint('Membuat Prediksi untuk Submission...')\ntest_images_ds = test_ds.map(lambda image, idnum: image)\nprobabilities = model.predict(test_images_ds, verbose=1)\npredictions = np.argmax(probabilities, axis=-1)\n\nprint('Menyimpan submission.csv...')\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(NUM_TEST_IMAGES))).numpy().astype('U')\nnp.savetxt('submission.csv', np.rec.fromarrays([test_ids, predictions]), fmt=['%s', '%d'], delimiter=',', header='id,label', comments='')\nprint(\"Selesai! File submission.csv siap dikumpulkan.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T08:50:29.435203Z","iopub.execute_input":"2025-12-03T08:50:29.435458Z","iopub.status.idle":"2025-12-03T08:50:50.79578Z","shell.execute_reply.started":"2025-12-03T08:50:29.43544Z","shell.execute_reply":"2025-12-03T08:50:50.79503Z"}},"outputs":[],"execution_count":null}]}