{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31194,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### 12S23043 Grace Tiodora","metadata":{}},{"cell_type":"markdown","source":"## 1. IMPORT LIBRARY & SETUP","metadata":{}},{"cell_type":"code","source":"\n\nimport os\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\n\nfrom sklearn.metrics import (\n    classification_report,\n    confusion_matrix,\n    f1_score,\n    precision_score,\n    recall_score,\n    roc_auc_score\n)\nfrom sklearn.preprocessing import label_binarize\n\nprint(\"TensorFlow version:\", tf.__version__)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:02:19.951239Z","iopub.execute_input":"2025-12-03T09:02:19.951539Z","iopub.status.idle":"2025-12-03T09:02:19.955463Z","shell.execute_reply.started":"2025-12-03T09:02:19.951511Z","shell.execute_reply":"2025-12-03T09:02:19.954707Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. LOAD DATASET TFRECORD (PAKAI RESOLUSI 192x192)","metadata":{}},{"cell_type":"code","source":"\n\nDATA_DIR = Path(\"/kaggle/input/tpu-getting-started\")\n\nIMAGE_SIZE = [192, 192]     # resolusi yang tersedia di dataset\nBATCH_SIZE = 32\nAUTO = tf.data.AUTOTUNE\n\nTFREC_DIR = DATA_DIR / f\"tfrecords-jpeg-{IMAGE_SIZE[0]}x{IMAGE_SIZE[1]}\"\n\nprint(os.listdir(TFREC_DIR))\n\nTRAINING_FILENAMES   = tf.io.gfile.glob(str(TFREC_DIR / \"train/*.tfrec\"))\nVALIDATION_FILENAMES = tf.io.gfile.glob(str(TFREC_DIR / \"val/*.tfrec\"))\nTEST_FILENAMES       = tf.io.gfile.glob(str(TFREC_DIR / \"test/*.tfrec\"))\n\nprint(\"Train files:\", len(TRAINING_FILENAMES))\nprint(\"Val files  :\", len(VALIDATION_FILENAMES))\nprint(\"Test files :\", len(TEST_FILENAMES))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:03:15.904770Z","iopub.execute_input":"2025-12-03T09:03:15.904992Z","iopub.status.idle":"2025-12-03T09:03:15.937044Z","shell.execute_reply.started":"2025-12-03T09:03:15.904975Z","shell.execute_reply":"2025-12-03T09:03:15.936260Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. FUNGSI DECODE DATASET","metadata":{}},{"cell_type":"code","source":"\n\nNUM_CLASSES = 104  # jumlah kelas bunga\n\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.image.resize(image, IMAGE_SIZE)\n    image = tf.cast(image, tf.float32) / 255.0\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFR_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64)\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFR_FORMAT)\n    return decode_image(example[\"image\"]), tf.cast(example[\"class\"], tf.int32)\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFR_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string)\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFR_FORMAT)\n    return decode_image(example[\"image\"]), example[\"id\"]\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    options = tf.data.Options()\n    options.experimental_deterministic = ordered\n\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    dataset = dataset.with_options(options)\n    dataset = dataset.map(\n        read_labeled_tfrecord if labeled else read_unlabeled_tfrecord,\n        num_parallel_calls=AUTO\n    )\n    return dataset\n\ndef get_dataset(filenames, labeled=True, ordered=False, shuffle=False):\n    ds = load_dataset(filenames, labeled=labeled, ordered=ordered)\n    if shuffle:\n        ds = ds.shuffle(2048)\n    ds = ds.batch(BATCH_SIZE)\n    ds = ds.prefetch(AUTO)\n    return ds\n\ntrain_ds = get_dataset(TRAINING_FILENAMES, labeled=True, shuffle=True)\nval_ds   = get_dataset(VALIDATION_FILENAMES, labeled=True)\ntest_ds  = get_dataset(TEST_FILENAMES, labeled=False)\n\nprint(train_ds)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:04:16.239118Z","iopub.execute_input":"2025-12-03T09:04:16.239376Z","iopub.status.idle":"2025-12-03T09:04:16.382864Z","shell.execute_reply.started":"2025-12-03T09:04:16.239358Z","shell.execute_reply":"2025-12-03T09:04:16.382020Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. MLP MODEL (HIGH PERFORMANCE)","metadata":{}},{"cell_type":"code","source":"\n\ntf.keras.backend.clear_session()\n\nH1 = 1024\nH2 = 512\nH3 = 256\n\nD1 = 0.3\nD2 = 0.3\nD3 = 0.2\n\nLEARNING_RATE = 5e-4\nEPOCHS = 12\n\ninputs = layers.Input(shape=(*IMAGE_SIZE, 3))\n\nx = layers.Flatten()(inputs)\n\nx = layers.Dense(H1, activation=\"relu\")(x)\nx = layers.BatchNormalization()(x)\nx = layers.Dropout(D1)(x)\n\nx = layers.Dense(H2, activation=\"relu\")(x)\nx = layers.BatchNormalization()(x)\nx = layers.Dropout(D2)(x)\n\nx = layers.Dense(H3, activation=\"relu\")(x)\nx = layers.BatchNormalization()(x)\nx = layers.Dropout(D3)(x)\n\noutputs = layers.Dense(NUM_CLASSES, activation=\"softmax\")(x)\n\nmodel = models.Model(inputs, outputs)\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=LEARNING_RATE),\n    loss=\"sparse_categorical_crossentropy\",\n    metrics=[\"accuracy\"]\n)\n\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:04:30.517275Z","iopub.execute_input":"2025-12-03T09:04:30.517530Z","iopub.status.idle":"2025-12-03T09:04:30.851429Z","shell.execute_reply.started":"2025-12-03T09:04:30.517514Z","shell.execute_reply":"2025-12-03T09:04:30.850529Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  5. TRAINING","metadata":{}},{"cell_type":"code","source":"\ncallbacks = [\n    tf.keras.callbacks.EarlyStopping(\n        monitor=\"val_loss\",\n        patience=3,\n        restore_best_weights=True,\n        verbose=1\n    ),\n    tf.keras.callbacks.ReduceLROnPlateau(\n        monitor=\"val_loss\",\n        factor=0.5,\n        patience=2,\n        verbose=1\n    ),\n]\n\nhistory = model.fit(\n    train_ds,\n    validation_data=val_ds,\n    epochs=EPOCHS,\n    callbacks=callbacks,\n    verbose=1\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:05:29.420923Z","iopub.execute_input":"2025-12-03T09:05:29.421150Z","iopub.status.idle":"2025-12-03T09:28:03.403572Z","shell.execute_reply.started":"2025-12-03T09:05:29.421134Z","shell.execute_reply":"2025-12-03T09:28:03.402145Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  6. PLOT LOSS & ACCURACY","metadata":{}},{"cell_type":"code","source":"\nhist = history.history\nepochs = range(1, len(hist[\"loss\"]) + 1)\n\nplt.figure(figsize=(15,5))\n\nplt.subplot(1,2,1)\nplt.plot(epochs, hist[\"loss\"], label=\"Train Loss\")\nplt.plot(epochs, hist[\"val_loss\"], label=\"Val Loss\")\nplt.legend(); plt.grid()\nplt.title(\"Loss Plot\")\n\nplt.subplot(1,2,2)\nplt.plot(epochs, hist[\"accuracy\"], label=\"Train Acc\")\nplt.plot(epochs, hist[\"val_accuracy\"], label=\"Val Acc\")\nplt.legend(); plt.grid()\nplt.title(\"Accuracy Plot\")\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:28:08.708648Z","iopub.execute_input":"2025-12-03T09:28:08.708992Z","iopub.status.idle":"2025-12-03T09:28:08.977348Z","shell.execute_reply.started":"2025-12-03T09:28:08.708972Z","shell.execute_reply":"2025-12-03T09:28:08.976493Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. EVALUASI MODEL","metadata":{}},{"cell_type":"code","source":"\ny_true = []\ny_pred_prob = []\n\nfor x_batch, y_batch in val_ds:\n    preds = model.predict(x_batch, verbose=0)\n    y_pred_prob.append(preds)\n    y_true.append(y_batch.numpy())\n\ny_true = np.concatenate(y_true)\ny_pred_prob = np.concatenate(y_pred_prob)\ny_pred = np.argmax(y_pred_prob, axis=1)\n\nacc = np.mean(y_pred == y_true)\nprec = precision_score(y_true, y_pred, average=\"macro\", zero_division=0)\nrec = recall_score(y_true, y_pred, average=\"macro\", zero_division=0)\nf1 = f1_score(y_true, y_pred, average=\"macro\", zero_division=0)\n\nprint(\"Accuracy :\", acc)\nprint(\"Precision:\", prec)\nprint(\"Recall   :\", rec)\nprint(\"F1 Score :\", f1)\n\n# AUC macro\ny_true_bin = label_binarize(y_true, classes=list(range(NUM_CLASSES)))\nauc = roc_auc_score(y_true_bin, y_pred_prob, multi_class=\"ovr\")\nprint(\"AUC :\", auc)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:28:14.373460Z","iopub.execute_input":"2025-12-03T09:28:14.373791Z","iopub.status.idle":"2025-12-03T09:28:24.108701Z","shell.execute_reply.started":"2025-12-03T09:28:14.373769Z","shell.execute_reply":"2025-12-03T09:28:24.107389Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 8. SUBMISSION","metadata":{}},{"cell_type":"code","source":"\n\ntest_ids = []\ntest_preds = []\n\nfor images, ids in test_ds:\n    probs = model.predict(images, verbose=0)\n    preds = np.argmax(probs, axis=1)\n\n    ids = [i.decode(\"utf-8\") for i in ids.numpy()]\n    test_ids.extend(ids)\n    test_preds.extend(preds)\n\nsubmission = pd.DataFrame({\n    \"id\": test_ids,\n    \"label\": test_preds\n})\n\nsubmission.to_csv(\"submission.csv\", index=False)\nsubmission.head()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:34:07.558486Z","iopub.execute_input":"2025-12-03T09:34:07.558781Z","iopub.status.idle":"2025-12-03T09:34:26.688335Z","shell.execute_reply.started":"2025-12-03T09:34:07.558763Z","shell.execute_reply":"2025-12-03T09:34:26.687169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"=== RESULT SUMMARY ===\")\nprint(\"Accuracy :\", acc)\nprint(\"F1 Score :\", f1)\nprint(\"AUC      :\", auc)\nprint(\"\\nsubmission.csv created!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:34:34.795937Z","iopub.execute_input":"2025-12-03T09:34:34.796233Z","iopub.status.idle":"2025-12-03T09:34:34.800556Z","shell.execute_reply.started":"2025-12-03T09:34:34.796213Z","shell.execute_reply":"2025-12-03T09:34:34.799589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}