{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31194,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:17:06.226188Z","iopub.execute_input":"2025-12-03T09:17:06.226448Z","iopub.status.idle":"2025-12-03T09:17:06.268840Z","shell.execute_reply.started":"2025-12-03T09:17:06.226429Z","shell.execute_reply":"2025-12-03T09:17:06.267872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 12S23046 - Anastasya T.B. Siahaan\n# Flower Classification - Neural Network (CNN + MLP)\n\nimport os, math, random\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nimport matplotlib.pyplot as plt\n\nprint(\"TensorFlow:\", tf.__version__)\n\n# ------------------------------------------------------\n# Strategi TPU kalau ada, kalau tidak pakai GPU/CPU\n# ------------------------------------------------------\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\n    print(\"✅ Running on TPU:\", tpu.master())\nexcept:\n    strategy = tf.distribute.get_strategy()\n    print(\"✅ Running on default strategy (GPU/CPU)\")\n\nprint(\"REPLICAS:\", strategy.num_replicas_in_sync)\n\n# ------------------------------------------------------\n# Konstanta utama\n# ------------------------------------------------------\nIMAGE_SIZE   = [192, 192]\nIMG_SIZE     = 192\nN_CLASSES    = 104                 # 104 jenis bunga\nAUTO         = tf.data.AUTOTUNE\nSEED         = 42\nBATCH_SIZE   = 32 * strategy.num_replicas_in_sync\nEPOCHS       = 20                  # boleh dinaikkan ke 25 kalau waktunya cukup\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:17:06.269189Z","iopub.execute_input":"2025-12-03T09:17:06.269347Z","iopub.status.idle":"2025-12-03T09:17:06.274222Z","shell.execute_reply.started":"2025-12-03T09:17:06.269333Z","shell.execute_reply":"2025-12-03T09:17:06.273440Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ------------------------------------------------------\n# Lokasi data TFRecord di Kaggle\n# ------------------------------------------------------\nGCS_PATH = \"/kaggle/input/tpu-getting-started/tfrecords-jpeg-192x192\"\n\ntrain_files = tf.io.gfile.glob(os.path.join(GCS_PATH, \"train/*.tfrec\"))\nval_files   = tf.io.gfile.glob(os.path.join(GCS_PATH, \"val/*.tfrec\"))\ntest_files  = tf.io.gfile.glob(os.path.join(GCS_PATH, \"test/*.tfrec\"))\n\nprint(\"Jumlah file TFRecord\")\nprint(\"Train :\", len(train_files))\nprint(\"Val   :\", len(val_files))\nprint(\"Test  :\", len(test_files))\n\n\n# ------------------------------------------------------\n# Fungsi bantu untuk menghitung jumlah image di TFRecord\n#   (untuk info & laporan)\n# ------------------------------------------------------\ndef count_data_items(filenames):\n    # contoh nama file: flowers00-192x192-230.tfrec -> ambil \"230\"\n    n = [\n        int(os.path.basename(fname).split(\"-\")[-1].split(\".\")[0])\n        for fname in filenames\n    ]\n    return np.sum(n)\n\nNUM_TRAIN_IMAGES = count_data_items(train_files)\nNUM_VAL_IMAGES   = count_data_items(val_files)\nNUM_TEST_IMAGES  = count_data_items(test_files)\n\nprint(\"NUM_TRAIN_IMAGES :\", NUM_TRAIN_IMAGES)\nprint(\"NUM_VAL_IMAGES   :\", NUM_VAL_IMAGES)\nprint(\"NUM_TEST_IMAGES  :\", NUM_TEST_IMAGES)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:17:06.274982Z","iopub.execute_input":"2025-12-03T09:17:06.275150Z","iopub.status.idle":"2025-12-03T09:17:06.308649Z","shell.execute_reply.started":"2025-12-03T09:17:06.275119Z","shell.execute_reply":"2025-12-03T09:17:06.307670Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ------------------------------------------------------\n# TFRecord parser + preprocessing\n#   - decode JPEG\n#   - resize ke 192x192\n#   - normalisasi 0-1\n# ------------------------------------------------------\n\ndef read_tfrecord(example, labeled=True):\n    # ⚠️ PERHATIKAN: gunakan key 'image', 'class', 'id'\n    tfrec_format = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\"   : tf.io.FixedLenFeature([], tf.string),\n    }\n    if labeled:\n        tfrec_format[\"class\"] = tf.io.FixedLenFeature([], tf.int64)\n\n    example = tf.io.parse_single_example(example, tfrec_format)\n\n    # image\n    image = tf.io.decode_jpeg(example[\"image\"], channels=3)\n    image = tf.image.resize(image, IMAGE_SIZE)\n    image = tf.cast(image, tf.float32) / 255.0\n\n    if labeled:\n        # ambil label dari feature 'class'\n        label = tf.cast(example[\"class\"], tf.int32)\n        return image, label\n    else:\n        img_id = example[\"id\"]\n        return image, img_id\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:17:06.309476Z","iopub.execute_input":"2025-12-03T09:17:06.309697Z","iopub.status.idle":"2025-12-03T09:17:06.314285Z","shell.execute_reply.started":"2025-12-03T09:17:06.309679Z","shell.execute_reply":"2025-12-03T09:17:06.313556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================================================\n# Data augmentation (opsional, tapi bagus untuk rubrik)\n# =====================================================\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n\nIMAGE_SIZE = (192, 192)\n\ndata_augmentation = keras.Sequential(\n    [\n        layers.RandomFlip(\"horizontal\"),\n        layers.RandomRotation(0.1),\n        layers.RandomZoom(0.1),\n    ],\n    name=\"data_augmentation\",\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:17:06.314939Z","iopub.execute_input":"2025-12-03T09:17:06.315124Z","iopub.status.idle":"2025-12-03T09:17:06.336370Z","shell.execute_reply.started":"2025-12-03T09:17:06.315109Z","shell.execute_reply":"2025-12-03T09:17:06.335366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ------------------------------------------------------\n# Fungsi membuat dataset dari daftar file TFRecord\n# ------------------------------------------------------\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False  # biar bisa acak dan cepat\n\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(lambda x: read_tfrecord(x, labeled),\n                          num_parallel_calls=AUTO)\n    return dataset\n\n\ndef get_train_dataset():\n    ds = load_dataset(train_files, labeled=True, ordered=False)\n    ds = ds.shuffle(2048, seed=SEED)\n    ds = ds.map(lambda x, y: (data_augmentation(x, training=True), y),\n                num_parallel_calls=AUTO)\n    ds = ds.batch(BATCH_SIZE).prefetch(AUTO)\n    return ds\n\n\ndef get_val_dataset():\n    ds = load_dataset(val_files, labeled=True, ordered=True)\n    ds = ds.batch(BATCH_SIZE).prefetch(AUTO)\n    return ds\n\n\ndef get_test_dataset():\n    ds = load_dataset(test_files, labeled=False, ordered=True)\n    ds = ds.batch(BATCH_SIZE).prefetch(AUTO)\n    return ds\n\n\ntrain_ds = get_train_dataset()\nval_ds   = get_val_dataset()\ntest_ds  = get_test_dataset()\n\nprint(\"✅ Dataset siap (train_ds, val_ds, test_ds)\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:17:06.336710Z","iopub.execute_input":"2025-12-03T09:17:06.336876Z","iopub.status.idle":"2025-12-03T09:17:06.736656Z","shell.execute_reply.started":"2025-12-03T09:17:06.336861Z","shell.execute_reply":"2025-12-03T09:17:06.735773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ------------------------------------------------------\n# Arsitektur CNN + MLP (sesuai materi NN/MLP)\n# CNN = feature extractor, Dense = MLP classifier\n# ------------------------------------------------------\n\ndef build_model():\n    inputs = layers.Input(shape=(IMG_SIZE, IMG_SIZE, 3))\n\n    # Feature extractor (CNN)\n    x = layers.Conv2D(32, (3, 3), padding=\"same\", activation=\"relu\")(inputs)\n    x = layers.BatchNormalization()(x)\n    x = layers.MaxPooling2D((2, 2))(x)\n\n    x = layers.Conv2D(64, (3, 3), padding=\"same\", activation=\"relu\")(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.MaxPooling2D((2, 2))(x)\n\n    x = layers.Conv2D(128, (3, 3), padding=\"same\", activation=\"relu\")(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.MaxPooling2D((2, 2))(x)\n\n    x = layers.Conv2D(256, (3, 3), padding=\"same\", activation=\"relu\")(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.MaxPooling2D((2, 2))(x)\n\n    # Flatten -> masuk ke MLP (Dense layers)\n    x = layers.Flatten()(x)\n\n    x = layers.Dense(512, activation=\"relu\")(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(0.5)(x)\n\n    x = layers.Dense(256, activation=\"relu\")(x)\n    x = layers.Dropout(0.4)(x)\n\n    outputs = layers.Dense(N_CLASSES, activation=\"softmax\")(x)\n\n    model = keras.Model(inputs=inputs, outputs=outputs, name=\"cnn_mlp_flower\")\n    return model\n\n\nwith strategy.scope():\n    model = build_model()\n    model.compile(\n        optimizer = keras.optimizers.Adam(learning_rate=1e-3),\n        loss      = \"sparse_categorical_crossentropy\",\n        metrics   = [\"accuracy\"]\n    )\n\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:17:06.737399Z","iopub.execute_input":"2025-12-03T09:17:06.737564Z","iopub.status.idle":"2025-12-03T09:17:06.862525Z","shell.execute_reply.started":"2025-12-03T09:17:06.737549Z","shell.execute_reply":"2025-12-03T09:17:06.861669Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ======================================================\n# Training + callback (tuning dasar + cegah overfitting)\n# ======================================================\n\nEPOCHS = 20   # boleh ubah ke 25 kalau runtime cukup\n\ncallbacks = [\n    keras.callbacks.EarlyStopping(\n        monitor=\"val_loss\",\n        patience=3,\n        restore_best_weights=True\n    ),\n    keras.callbacks.ReduceLROnPlateau(\n        monitor=\"val_loss\",\n        factor=0.5,\n        patience=2,\n        verbose=1\n    )\n]\n\nhistory = model.fit(\n    train_ds,\n    validation_data=val_ds,\n    epochs=EPOCHS,\n    callbacks=callbacks,\n    verbose=1\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:17:06.863085Z","iopub.execute_input":"2025-12-03T09:17:06.863264Z","iopub.status.idle":"2025-12-03T10:39:03.840325Z","shell.execute_reply.started":"2025-12-03T09:17:06.863250Z","shell.execute_reply":"2025-12-03T10:39:03.839170Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ======================================================\n# Analisis Konvergensi: Loss & Accuracy vs Epoch\n# ======================================================\n\nplt.figure(figsize=(8, 4))\nplt.plot(history.history[\"loss\"],     label=\"Train Loss\")\nplt.plot(history.history[\"val_loss\"], label=\"Val Loss\")\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Loss\")\nplt.title(\"Train vs Val Loss\")\nplt.legend()\nplt.show()\n\nplt.figure(figsize=(8, 4))\nplt.plot(history.history[\"accuracy\"],     label=\"Train Acc\")\nplt.plot(history.history[\"val_accuracy\"], label=\"Val Acc\")\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Accuracy\")\nplt.title(\"Train vs Val Accuracy\")\nplt.legend()\nplt.show()\n\nprint(\"\"\"\nInterpretasi contoh (bisa kamu kembangkan di laporan):\n- Jika train_loss dan val_loss turun lalu stabil -> model belajar dengan baik.\n- Jika train_loss turun tapi val_loss naik tajam -> indikasi overfitting.\n- Jika keduanya masih tinggi -> model underfitting (butuh arsitektur/epoch lebih besar).\n\"\"\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T10:39:03.840785Z","iopub.execute_input":"2025-12-03T10:39:03.840962Z","iopub.status.idle":"2025-12-03T10:39:04.148645Z","shell.execute_reply.started":"2025-12-03T10:39:03.840945Z","shell.execute_reply":"2025-12-03T10:39:04.147532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ======================================================\n# Evaluasi Model di validation set:\n#   - Accuracy\n#   - Precision (macro)\n#   - Recall (macro)\n#   - F1-score (macro)\n# ======================================================\n\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, classification_report\n\ny_true = []\ny_pred = []\n\nfor images, labels in val_ds:\n    preds = model.predict(images, verbose=0)\n    y_true.extend(labels.numpy())\n    y_pred.extend(np.argmax(preds, axis=1))\n\ny_true = np.array(y_true)\ny_pred = np.array(y_pred)\n\nacc  = accuracy_score(y_true, y_pred)\nprec = precision_score(y_true, y_pred, average=\"macro\", zero_division=0)\nrec  = recall_score(y_true, y_pred, average=\"macro\", zero_division=0)\nf1   = f1_score(y_true, y_pred, average=\"macro\", zero_division=0)\n\nprint(f\"Accuracy  : {acc:.4f}\")\nprint(f\"Precision : {prec:.4f}\")\nprint(f\"Recall    : {rec:.4f}\")\nprint(f\"F1-score  : {f1:.4f}\")\n\nprint(\"\\nClassification Report (macro):\\n\")\nprint(classification_report(y_true, y_pred, zero_division=0))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T10:39:04.149129Z","iopub.execute_input":"2025-12-03T10:39:04.149310Z","iopub.status.idle":"2025-12-03T10:39:19.804227Z","shell.execute_reply.started":"2025-12-03T10:39:04.149295Z","shell.execute_reply":"2025-12-03T10:39:19.802937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ======================================================\n# Prediksi test set & membuat submission.csv\n# ======================================================\n\ntest_ids   = []\ntest_preds = []\n\nfor images, ids in test_ds:\n    preds = model.predict(images, verbose=0)\n    test_preds.extend(np.argmax(preds, axis=1))\n    test_ids.extend(ids.numpy())\n\n# decode bytes -> string\ntest_ids = [x.decode(\"utf-8\") for x in test_ids]\n\nsubmission = pd.DataFrame({\n    \"id\": test_ids,\n    \"label\": test_preds\n})\n\nsubmission.to_csv(\"/kaggle/working/submission.csv\", index=False)\nsubmission.head()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}