{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-02T15:07:16.037744Z","iopub.execute_input":"2025-12-02T15:07:16.038045Z","iopub.status.idle":"2025-12-02T15:07:16.389027Z","shell.execute_reply.started":"2025-12-02T15:07:16.038023Z","shell.execute_reply":"2025-12-02T15:07:16.387614Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 12S23046 - Anastasya T.B. Siahaan\n# Flower Classification - Neural Network (MLP)\n\nimport os, math, random\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nimport matplotlib.pyplot as plt\n\n\nprint(\"TensorFlow:\", tf.__version__)\n\n# Strategy TPU kalau ada, kalau tidak pakai GPU/CPU\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\n    print(\"✅ Running on TPU:\", tpu.master())\nexcept:\n    strategy = tf.distribute.get_strategy()\n    print(\"✅ Running on default strategy (GPU/CPU)\")\n\nprint(\"REPLICAS:\", strategy.num_replicas_in_sync)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T15:07:16.391059Z","iopub.execute_input":"2025-12-02T15:07:16.391577Z","iopub.status.idle":"2025-12-02T15:07:39.558991Z","shell.execute_reply.started":"2025-12-02T15:07:16.391549Z","shell.execute_reply":"2025-12-02T15:07:39.557804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Path dataset dari halaman kompetisi Kaggle\nDATA_DIR = \"/kaggle/input/tpu-getting-started\"\nBASE_DIR = os.path.join(DATA_DIR, \"tfrecords-jpeg-192x192\")\n\nprint(\"DATA_DIR :\", DATA_DIR)\nprint(\"BASE_DIR :\", BASE_DIR)\n\ntrain_files = tf.io.gfile.glob(os.path.join(BASE_DIR, \"train/*.tfrec\"))\nval_files   = tf.io.gfile.glob(os.path.join(BASE_DIR, \"val/*.tfrec\"))\ntest_files  = tf.io.gfile.glob(os.path.join(BASE_DIR, \"test/*.tfrec\"))\n\nprint(\"Train TFRecords:\", len(train_files))\nprint(\"Val TFRecords  :\", len(val_files))\nprint(\"Test TFRecords :\", len(test_files))\n\nassert len(train_files) > 0\nassert len(val_files)   > 0\nassert len(test_files)  > 0\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T15:07:39.559939Z","iopub.execute_input":"2025-12-02T15:07:39.560575Z","iopub.status.idle":"2025-12-02T15:07:39.582361Z","shell.execute_reply.started":"2025-12-02T15:07:39.560551Z","shell.execute_reply":"2025-12-02T15:07:39.5814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# TFRecord parser + preprocessing\nAUTO       = tf.data.AUTOTUNE\nIMG_SIZE   = 192\nN_CLASSES  = 104\nBATCH_SIZE = 32 * strategy.num_replicas_in_sync\n\ndef read_tfrecord(example, labeled=True):\n    \"\"\"\n    Struktur TFRecord:\n      - image : bytestring JPG\n      - class : label (train/val)\n      - id    : string id (test)\n    \"\"\"\n    if labeled:\n        TFREC_FORMAT = {\n            \"image\": tf.io.FixedLenFeature([], tf.string),\n            \"class\": tf.io.FixedLenFeature([], tf.int64),\n        }\n    else:\n        TFREC_FORMAT = {\n            \"image\": tf.io.FixedLenFeature([], tf.string),\n            \"id\":    tf.io.FixedLenFeature([], tf.string),\n        }\n\n    example = tf.io.parse_single_example(example, TFREC_FORMAT)\n\n    # decode jpeg -> resize -> float32\n    image = tf.io.decode_jpeg(example[\"image\"], channels=3)\n    image = tf.image.resize(image, (IMG_SIZE, IMG_SIZE))\n    image = tf.cast(image, tf.float32)  # normalisasi 0–255 ke float\n\n    if labeled:\n        label = tf.cast(example[\"class\"], tf.int32)\n        return image, label\n    else:\n        img_id = example[\"id\"]\n        return image, img_id\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ds = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    ds = ds.map(lambda x: read_tfrecord(x, labeled), num_parallel_calls=AUTO)\n    if not ordered:\n        ds = ds.shuffle(2048)\n    ds = ds.batch(BATCH_SIZE).prefetch(AUTO)\n    return ds\n\ntrain_ds = load_dataset(train_files, labeled=True, ordered=False)\nval_ds   = load_dataset(val_files,   labeled=True, ordered=True)\ntest_ds  = load_dataset(test_files,  labeled=False, ordered=True)\n\nprint(\"✅ Dataset ready\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T15:07:39.584065Z","iopub.execute_input":"2025-12-02T15:07:39.584268Z","iopub.status.idle":"2025-12-02T15:07:39.81222Z","shell.execute_reply.started":"2025-12-02T15:07:39.584252Z","shell.execute_reply":"2025-12-02T15:07:39.811551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"images_batch, labels_batch = next(iter(train_ds))\n\nplt.figure(figsize=(10, 10))\nfor i in range(9):\n    plt.subplot(3, 3, i+1)\n    plt.imshow(images_batch[i].numpy().astype(\"uint8\"))\n    plt.title(int(labels_batch[i].numpy()))\n    plt.axis(\"off\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T15:07:39.812955Z","iopub.execute_input":"2025-12-02T15:07:39.813167Z","iopub.status.idle":"2025-12-02T15:07:42.103108Z","shell.execute_reply.started":"2025-12-02T15:07:39.813148Z","shell.execute_reply":"2025-12-02T15:07:42.102017Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Pembangunan Model: MLP dengan arsitektur lengkap\ndef build_mlp():\n    model = keras.Sequential([\n        layers.Input(shape=(IMG_SIZE, IMG_SIZE, 3)),\n        # normalisasi 0–1\n        layers.Rescaling(1./255),\n\n        # Hidden layers (fitur tinggi)\n        layers.Flatten(),\n        layers.Dense(1024, activation='relu'),\n        layers.Dense(512,  activation='relu'),\n        layers.Dense(256,  activation='relu'),\n        layers.Dropout(0.3),\n\n        # Output 104 kelas bunga (softmax)\n        layers.Dense(N_CLASSES, activation='softmax')\n    ])\n    return model\n\nwith strategy.scope():\n    model = build_mlp()\n    model.compile(\n        optimizer=keras.optimizers.Adam(learning_rate=1e-3),\n        loss=\"sparse_categorical_crossentropy\",\n        metrics=[\"accuracy\"]\n    )\n\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T15:07:42.104186Z","iopub.execute_input":"2025-12-02T15:07:42.104473Z","iopub.status.idle":"2025-12-02T15:07:42.791339Z","shell.execute_reply.started":"2025-12-02T15:07:42.104456Z","shell.execute_reply":"2025-12-02T15:07:42.790179Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# Training + callback (tuning dasar + cegah overfitting)\n# ==========================================\nEPOCHS = 25\n\ncallbacks = [\n    keras.callbacks.EarlyStopping(\n        monitor=\"val_loss\", patience=3, restore_best_weights=True\n    ),\n    keras.callbacks.ReduceLROnPlateau(\n        monitor=\"val_loss\", factor=0.5, patience=2\n    )\n]\n\n# --- hitung steps_per_epoch secara eksplisit ---\nNUM_TRAIN_IMAGES = 12753\nNUM_VAL_IMAGES   = 3712\n\n# ceil(N / BATCH_SIZE) tanpa import math\n\nBATCH_SIZE = 128 * strategy.num_replicas_in_sync\n\ntrain_steps = (NUM_TRAIN_IMAGES + BATCH_SIZE - 1) // BATCH_SIZE\nval_steps   = (NUM_VAL_IMAGES   + BATCH_SIZE - 1) // BATCH_SIZE\n\nprint(\"train_steps:\", train_steps, \"val_steps:\", val_steps)\n\n# repeat() + steps_per_epoch -> tidak \"input ran out of data\"\ntrain_ds_rep = train_ds.repeat()\nval_ds_rep   = val_ds.repeat()\n\nhistory = model.fit(\n    train_ds_rep,\n    validation_data=val_ds_rep,\n    epochs=EPOCHS,\n    steps_per_epoch=train_steps,\n    validation_steps=val_steps,\n    callbacks=callbacks,\n    verbose=1\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T15:07:42.79241Z","iopub.execute_input":"2025-12-02T15:07:42.79272Z","iopub.status.idle":"2025-12-02T15:56:13.198211Z","shell.execute_reply.started":"2025-12-02T15:07:42.792696Z","shell.execute_reply":"2025-12-02T15:56:13.197357Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Grafik Loss & Accuracy untuk deteksi overfitting/underfitting\nplt.figure()\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title(\"Loss vs Epoch\")\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Loss\")\nplt.legend([\"Train Loss\", \"Val Loss\"])\nplt.show()\n\nplt.figure()\nplt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.title(\"Accuracy vs Epoch\")\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Accuracy\")\nplt.legend([\"Train Acc\", \"Val Acc\"])\nplt.show()\n\nprint(\"\"\"\nInterpretasi:\n- Train_loss & val_loss sama-sama turun -> model belajar baik.\n- Train_loss turun tapi val_loss naik tajam -> indikasi overfitting.\n- Keduanya tinggi -> model masih underfitting.\n\"\"\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T15:56:13.199083Z","iopub.execute_input":"2025-12-02T15:56:13.19937Z","iopub.status.idle":"2025-12-02T15:56:13.502266Z","shell.execute_reply.started":"2025-12-02T15:56:13.199339Z","shell.execute_reply":"2025-12-02T15:56:13.500954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import (\n    accuracy_score, precision_score, recall_score, f1_score,\n    classification_report\n)\n\ny_true = []\ny_pred = []\n\nfor images, labels in val_ds:\n    preds = model.predict(images, verbose=0)\n    y_true.extend(labels.numpy())\n    y_pred.extend(tf.argmax(preds, axis=1).numpy())\n\nacc  = accuracy_score(y_true, y_pred)\nprec = precision_score(y_true, y_pred, average='macro', zero_division=0)\nrec  = recall_score(y_true, y_pred, average='macro', zero_division=0)\nf1   = f1_score(y_true, y_pred, average='macro', zero_division=0)\n\nprint(\"Accuracy  :\", acc)\nprint(\"Precision :\", prec)\nprint(\"Recall    :\", rec)\nprint(\"F1 Score  :\", f1)\n\nprint(\"\\nClassification Report (macro):\\n\")\nprint(classification_report(y_true, y_pred, zero_division=0))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T15:56:13.503019Z","iopub.execute_input":"2025-12-02T15:56:13.503223Z","iopub.status.idle":"2025-12-02T15:56:44.98435Z","shell.execute_reply.started":"2025-12-02T15:56:13.503207Z","shell.execute_reply":"2025-12-02T15:56:44.983489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"###### Prediksi test set & membuat file submission.csv\ntest_ids   = []\ntest_preds = []\n\nfor images, ids in test_ds:\n    preds = model.predict(images, verbose=0)\n    test_preds.extend(tf.argmax(preds, axis=1).numpy())\n    test_ids.extend(ids.numpy())\n\ntest_ids = [x.decode(\"utf-8\") for x in test_ids]\n\nsubmission = pd.DataFrame({\n    \"id\": test_ids,\n    \"label\": test_preds\n})\n\nsubmission.to_csv(\"/kaggle/working/submission.csv\", index=False)\nprint(\"✅ submission.csv saved to /kaggle/working/submission.csv\")\nsubmission.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T15:56:44.986223Z","iopub.execute_input":"2025-12-02T15:56:44.986463Z","iopub.status.idle":"2025-12-02T15:57:47.851462Z","shell.execute_reply.started":"2025-12-02T15:56:44.986448Z","shell.execute_reply":"2025-12-02T15:57:47.85012Z"}},"outputs":[],"execution_count":null}]}