{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-02T12:59:40.636276Z","iopub.execute_input":"2025-12-02T12:59:40.636465Z","iopub.status.idle":"2025-12-02T12:59:42.871215Z","shell.execute_reply.started":"2025-12-02T12:59:40.636448Z","shell.execute_reply":"2025-12-02T12:59:42.869701Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n## STEP 1 – Import dan Konfigurasi Global","metadata":{"execution":{"iopub.status.busy":"2025-12-03T04:51:35.659776Z","iopub.execute_input":"2025-12-03T04:51:35.659962Z","iopub.status.idle":"2025-12-03T04:51:41.719480Z","shell.execute_reply.started":"2025-12-03T04:51:35.659944Z","shell.execute_reply":"2025-12-03T04:51:41.718381Z"}}},{"cell_type":"markdown","source":"Step ini memuat library, set ukuran gambar, batch size, jumlah kelas, seed, dan path dataset Kaggle.","metadata":{}},{"cell_type":"code","source":"!pip install \"protobuf==3.20.3\" --quiet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:16:55.624289Z","iopub.execute_input":"2025-12-03T09:16:55.624739Z","iopub.status.idle":"2025-12-03T09:16:59.120210Z","shell.execute_reply.started":"2025-12-03T09:16:55.624708Z","shell.execute_reply":"2025-12-03T09:16:59.119060Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, re, numpy as np, pandas as pd, tensorflow as tf\nimport matplotlib.pyplot as plt\n\nfrom tensorflow.keras import layers, models, Input, Model\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom sklearn.metrics import (\n    accuracy_score,\n    precision_recall_fscore_support,\n    f1_score,\n    roc_auc_score,\n)\n\nprint(\"TensorFlow version:\", tf.__version__)\n\n# Konfigurasi dasar\nAUTO        = tf.data.AUTOTUNE\nIMAGE_SIZE  = [224, 224]\nBATCH_SIZE  = 32\nNUM_CLASSES = 104\nSEED        = 42\n\ntf.random.set_seed(SEED)\nnp.random.seed(SEED)\n\n# Path dataset di Kaggle\nBASE_DIR     = \"/kaggle/input/tpu-getting-started\"\nTFREC_DIR    = os.path.join(BASE_DIR, \"tfrecords-jpeg-224x224\")\nTRAIN_FILES  = tf.io.gfile.glob(os.path.join(TFREC_DIR, \"train/*.tfrec\"))\nVAL_FILES    = tf.io.gfile.glob(os.path.join(TFREC_DIR, \"val/*.tfrec\"))\nTEST_FILES   = tf.io.gfile.glob(os.path.join(TFREC_DIR, \"test/*.tfrec\"))\n\nprint(\"Jumlah file train:\", len(TRAIN_FILES))\nprint(\"Jumlah file val  :\", len(VAL_FILES))\nprint(\"Jumlah file test :\", len(TEST_FILES))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T04:51:41.721791Z","iopub.execute_input":"2025-12-03T04:51:41.722065Z","iopub.status.idle":"2025-12-03T04:52:01.438201Z","shell.execute_reply.started":"2025-12-03T04:51:41.722037Z","shell.execute_reply":"2025-12-03T04:52:01.437242Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## STEP 2 – Fungsi Menghitung Total Sample","metadata":{}},{"cell_type":"markdown","source":"Step ini untuk validasi agar mengetahui “jumlah gambar train/val/test” di laporan.","metadata":{}},{"cell_type":"code","source":"def count_data_items(filenames):\n    \"\"\"Menghitung total sample berdasarkan angka di nama file TFRecord.\"\"\"\n    n = 0\n    for f in filenames:\n        m = re.search(r\"-([0-9]+)\\.tfrec\", f)\n        if m:\n            n += int(m.group(1))\n    return n\n\nNUM_TRAIN = count_data_items(TRAIN_FILES)\nNUM_VAL   = count_data_items(VAL_FILES)\nNUM_TEST  = count_data_items(TEST_FILES)\n\nprint(\"Total train images:\", NUM_TRAIN)\nprint(\"Total val images  :\", NUM_VAL)\nprint(\"Total test images :\", NUM_TEST)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T04:52:01.439101Z","iopub.execute_input":"2025-12-03T04:52:01.439743Z","iopub.status.idle":"2025-12-03T04:52:01.446039Z","shell.execute_reply.started":"2025-12-03T04:52:01.439713Z","shell.execute_reply":"2025-12-03T04:52:01.445145Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## STEP 3 – Decoder & Parser TFRecord (Preprocessing & Normalisasi)","metadata":{}},{"cell_type":"markdown","source":"baca TFRecord → decode JPEG → normalisasi [0,1] → resize 224×224.\nUntuk train/val kita baca image + class, untuk test image + id.","metadata":{}},{"cell_type":"code","source":"def decode_image(image_bytes):\n    \"\"\"Decode JPEG ke tensor float32 [0,1] berukuran IMAGE_SIZE.\"\"\"\n    image = tf.image.decode_jpeg(image_bytes, channels=3)\n    image = tf.image.convert_image_dtype(image, tf.float32)  # otomatis /255\n    image = tf.image.resize(image, IMAGE_SIZE)\n    return image\n\ndef parse_tfrecord_train(example):\n    \"\"\"Parser untuk TFRecord train/val: image + label class (int).\"\"\"\n    feature_description = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, feature_description)\n    image = decode_image(example[\"image\"])\n    label = tf.cast(example[\"class\"], tf.int32)\n    return image, label\n\ndef parse_tfrecord_test(example):\n    \"\"\"Parser untuk TFRecord test: image + id (string).\"\"\"\n    feature_description = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, feature_description)\n    image = decode_image(example[\"image\"])\n    image_id = example[\"id\"]\n    return image, image_id\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T04:52:01.450703Z","iopub.execute_input":"2025-12-03T04:52:01.450949Z","iopub.status.idle":"2025-12-03T04:52:01.484102Z","shell.execute_reply.started":"2025-12-03T04:52:01.450929Z","shell.execute_reply":"2025-12-03T04:52:01.483114Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## STEP 4 – Membangun tf.data Pipeline (Holdout Validation + Augmentasi)","metadata":{}},{"cell_type":"markdown","source":"Train: ada augmentasi (flip, brightness, contrast).\nVal & test: tanpa augmentasi, hanya normalisasi.","metadata":{}},{"cell_type":"code","source":"def get_train_dataset():\n    ds = tf.data.TFRecordDataset(TRAIN_FILES, num_parallel_reads=AUTO)\n    ds = ds.map(parse_tfrecord_train, num_parallel_calls=AUTO)\n\n    # Augmentasi ringan untuk meningkatkan generalisasi\n    def augment(image, label):\n        image = tf.image.random_flip_left_right(image)\n        image = tf.image.random_brightness(image, 0.2)\n        image = tf.image.random_contrast(image, 0.7, 1.3)\n        return image, label\n\n    ds = ds.map(augment, num_parallel_calls=AUTO)\n    ds = ds.shuffle(2048, seed=SEED)\n    ds = ds.batch(BATCH_SIZE)\n    ds = ds.prefetch(AUTO)\n    return ds\n\ndef get_val_dataset():\n    ds = tf.data.TFRecordDataset(VAL_FILES, num_parallel_reads=AUTO)\n    ds = ds.map(parse_tfrecord_train, num_parallel_calls=AUTO)\n    ds = ds.batch(BATCH_SIZE)\n    ds = ds.cache()\n    ds = ds.prefetch(AUTO)\n    return ds\n\ndef get_test_dataset():\n    ds = tf.data.TFRecordDataset(TEST_FILES, num_parallel_reads=AUTO)\n    ds = ds.map(parse_tfrecord_test, num_parallel_calls=AUTO)\n    ds = ds.batch(BATCH_SIZE)\n    ds = ds.prefetch(AUTO)\n    return ds\n\ntrain_ds = get_train_dataset()\nval_ds   = get_val_dataset()\ntest_ds  = get_test_dataset()\n\ntrain_ds, val_ds, test_ds\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T04:52:07.726861Z","iopub.execute_input":"2025-12-03T04:52:07.727160Z","iopub.status.idle":"2025-12-03T04:52:07.897419Z","shell.execute_reply.started":"2025-12-03T04:52:07.727136Z","shell.execute_reply":"2025-12-03T04:52:07.896760Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## STEP 5 – Cek Visual Beberapa Gambar Train","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 8))\nfor images, labels in train_ds.take(1):\n    for i in range(9):\n        plt.subplot(3, 3, i + 1)\n        plt.imshow(images[i].numpy())\n        plt.title(f\"Label: {int(labels[i].numpy())}\")\n        plt.axis(\"off\")\nplt.suptitle(\"Contoh gambar dari train set\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T04:52:17.631221Z","iopub.execute_input":"2025-12-03T04:52:17.632144Z","iopub.status.idle":"2025-12-03T04:52:20.890424Z","shell.execute_reply.started":"2025-12-03T04:52:17.632111Z","shell.execute_reply":"2025-12-03T04:52:20.889484Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## STEP 6 – Bangun Model: EfficientNetB0 + MLP Head","metadata":{}},{"cell_type":"markdown","source":"CNN (EfficientNetB0) hanya untuk ekstraksi fitur → keluar vektor 1280 dimensi.\nMLP head: beberapa Dense (ReLU) + Dropout + Output Softmax.","metadata":{}},{"cell_type":"code","source":"def build_cnn_mlp_model(hidden_units=[512, 256], dropout_rate=0.4, lr=1e-4):\n    \"\"\"\n    CNN EfficientNetB0 (pretrained ImageNet) sebagai feature extractor,\n    kemudian MLP classifier di atasnya (sesuai rubrik MLP).\n    \"\"\"\n    base_model = EfficientNetB0(\n        include_top=False,\n        weights=\"imagenet\",\n        input_shape=(IMAGE_SIZE[0], IMAGE_SIZE[1], 3),\n        pooling=\"avg\",  # langsung jadi vektor fitur\n    )\n    base_model.trainable = False  # tahap awal: freeze CNN\n\n    inputs = Input(shape=(IMAGE_SIZE[0], IMAGE_SIZE[1], 3), name=\"input_image\")\n    x = base_model(inputs, training=False)  # shape (None, 1280)\n\n    # MLP head: dense + relu + dropout\n    for i, units in enumerate(hidden_units):\n        x = layers.Dense(units, activation=\"relu\", name=f\"dense_{i+1}\")(x)\n        x = layers.Dropout(dropout_rate, name=f\"dropout_{i+1}\")(x)\n\n    outputs = layers.Dense(NUM_CLASSES, activation=\"softmax\", name=\"output\")(x)\n\n    model = Model(inputs, outputs, name=\"EffNetB0_MLP\")\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=lr),\n        loss=\"sparse_categorical_crossentropy\",\n        metrics=[\"accuracy\"],\n    )\n    return model\n\nmodel = build_cnn_mlp_model()\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T04:52:29.684747Z","iopub.execute_input":"2025-12-03T04:52:29.685054Z","iopub.status.idle":"2025-12-03T04:52:31.433288Z","shell.execute_reply.started":"2025-12-03T04:52:29.685029Z","shell.execute_reply":"2025-12-03T04:52:31.432358Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## STEP 7 – Hyperparameter Tuning ","metadata":{}},{"cell_type":"markdown","source":"Kita coba 3 kombinasi MLP: jumlah neuron, dropout, learning rate.\nMetrik pilihan: Macro F1 di validation","metadata":{}},{"cell_type":"code","source":"configs = [\n    {\"hidden_units\": [512, 256],      \"dropout\": 0.4, \"lr\": 1e-4},\n    {\"hidden_units\": [1024, 512],     \"dropout\": 0.5, \"lr\": 1e-4},\n    {\"hidden_units\": [512, 256, 128], \"dropout\": 0.4, \"lr\": 5e-5},\n]\n\nEPOCHS_TUNING = 3  # pendek, hanya untuk memilih konfigurasi terbaik\n\nbest_f1 = -1\nbest_cfg = None\nbest_model = None\nbest_history = None\n\nfor i, cfg in enumerate(configs, start=1):\n    print(f\"\\n=== Training CONFIG {i}/{len(configs)}: {cfg} ===\")\n    m = build_cnn_mlp_model(\n        hidden_units=cfg[\"hidden_units\"],\n        dropout_rate=cfg[\"dropout\"],\n        lr=cfg[\"lr\"],\n    )\n\n    h = m.fit(\n        train_ds,\n        epochs=EPOCHS_TUNING,\n        validation_data=val_ds,\n        verbose=1,\n    )\n\n    # Hitung Macro F1 di validation\n    y_true, y_pred = [], []\n    for x_batch, y_batch in val_ds:\n        probs = m.predict(x_batch, verbose=0)\n        preds = np.argmax(probs, axis=1)\n        y_true.extend(y_batch.numpy())\n        y_pred.extend(preds)\n\n    y_true = np.array(y_true)\n    y_pred = np.array(y_pred)\n\n    macro_f1 = f1_score(y_true, y_pred, average=\"macro\")\n    acc = accuracy_score(y_true, y_pred)\n\n    print(f\"Val Accuracy: {acc:.4f} | Val Macro F1: {macro_f1:.4f}\")\n\n    if macro_f1 > best_f1:\n        best_f1 = macro_f1\n        best_cfg = cfg\n        best_model = m\n        best_history = h\n\nprint(\"\\n=== HASIL TUNING ===\")\nprint(\"Config terbaik:\", best_cfg)\nprint(\"Macro F1 terbaik (val):\", best_f1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T04:52:39.502467Z","iopub.execute_input":"2025-12-03T04:52:39.502807Z","iopub.status.idle":"2025-12-03T06:26:55.330930Z","shell.execute_reply.started":"2025-12-03T04:52:39.502782Z","shell.execute_reply":"2025-12-03T06:26:55.329990Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## STEP 8 – Fine-Tuning CNN (Unfreeze Sebagian Layer Atas)","metadata":{}},{"cell_type":"markdown","source":"Setelah MLP head bagus, kita buka sebagian besar layer terakhir EfficientNet dan latih dengan learning rate kecil supaya performa naik tanpa hancur.","metadata":{}},{"cell_type":"code","source":"# Ambil base_model dari best_model\nbase_model = None\nfor layer in best_model.layers:\n    if isinstance(layer, tf.keras.Model) and layer.name.startswith(\"efficientnetb0\"):\n        base_model = layer\n        break\n\nprint(\"Base model ditemukan:\", base_model is not None)\n\nif base_model is not None:\n    base_model.trainable = True\n    # freeze layer awal, buka 1/3 terakhir\n    fine_tune_at = len(base_model.layers) * 2 // 3\n    for i, layer in enumerate(base_model.layers):\n        layer.trainable = (i >= fine_tune_at)\n\nbest_model.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-5),\n    loss=\"sparse_categorical_crossentropy\",\n    metrics=[\"accuracy\"],\n)\n\nEPOCHS_FINE_TUNE = 10  # bisa kamu naikkan kalau waktunya cukup\n\nhistory_ft = best_model.fit(\n    train_ds,\n    epochs=EPOCHS_FINE_TUNE,\n    validation_data=val_ds,\n    verbose=1,\n)\n\n# gabungkan history tuning + fine-tuning\nfull_history = {}\nfor k in best_history.history.keys():\n    full_history[k] = best_history.history[k] + history_ft.history[k]\n ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T06:30:14.564217Z","iopub.execute_input":"2025-12-03T06:30:14.564988Z","iopub.status.idle":"2025-12-03T08:32:11.644023Z","shell.execute_reply.started":"2025-12-03T06:30:14.564956Z","shell.execute_reply":"2025-12-03T08:32:11.642778Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## STEP 9 – Plot Loss & Accuracy","metadata":{}},{"cell_type":"markdown","source":"step ini untuk menjelaskan di laporan apakah model overfitting / underfitting.","metadata":{}},{"cell_type":"code","source":"epochs = range(1, len(full_history[\"loss\"]) + 1)\n\nplt.figure(figsize=(12, 5))\n\nplt.subplot(1, 2, 1)\nplt.plot(epochs, full_history[\"loss\"], label=\"Train Loss\")\nplt.plot(epochs, full_history[\"val_loss\"], label=\"Val Loss\")\nplt.xlabel(\"Epoch\"); plt.ylabel(\"Loss\")\nplt.title(\"Loss vs Epoch\")\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs, full_history[\"accuracy\"], label=\"Train Acc\")\nplt.plot(epochs, full_history[\"val_accuracy\"], label=\"Val Acc\")\nplt.xlabel(\"Epoch\"); plt.ylabel(\"Accuracy\")\nplt.title(\"Accuracy vs Epoch\")\nplt.legend()\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T08:37:33.566662Z","iopub.execute_input":"2025-12-03T08:37:33.567144Z","iopub.status.idle":"2025-12-03T08:37:34.030030Z","shell.execute_reply.started":"2025-12-03T08:37:33.567114Z","shell.execute_reply":"2025-12-03T08:37:34.029287Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## STEP 10 – Evaluasi Lengkap: Accuracy, Precision, Recall, F1, AUC","metadata":{}},{"cell_type":"markdown","source":"lakukan evaluasi pada model","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix\nimport numpy as np\n\ny_true = []\ny_pred = []\n\n# PAKAI val_ds, BUKAN test_ds\nfor images, labels in val_ds:\n    probs = best_model.predict(images, verbose=0)\n    preds = np.argmax(probs, axis=1)\n\n    # extend = masukin satu-satu (bukan list di dalam list)\n    y_pred.extend(preds)\n    y_true.extend(labels.numpy())\n\ny_true = np.array(y_true)\ny_pred = np.array(y_pred)\n\nprint(\"Shapes:\", y_true.shape, y_pred.shape)\n\nreport = classification_report(\n    y_true,\n    y_pred,\n    digits=4,\n    zero_division=0\n)\nprint(\"=== Classification Report (VAL SET) ===\")\nprint(report)\n\ncm = confusion_matrix(y_true, y_pred)\nprint(\"\\nConfusion Matrix shape:\", cm.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T08:58:16.163447Z","iopub.execute_input":"2025-12-03T08:58:16.163793Z","iopub.status.idle":"2025-12-03T09:00:27.645193Z","shell.execute_reply.started":"2025-12-03T08:58:16.163770Z","shell.execute_reply":"2025-12-03T09:00:27.644316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import csv\nimport numpy as np\n\n# Gunakan model terbaik\nmodel_for_submission = best_model\n\nsubmission_filename = \"submission.csv\"\n\nwith open(submission_filename, mode='w', newline='') as file:\n    writer = csv.writer(file)\n    writer.writerow([\"id\", \"label\"])\n\n    for images, image_ids in test_ds:\n        probs = model_for_submission.predict(images, verbose=0)\n        preds = np.argmax(probs, axis=1)\n\n        for img_id, pred in zip(image_ids.numpy(), preds):\n            writer.writerow([img_id.decode(\"utf-8\"), pred])\n\nprint(\"Submission file saved →\", submission_filename)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:03:35.044988Z","iopub.execute_input":"2025-12-03T09:03:35.045701Z","iopub.status.idle":"2025-12-03T09:07:56.630875Z","shell.execute_reply.started":"2025-12-03T09:03:35.045668Z","shell.execute_reply":"2025-12-03T09:07:56.629971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}