{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:28:08.568378Z","iopub.execute_input":"2025-12-03T09:28:08.568812Z","iopub.status.idle":"2025-12-03T09:28:08.902335Z","shell.execute_reply.started":"2025-12-03T09:28:08.568784Z","shell.execute_reply":"2025-12-03T09:28:08.901769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom kaggle_datasets import KaggleDatasets\n\nfrom sklearn.metrics import (\n    classification_report,\n    accuracy_score,\n    precision_score,\n    recall_score,\n    f1_score,\n    roc_auc_score\n)\nfrom sklearn.preprocessing import label_binarize\n\nprint(\"TensorFlow version:\", tf.__version__)\n\n# Strategy: pakai semua GPU kalau ada, kalau tidak fallback ke CPU\ngpus = tf.config.list_physical_devices('GPU')\nprint(\"GPUs detected:\", gpus)\n\nif gpus:\n    try:\n        for gpu in gpus:\n            tf.config.experimental.set_memory_growth(gpu, True)\n    except Exception as e:\n        print(\"Cannot set memory growth:\", e)\n\n    strategy = tf.distribute.MirroredStrategy()\nelse:\n    print(\"No GPU found, using default CPU strategy.\")\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS:\", strategy.num_replicas_in_sync)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:28:08.902957Z","iopub.execute_input":"2025-12-03T09:28:08.903203Z","iopub.status.idle":"2025-12-03T09:28:14.335994Z","shell.execute_reply.started":"2025-12-03T09:28:08.903187Z","shell.execute_reply":"2025-12-03T09:28:14.335253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Path dataset dari Kaggle\ntry:\n    GCS_DS_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started')\nexcept:\n    GCS_DS_PATH = \"/kaggle/input/tpu-getting-started\"\n\nprint(\"GCS path:\", GCS_DS_PATH)\n\nIMAGE_SIZE = [192, 192]\nNUM_CLASSES = 104\n\nNUM_TRAIN_IMAGES = 12753\nNUM_VAL_IMAGES   = 3712\nNUM_TEST_IMAGES  = 7382\n\nEPOCHS_FINAL = 10   # training akhir\nEPOCHS_TUNE  = 3    # per config saat tuning\n\nBASE_BATCH_SIZE = 32\nBATCH_SIZE = BASE_BATCH_SIZE * strategy.num_replicas_in_sync\n\nprint(\"Global batch size:\", BATCH_SIZE)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:28:14.336744Z","iopub.execute_input":"2025-12-03T09:28:14.337263Z","iopub.status.idle":"2025-12-03T09:28:14.572017Z","shell.execute_reply.started":"2025-12-03T09:28:14.337236Z","shell.execute_reply":"2025-12-03T09:28:14.571425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"AUTO = tf.data.AUTOTUNE\n\ndef decode_image(image_data):\n    \"\"\"Decode JPEG -> float32 [0,1] dan reshape eksplisit.\"\"\"\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0      # normalisasi (krusial untuk MLP)\n    image = tf.reshape(image, [*IMAGE_SIZE, 3])\n    return image\n\ndef read_labeled_tfrecord(example_proto):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example_proto, LABELED_TFREC_FORMAT)\n    image = decode_image(example[\"image\"])\n    label = tf.cast(example[\"class\"], tf.int32)\n    return image, label\n\ndef read_unlabeled_tfrecord(example_proto):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example_proto, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example[\"image\"])\n    image_id = example[\"id\"]\n    return image, image_id\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    \"\"\"Load TFRecord -> tf.data.Dataset.\"\"\"\n    options = tf.data.Options()\n    if not ordered:\n        options.experimental_deterministic = False\n\n    dataset = tf.data.TFRecordDataset(\n        filenames,\n        num_parallel_reads=AUTO\n    )\n    dataset = dataset.with_options(options)\n\n    parse_fn = read_labeled_tfrecord if labeled else read_unlabeled_tfrecord\n    dataset = dataset.map(parse_fn, num_parallel_calls=AUTO)\n    return dataset\n\ndef get_training_dataset(batch_size=BATCH_SIZE):\n    train_fns = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/train/*.tfrec')\n    ds = load_dataset(train_fns, labeled=True, ordered=False)\n    ds = ds.shuffle(2048)\n    # TIDAK pakai .repeat() → 1 epoch = 1x seluruh data\n    ds = ds.batch(batch_size)\n    ds = ds.prefetch(AUTO)\n    return ds\n\ndef get_validation_dataset(batch_size=BATCH_SIZE):\n    val_fns = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/val/*.tfrec')\n    ds = load_dataset(val_fns, labeled=True, ordered=True)\n    ds = ds.batch(batch_size)\n    ds = ds.cache()\n    ds = ds.prefetch(AUTO)\n    return ds\n\ndef get_test_dataset(batch_size=BATCH_SIZE, ordered=True):\n    test_fns = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/test/*.tfrec')\n    ds = load_dataset(test_fns, labeled=False, ordered=ordered)\n    ds = ds.batch(batch_size)\n    ds = ds.prefetch(AUTO)\n    return ds\n\n# cek satu batch\ntrain_ds_tmp = get_training_dataset()\nfor imgs, labels in train_ds_tmp.take(1):\n    print(\"Train batch shape:\", imgs.shape, labels.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:28:14.57387Z","iopub.execute_input":"2025-12-03T09:28:14.57412Z","iopub.status.idle":"2025-12-03T09:28:15.930407Z","shell.execute_reply.started":"2025-12-03T09:28:14.574104Z","shell.execute_reply":"2025-12-03T09:28:15.929787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_mlp_model(dense_units=512, dropout_rate=0.3, learning_rate=1e-4, train_base=False):\n    \"\"\"\n    CNN feature extractor (VGG16) + MLP classifier head.\n    MLP head = GlobalAveragePooling -> Dense(hidden) -> Dense(output).\n    \"\"\"\n    with strategy.scope():\n        # Pretrained VGG16 sebagai extractor fitur\n        base_model = tf.keras.applications.VGG16(\n            weights='imagenet',\n            include_top=False,\n            input_shape=(*IMAGE_SIZE, 3)\n        )\n        base_model.trainable = train_base  # False = hanya head yang dilatih (transfer learning)\n\n        inputs = tf.keras.Input(shape=(*IMAGE_SIZE, 3))\n\n        # Sedikit augmentasi (optional tapi membantu)\n        x = tf.keras.layers.RandomFlip(\"horizontal\")(inputs)\n        x = tf.keras.layers.RandomRotation(0.1)(x)\n        x = tf.keras.layers.RandomZoom(0.1)(x)\n\n        # Extract fitur visual\n        x = base_model(x, training=False)\n        x = tf.keras.layers.GlobalAveragePooling2D()(x)\n\n        # --- MLP head (ini yang kamu jelaskan di rubrik) ---\n        x = tf.keras.layers.Dense(dense_units, activation=\"relu\")(x)   # hidden layer\n        x = tf.keras.layers.Dropout(dropout_rate)(x)\n        outputs = tf.keras.layers.Dense(NUM_CLASSES, activation=\"softmax\")(x)  # output layer\n\n        model = tf.keras.Model(inputs, outputs)\n\n        optimizer = tf.keras.optimizers.Adam(learning_rate=learning_rate)\n\n        model.compile(\n            optimizer=optimizer,\n            loss=\"sparse_categorical_crossentropy\",\n            metrics=[\"sparse_categorical_accuracy\"],\n        )\n\n    return model\n\n\n# Beberapa konfigurasi hyperparameter untuk tuning\nHP_CONFIGS = [\n    # hanya head yang dilatih, base_model dibekukan\n    {\"name\": \"cfg_head_256_lr1e-3\", \"dense_units\": 256, \"dropout\": 0.3, \"lr\": 1e-3, \"train_base\": False},\n    {\"name\": \"cfg_head_512_lr5e-4\", \"dense_units\": 512, \"dropout\": 0.4, \"lr\": 5e-4, \"train_base\": False},\n    # opsi fine-tune: unfreeze base_model sebagian (lebih mahal, tapi bisa coba)\n    {\"name\": \"cfg_head_512_ft_lr1e-4\", \"dense_units\": 512, \"dropout\": 0.3, \"lr\": 1e-4, \"train_base\": True},\n]\n\nHP_CONFIGS\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:28:15.93115Z","iopub.execute_input":"2025-12-03T09:28:15.931404Z","iopub.status.idle":"2025-12-03T09:28:15.942012Z","shell.execute_reply.started":"2025-12-03T09:28:15.931376Z","shell.execute_reply":"2025-12-03T09:28:15.941263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tune_results = []\n\nfor cfg in HP_CONFIGS:\n    print(\"\\n=== Training\", cfg[\"name\"], \"===\")\n    tf.keras.backend.clear_session()\n\n    model_tune = build_mlp_model(\n        dense_units=cfg[\"dense_units\"],\n        dropout_rate=cfg[\"dropout\"],\n        learning_rate=cfg[\"lr\"],\n        train_base=cfg[\"train_base\"]\n    )\n\n\n    train_ds_tune = get_training_dataset()\n    val_ds_tune   = get_validation_dataset()\n\n    history_tune = model_tune.fit(\n        train_ds_tune,\n        epochs=EPOCHS_TUNE,\n        validation_data=val_ds_tune,\n        verbose=1\n    )\n\n    val_acc_last = history_tune.history['val_sparse_categorical_accuracy'][-1]\n    tune_results.append({\n        \"name\": cfg[\"name\"],\n        \"config\": cfg,\n        \"val_acc\": float(val_acc_last)\n    })\n    print(\"Config:\", cfg, \"→ val_acc:\", val_acc_last)\n\ntune_results_df = pd.DataFrame(tune_results)\ntune_results_df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:28:15.94281Z","iopub.execute_input":"2025-12-03T09:28:15.943072Z","iopub.status.idle":"2025-12-03T09:37:13.606675Z","shell.execute_reply.started":"2025-12-03T09:28:15.943049Z","shell.execute_reply":"2025-12-03T09:37:13.60595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Pilih konfigurasi dengan validation accuracy tertinggi\nbest_cfg = max(tune_results, key=lambda x: x[\"val_acc\"])[\"config\"]\nprint(\"Best config:\", best_cfg)\n\ntf.keras.backend.clear_session()\nmodel = build_mlp_model(\n    dense_units=best_cfg[\"dense_units\"],\n    dropout_rate=best_cfg[\"dropout\"],\n    learning_rate=best_cfg[\"lr\"],\n    train_base=best_cfg[\"train_base\"]\n)\n\ntrain_ds = get_training_dataset()\nval_ds   = get_validation_dataset()\n\nhistory = model.fit(\n    train_ds,\n    epochs=EPOCHS_FINAL,\n    validation_data=val_ds,\n    verbose=1\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:37:13.608221Z","iopub.execute_input":"2025-12-03T09:37:13.608437Z","iopub.status.idle":"2025-12-03T09:51:44.65384Z","shell.execute_reply.started":"2025-12-03T09:37:13.60842Z","shell.execute_reply":"2025-12-03T09:51:44.653022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Loss vs Epoch\nplt.figure(figsize=(8,5))\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Val Loss')\nplt.xlabel('Epoch'); plt.ylabel('Loss')\nplt.title('Loss vs Epoch')\nplt.legend(); plt.grid(True)\nplt.show()\n\n# Accuracy vs Epoch\nplt.figure(figsize=(8,5))\nplt.plot(history.history['sparse_categorical_accuracy'], label='Train Accuracy')\nplt.plot(history.history['val_sparse_categorical_accuracy'], label='Val Accuracy')\nplt.xlabel('Epoch'); plt.ylabel('Accuracy')\nplt.title('Accuracy vs Epoch')\nplt.legend(); plt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:51:44.654996Z","iopub.execute_input":"2025-12-03T09:51:44.655252Z","iopub.status.idle":"2025-12-03T09:51:44.960488Z","shell.execute_reply.started":"2025-12-03T09:51:44.655232Z","shell.execute_reply":"2025-12-03T09:51:44.959963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_true = []\ny_pred = []\ny_prob = []\n\nfor images, labels in val_ds:\n    probs = model.predict(images, verbose=0)\n    preds = np.argmax(probs, axis=-1)\n\n    y_true.extend(labels.numpy())\n    y_pred.extend(preds)\n    y_prob.extend(probs)\n\ny_true = np.array(y_true)\ny_pred = np.array(y_pred)\ny_prob = np.array(y_prob)\n\nacc  = accuracy_score(y_true, y_pred)\nprec = precision_score(y_true, y_pred, average='macro', zero_division=0)\nrec  = recall_score(y_true, y_pred, average='macro', zero_division=0)\nf1   = f1_score(y_true, y_pred, average='macro', zero_division=0)\n\nprint(\"Validation Accuracy :\", acc)\nprint(\"Validation Precision:\", prec)\nprint(\"Validation Recall   :\", rec)\nprint(\"Validation F1-score :\", f1)\n\n# Macro AUC (multi-class one-vs-rest)\ny_true_bin = label_binarize(y_true, classes=np.arange(NUM_CLASSES))\ntry:\n    auc_macro = roc_auc_score(y_true_bin, y_prob, average='macro', multi_class='ovr')\n    print(\"Validation Macro AUC:\", auc_macro)\nexcept ValueError as e:\n    print(\"AUC could not be computed:\", e)\n\nprint(\"\\nClassification report (ringkas):\")\nprint(classification_report(y_true, y_pred, digits=3))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:51:44.961163Z","iopub.execute_input":"2025-12-03T09:51:44.961412Z","iopub.status.idle":"2025-12-03T09:52:07.906915Z","shell.execute_reply.started":"2025-12-03T09:51:44.961388Z","shell.execute_reply":"2025-12-03T09:52:07.906192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_ds = get_test_dataset(batch_size=BATCH_SIZE, ordered=True)\n\nprint(\"Computing predictions on test set...\")\ntest_images_ds = test_ds.map(lambda image, image_id: image)\nprobs_test = model.predict(test_images_ds)\npreds_test = np.argmax(probs_test, axis=-1)\nprint(\"Predictions shape:\", preds_test.shape)\n\n# Ambil id\ntest_ids_ds = test_ds.map(lambda image, image_id: image_id).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(NUM_TEST_IMAGES))).numpy().astype('U')\nprint(\"IDs shape:\", test_ids.shape)\n\n# Pastikan sesuai dengan sample_submission\nsample_sub = pd.read_csv('/kaggle/input/tpu-getting-started/sample_submission.csv')\nsubmission = pd.DataFrame({'id': test_ids, 'label': preds_test})\nsubmission = sample_sub[['id']].merge(submission, on='id', how='left')\n\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)\nsubmission.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:52:07.907797Z","iopub.execute_input":"2025-12-03T09:52:07.908269Z","iopub.status.idle":"2025-12-03T09:52:24.025557Z","shell.execute_reply.started":"2025-12-03T09:52:07.90825Z","shell.execute_reply":"2025-12-03T09:52:24.024986Z"}},"outputs":[],"execution_count":null}]}