{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Proyek Klasterisasi Penguin: K-Means & Evaluasi Mendalam","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"#### 12S23043 - Grace Tiodora","metadata":{}},{"cell_type":"markdown","source":"## 1. Pendahuluan, Setup, dan Pemuatan Data","metadata":{}},{"cell_type":"code","source":"\n\nimport os\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\n\nfrom sklearn.metrics import (\n    classification_report,\n    confusion_matrix,\n    f1_score,\n    precision_score,\n    recall_score,\n    roc_auc_score\n)\nfrom sklearn.preprocessing import label_binarize\n\nprint(\"TensorFlow version:\", tf.__version__)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### PATH DATASET TFRECORD","metadata":{}},{"cell_type":"code","source":"\n\nDATA_DIR = Path(\"/kaggle/input/tpu-getting-started\")\n\nIMAGE_SIZE = [192, 192]     # resolusi yang tersedia di dataset\nBATCH_SIZE = 32\nAUTO = tf.data.AUTOTUNE\n\nTFREC_DIR = DATA_DIR / f\"tfrecords-jpeg-{IMAGE_SIZE[0]}x{IMAGE_SIZE[1]}\"\n\nprint(os.listdir(TFREC_DIR))\n\nTRAINING_FILENAMES   = tf.io.gfile.glob(str(TFREC_DIR / \"train/*.tfrec\"))\nVALIDATION_FILENAMES = tf.io.gfile.glob(str(TFREC_DIR / \"val/*.tfrec\"))\nTEST_FILENAMES       = tf.io.gfile.glob(str(TFREC_DIR / \"test/*.tfrec\"))\n\nprint(\"Train files:\", len(TRAINING_FILENAMES))\nprint(\"Val files  :\", len(VALIDATION_FILENAMES))\nprint(\"Test files :\", len(TEST_FILENAMES))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:28:03.168097Z","iopub.execute_input":"2025-12-03T11:28:03.168637Z","iopub.status.idle":"2025-12-03T11:28:03.305886Z","shell.execute_reply.started":"2025-12-03T11:28:03.168615Z","shell.execute_reply":"2025-12-03T11:28:03.304888Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## FUNGSI BACA TFRECORD","metadata":{}},{"cell_type":"code","source":"\n\nNUM_CLASSES = 104  # jumlah kelas bunga\n\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.image.resize(image, IMAGE_SIZE)\n    image = tf.cast(image, tf.float32) / 255.0\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFR_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64)\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFR_FORMAT)\n    return decode_image(example[\"image\"]), tf.cast(example[\"class\"], tf.int32)\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFR_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string)\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFR_FORMAT)\n    return decode_image(example[\"image\"]), example[\"id\"]\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    options = tf.data.Options()\n    options.experimental_deterministic = ordered\n\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    dataset = dataset.with_options(options)\n    dataset = dataset.map(\n        read_labeled_tfrecord if labeled else read_unlabeled_tfrecord,\n        num_parallel_calls=AUTO\n    )\n    return dataset\n\ndef get_dataset(filenames, labeled=True, ordered=False, shuffle=False):\n    ds = load_dataset(filenames, labeled=labeled, ordered=ordered)\n    if shuffle:\n        ds = ds.shuffle(2048)\n    ds = ds.batch(BATCH_SIZE)\n    ds = ds.prefetch(AUTO)\n    return ds\n\ntrain_ds = get_dataset(TRAINING_FILENAMES, labeled=True, shuffle=True)\nval_ds   = get_dataset(VALIDATION_FILENAMES, labeled=True)\ntest_ds  = get_dataset(TEST_FILENAMES, labeled=False)\n\nprint(train_ds)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:28:03.309684Z","iopub.execute_input":"2025-12-03T11:28:03.309952Z","iopub.status.idle":"2025-12-03T11:28:03.585504Z","shell.execute_reply.started":"2025-12-03T11:28:03.309930Z","shell.execute_reply":"2025-12-03T11:28:03.584487Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## PEMBANGUNAN MODEL MLP","metadata":{}},{"cell_type":"code","source":"\n\ntf.keras.backend.clear_session()\n\nH1 = 1024\nH2 = 512\nH3 = 256\n\nD1 = 0.3\nD2 = 0.3\nD3 = 0.2\n\nLEARNING_RATE = 5e-4\nEPOCHS = 12\n\ninputs = layers.Input(shape=(*IMAGE_SIZE, 3))\n\nx = layers.Flatten()(inputs)\n\nx = layers.Dense(H1, activation=\"relu\")(x)\nx = layers.BatchNormalization()(x)\nx = layers.Dropout(D1)(x)\n\nx = layers.Dense(H2, activation=\"relu\")(x)\nx = layers.BatchNormalization()(x)\nx = layers.Dropout(D2)(x)\n\nx = layers.Dense(H3, activation=\"relu\")(x)\nx = layers.BatchNormalization()(x)\nx = layers.Dropout(D3)(x)\n\noutputs = layers.Dense(NUM_CLASSES, activation=\"softmax\")(x)\n\nmodel = models.Model(inputs, outputs)\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=LEARNING_RATE),\n    loss=\"sparse_categorical_crossentropy\",\n    metrics=[\"accuracy\"]\n)\n\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:28:03.587658Z","iopub.execute_input":"2025-12-03T11:28:03.587935Z","iopub.status.idle":"2025-12-03T11:28:04.934456Z","shell.execute_reply.started":"2025-12-03T11:28:03.587913Z","shell.execute_reply":"2025-12-03T11:28:04.933525Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## TRAINING MODEL","metadata":{}},{"cell_type":"code","source":"\n\ncallbacks = [\n    tf.keras.callbacks.EarlyStopping(\n        monitor=\"val_loss\",\n        patience=3,\n        restore_best_weights=True,\n        verbose=1\n    ),\n    tf.keras.callbacks.ReduceLROnPlateau(\n        monitor=\"val_loss\",\n        factor=0.5,\n        patience=2,\n        verbose=1\n    )\n]\n\nhistory = model.fit(\n    train_ds,\n    validation_data=val_ds,\n    epochs=EPOCHS,\n    callbacks=callbacks,\n    verbose=1\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:28:04.935559Z","iopub.execute_input":"2025-12-03T11:28:04.935904Z","execution_failed":"2025-12-03T11:28:14.532Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## GRAFIK KONVERGENSI","metadata":{}},{"cell_type":"code","source":"\n\ndef plot_history(history):\n    hist = history.history\n    epochs_range = range(1, len(hist[\"loss\"]) + 1)\n\n    plt.figure(figsize=(14,5))\n\n    # Loss\n    plt.subplot(1,2,1)\n    plt.plot(epochs_range, hist[\"loss\"], label=\"Train Loss\")\n    plt.plot(epochs_range, hist[\"val_loss\"], label=\"Val Loss\")\n    plt.xlabel(\"Epoch\")\n    plt.ylabel(\"Loss\")\n    plt.title(\"Training vs Validation Loss\")\n    plt.legend()\n    plt.grid(True)\n\n    # Accuracy\n    plt.subplot(1,2,2)\n    plt.plot(epochs_range, hist[\"accuracy\"], label=\"Train Acc\")\n    plt.plot(epochs_range, hist[\"val_accuracy\"], label=\"Val Acc\")\n    plt.xlabel(\"Epoch\")\n    plt.ylabel(\"Accuracy\")\n    plt.title(\"Training vs Validation Accuracy\")\n    plt.legend()\n    plt.grid(True)\n\n    plt.tight_layout()\n    plt.show()\n\nplot_history(history)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T07:03:54.169531Z","iopub.execute_input":"2025-12-03T07:03:54.169966Z","iopub.status.idle":"2025-12-03T07:03:54.711612Z","shell.execute_reply.started":"2025-12-03T07:03:54.169940Z","shell.execute_reply":"2025-12-03T07:03:54.710814Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EVALUASI MODEL ","metadata":{"execution":{"iopub.status.busy":"2025-12-03T07:09:17.597354Z","iopub.execute_input":"2025-12-03T07:09:17.598163Z","iopub.status.idle":"2025-12-03T07:09:17.602076Z","shell.execute_reply.started":"2025-12-03T07:09:17.598138Z","shell.execute_reply":"2025-12-03T07:09:17.601077Z"}}},{"cell_type":"code","source":"\ny_true = []\ny_proba = []\n\nfor images, labels in val_ds:\n    preds = model.predict(images, verbose=0)\n    y_proba.append(preds)\n    y_true.append(labels.numpy())\n\ny_true = np.concatenate(y_true)\ny_proba = np.concatenate(y_proba)\ny_pred = np.argmax(y_proba, axis=1)\n\nval_accuracy = np.mean(y_pred == y_true)\nval_precision = precision_score(y_true, y_pred, average=\"macro\", zero_division=0)\nval_recall    = recall_score(y_true, y_pred, average=\"macro\", zero_division=0)\nval_f1        = f1_score(y_true, y_pred, average=\"macro\", zero_division=0)\n\nprint(\"=== METRIK VALIDATION ===\")\nprint(f\"Accuracy  : {val_accuracy:.4f}\")\nprint(f\"Precision : {val_precision:.4f}\")\nprint(f\"Recall    : {val_recall:.4f}\")\nprint(f\"F1-Score  : {val_f1:.4f}\")\n\n# AUC macro (multi-class)\nclasses = list(range(NUM_CLASSES))\ny_true_bin = label_binarize(y_true, classes=classes)\ntry:\n    val_auc = roc_auc_score(y_true_bin, y_proba, multi_class=\"ovo\", average=\"macro\")\n    print(f\"AUC (macro, OVO): {val_auc:.4f}\")\nexcept ValueError as e:\n    print(\"AUC tidak dapat dihitung:\", e)\n\nprint(\"\\n=== Classification Report (ringkas) ===\")\nprint(classification_report(y_true, y_pred, zero_division=0))\n\ncm = confusion_matrix(y_true, y_pred)\nprint(\"Confusion matrix shape:\", cm.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T07:04:13.511249Z","iopub.execute_input":"2025-12-03T07:04:13.511579Z","iopub.status.idle":"2025-12-03T07:04:32.070569Z","shell.execute_reply.started":"2025-12-03T07:04:13.511556Z","shell.execute_reply":"2025-12-03T07:04:32.069710Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## PREDIKSI TEST & SUBMISSION","metadata":{}},{"cell_type":"code","source":"\n\ntest_ids = []\ntest_pred = []\n\nfor images, ids in test_ds:\n    proba = model.predict(images, verbose=0)\n    preds = np.argmax(proba, axis=1)\n    test_pred.extend(preds)\n    # ids: tf.Tensor string -> bytes -> str\n    ids = [i.decode(\"utf-8\") for i in ids.numpy()]\n    test_ids.extend(ids)\n\nsubmission = pd.DataFrame({\n    \"id\": test_ids,\n    \"label\": test_pred\n})\n\nsubmission.to_csv(\"submission.csv\", index=False)\nsubmission.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T07:04:58.171322Z","iopub.execute_input":"2025-12-03T07:04:58.172154Z","iopub.status.idle":"2025-12-03T07:05:35.365635Z","shell.execute_reply.started":"2025-12-03T07:04:58.172124Z","shell.execute_reply":"2025-12-03T07:05:35.364627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"=== RINGKASAN AKHIR ===\")\nprint(f\"Val Accuracy : {val_accuracy:.4f}\")\nprint(f\"Val F1 Macro : {val_f1:.4f}\")\nprint(f\"Val Precision: {val_precision:.4f}\")\nprint(f\"Val Recall   : {val_recall:.4f}\")\ntry:\n    print(f\"Val AUC      : {val_auc:.4f}\")\nexcept:\n    pass\n\nprint(\"\\nFile 'submission.csv' sudah dibuat dan siap di-submit ke Kaggle.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T07:06:02.886391Z","iopub.execute_input":"2025-12-03T07:06:02.886672Z","iopub.status.idle":"2025-12-03T07:06:02.892475Z","shell.execute_reply.started":"2025-12-03T07:06:02.886653Z","shell.execute_reply":"2025-12-03T07:06:02.891500Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}