{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:04:59.68523Z","iopub.execute_input":"2025-12-03T13:04:59.685632Z","iopub.status.idle":"2025-12-03T13:04:59.715533Z","shell.execute_reply.started":"2025-12-03T13:04:59.685609Z","shell.execute_reply":"2025-12-03T13:04:59.714906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Setup\nimport math, re, os, glob\nimport numpy as np\nimport tensorflow as tf\nfrom matplotlib import pyplot as plt\nfrom sklearn.metrics import classification_report\n\nprint(\"TensorFlow Version:\", tf.__version__)\n\n# --- 1. KONFIGURASI GPU ---\nstrategy = tf.distribute.MirroredStrategy()\nprint(f'✅ Running on {strategy.num_replicas_in_sync} GPU(s)!')\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\n\n# --- 2. SETUP DATASET ---\nDS_PATH = '/kaggle/input/tpu-getting-started'\n# WAJIB 224x224 untuk VGG16\nIMAGE_SIZE = [224, 224] \n\n# --- 3. FUNGSI DECODE & AUGMENTASI ---\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0\n    # Pastikan di-resize ke 224x224, bukan reshape\n    image = tf.reshape(image, [*IMAGE_SIZE, 3])\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label\n\ndef data_augment(image, label):\n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_flip_up_down(image)\n    image = tf.image.random_brightness(image, max_delta=0.2)\n    return image, label\n\ndef load_dataset(filenames, labeled=True, ordered=False, augment=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=tf.data.experimental.AUTOTUNE)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord, num_parallel_calls=tf.data.experimental.AUTOTUNE)\n    if augment:\n        dataset = dataset.map(data_augment, num_parallel_calls=tf.data.experimental.AUTOTUNE)\n    return dataset\n\n# Load File\nFILENAMES_TRAIN = tf.io.gfile.glob(DS_PATH + '/tfrecords-jpeg-224x224/train/*.tfrec')\nFILENAMES_VAL = tf.io.gfile.glob(DS_PATH + '/tfrecords-jpeg-224x224/val/*.tfrec')\n\ntraining_dataset = load_dataset(FILENAMES_TRAIN, labeled=True, augment=True).repeat().shuffle(2048).batch(BATCH_SIZE).prefetch(tf.data.experimental.AUTOTUNE)\nvalidation_dataset = load_dataset(FILENAMES_VAL, labeled=True, ordered=True).batch(BATCH_SIZE).prefetch(tf.data.experimental.AUTOTUNE)\n\nprint(\"✅ Langkah 1 Selesai. Data di-reset ke 224x224.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:04:59.71658Z","iopub.execute_input":"2025-12-03T13:04:59.716788Z","iopub.status.idle":"2025-12-03T13:04:59.842675Z","shell.execute_reply.started":"2025-12-03T13:04:59.716771Z","shell.execute_reply":"2025-12-03T13:04:59.841888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Transfer MLP\nwith strategy.scope():\n    # 1. Feature Extractor (Anggap ini Input Layer Canggih)\n    # Kita pakai VGG16 karena strukturnya paling mirip MLP (blok-blok lurus)\n    pretrained_model = tf.keras.applications.VGG16(\n        weights='imagenet', \n        include_top=False ,\n        input_shape=[*IMAGE_SIZE, 3]\n    )\n    pretrained_model.trainable = False # KITA BEKUKAN (Agar rubrik tetap fokus ke MLP kita)\n\n    model = tf.keras.Sequential([\n        # Input dari VGG16\n        pretrained_model,\n        \n        # --- MULAI ARSITEKTUR MLP SESUAI RUBRIK ---\n        # 1. Input Layer (Flattening)\n        tf.keras.layers.Flatten(),\n\n        # 2. Hidden Layer 1 (Jumbo)\n        tf.keras.layers.Dense(4096, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.5),\n\n        # 3. Hidden Layer 2 (Besar)\n        tf.keras.layers.Dense(4096, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.5),\n\n        # 4. Output Layer (104 Kelas)\n        tf.keras.layers.Dense(104, activation='softmax')\n    ])\n\n    model.compile(\n        optimizer='adam',\n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:04:59.843509Z","iopub.execute_input":"2025-12-03T13:04:59.843768Z","iopub.status.idle":"2025-12-03T13:05:02.1736Z","shell.execute_reply.started":"2025-12-03T13:04:59.843745Z","shell.execute_reply":"2025-12-03T13:05:02.17298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Training 20 epouch\nNUM_TRAINING_IMAGES = 12753\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\nEPOCHS = 20\n\n# Callback untuk menyimpan model terbaik & mengurangi LR\ncheckpoint = tf.keras.callbacks.ModelCheckpoint(\n    'best_model.keras', save_best_only=True, monitor='val_sparse_categorical_accuracy', mode='max'\n)\nlr_scheduler = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor='val_loss', factor=0.5, patience=3, min_lr=1e-6, verbose=1\n)\n\nprint(\"🚀 Memulai Training (Target Akurasi > 60%)...\")\n\nhistory = model.fit(\n    training_dataset,\n    steps_per_epoch=STEPS_PER_EPOCH,\n    epochs=EPOCHS,\n    validation_data=validation_dataset,\n    callbacks=[checkpoint, lr_scheduler]\n)\n\n# Plotting Grafik (Wajib Rubrik)\ndef plot_history(history):\n    acc = history.history['sparse_categorical_accuracy']\n    val_acc = history.history['val_sparse_categorical_accuracy']\n    loss = history.history['loss']\n    val_loss = history.history['val_loss']\n    epochs = range(1, len(acc) + 1)\n\n    plt.figure(figsize=(12, 5))\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs, acc, 'r', label='Training Acc')\n    plt.plot(epochs, val_acc, 'b', label='Validation Acc')\n    plt.title('Accuracy')\n    plt.legend()\n    plt.subplot(1, 2, 2)\n    plt.plot(epochs, loss, 'r', label='Training Loss')\n    plt.plot(epochs, val_loss, 'b', label='Validation Loss')\n    plt.title('Loss')\n    plt.legend()\n    plt.show()\n\nplot_history(history)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:05:02.174909Z","iopub.execute_input":"2025-12-03T13:05:02.17512Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Matriks\nprint(\"Sedang mengevaluasi model terbaik...\")\n# Load model terbaik yang tersimpan tadi\n# Pastikan file 'best_model.keras' sudah terbentuk dari Langkah 3\ntry:\n    model.load_weights('best_model.keras')\n    print(\"✅ Model terbaik berhasil dimuat.\")\nexcept:\n    print(\"⚠️ Warning: Menggunakan model terakhir (belum load best weights).\")\n\ny_true = []\ny_pred = []\n\n# Loop untuk mengambil prediksi\nfor images, labels in validation_dataset:\n    y_true.extend(labels.numpy())\n    probs = model.predict(images, verbose=0)\n    y_pred.extend(np.argmax(probs, axis=-1))\n\nprint(classification_report(y_true, y_pred))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Suubmission \nimport csv\n\nprint('Sedang membuat file Submission...')\n\ndef read_test_tfrecord(example):\n    TEST_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, TEST_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\nFILENAMES_TEST = tf.io.gfile.glob(DS_PATH + '/tfrecords-jpeg-224x224/test/*.tfrec')\ntest_dataset = tf.data.TFRecordDataset(FILENAMES_TEST, num_parallel_reads=tf.data.experimental.AUTOTUNE)\ntest_dataset = test_dataset.map(read_test_tfrecord, num_parallel_calls=tf.data.experimental.AUTOTUNE)\ntest_dataset = test_dataset.batch(BATCH_SIZE).prefetch(tf.data.experimental.AUTOTUNE)\n\ntest_ids = []\ntest_preds = []\n\nfor image, idnum in test_dataset:\n    test_ids.extend([x.decode('utf-8') for x in idnum.numpy()])\n    probs = model.predict(image, verbose=0)\n    test_preds.extend(np.argmax(probs, axis=-1))\n\n# Tulis ke CSV\nwith open('submission.csv', 'w', newline='') as f:\n    writer = csv.writer(f)\n    writer.writerow([\"id\", \"label\"])\n    for i in range(len(test_ids)):\n        writer.writerow([test_ids[i], test_preds[i]])\n\nprint(\"✅ SUKSES! File 'submission.csv' siap disubmit.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}