{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-02T08:10:49.998814Z","iopub.execute_input":"2025-12-02T08:10:49.999492Z","iopub.status.idle":"2025-12-02T08:10:50.031348Z","shell.execute_reply.started":"2025-12-02T08:10:49.999456Z","shell.execute_reply":"2025-12-02T08:10:50.030063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Instal versi protobuf yang stabil untuk TensorFlow di Kaggle\n!pip install -q \"protobuf<=3.20.3\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T08:11:56.125395Z","iopub.execute_input":"2025-12-02T08:11:56.125866Z","iopub.status.idle":"2025-12-02T08:12:00.040542Z","shell.execute_reply.started":"2025-12-02T08:11:56.125828Z","shell.execute_reply":"2025-12-02T08:12:00.039327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n# Pastikan pakai GPU\nprint(\"TensorFlow Version:\", tf.__version__)\n\n# --- KONFIGURASI ---\n# Kita turunkan resolusi ke 64x64 atau 128x128. \n# Kenapa? Karena MLP benci input yang terlalu besar (kebanyakan parameter bikin pusing).\n# Input 64x64 jauh lebih mudah dipelajari MLP daripada 192x192.\nIMAGE_SIZE = [64, 64] \nBATCH_SIZE = 128 # Batch size diperbesar biar training ngebut\n\nLOCAL_PATH = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-192x192'\nTRAINING_FILENAMES = tf.io.gfile.glob(LOCAL_PATH + '/train/*.tfrec')\nVALIDATION_FILENAMES = tf.io.gfile.glob(LOCAL_PATH + '/val/*.tfrec')\nTEST_FILENAMES = tf.io.gfile.glob(LOCAL_PATH + '/test/*.tfrec')\n\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0\n    image = tf.image.resize(image, IMAGE_SIZE) # Resize ke 64x64\n    return image\n\n# --- FUNGSI AUGMENTASI\ndef data_augment(image, label):\n    # Flip kiri-kanan secara acak\n    image = tf.image.random_flip_left_right(image)\n    # Ubah kecerahan/kontras sedikit\n    image = tf.image.random_brightness(image, max_delta=0.2)\n    image = tf.image.random_contrast(image, lower=0.8, upper=1.2)\n    return image, label\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), \n        \"class\": tf.io.FixedLenFeature([], tf.int64),  \n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = example['class']\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False \n\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=tf.data.AUTOTUNE)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, \n                          num_parallel_calls=tf.data.AUTOTUNE)\n    return dataset\n\n# Load Dataset dengan Augmentasi di Training set\ntrain_dataset = load_dataset(TRAINING_FILENAMES, labeled=True)\ntrain_dataset = train_dataset.map(data_augment, num_parallel_calls=tf.data.AUTOTUNE) # Masukkan augmentasi disini\ntrain_dataset = train_dataset.repeat().shuffle(2048).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\nval_dataset = load_dataset(VALIDATION_FILENAMES, labeled=True, ordered=False)\nval_dataset = val_dataset.batch(BATCH_SIZE).cache().prefetch(tf.data.AUTOTUNE)\n\ntest_dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=True)\ntest_dataset = test_dataset.batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\nprint(\"Data dengan Augmentasi Siap!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T08:12:11.537717Z","iopub.execute_input":"2025-12-02T08:12:11.538098Z","iopub.status.idle":"2025-12-02T08:12:11.842823Z","shell.execute_reply.started":"2025-12-02T08:12:11.538050Z","shell.execute_reply":"2025-12-02T08:12:11.841653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- ARSITEKTUR MLP SUPERCHARGED ---\n\nmodel = tf.keras.Sequential([\n    tf.keras.layers.Input(shape=(*IMAGE_SIZE, 3)),\n    tf.keras.layers.Flatten(),\n    \n    # Layer 1: Besar\n    tf.keras.layers.Dense(1024, activation='relu'), # Diperbesar neuronnya\n    tf.keras.layers.BatchNormalization(), # Penyeimbang\n    tf.keras.layers.Dropout(0.3),\n    \n    # Layer 2\n    tf.keras.layers.Dense(512, activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.Dropout(0.3),\n    \n    # Layer 3\n    tf.keras.layers.Dense(256, activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.Dropout(0.3),\n    \n    # Output\n    tf.keras.layers.Dense(104, activation='softmax')\n])\n\n# Menggunakan Learning Rate Schedule\n# Mulai cepat, lalu melambat agar presisi\nlr_schedule = tf.keras.optimizers.schedules.ExponentialDecay(\n    initial_learning_rate=0.001,\n    decay_steps=1000,\n    decay_rate=0.9\n)\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=lr_schedule),\n    loss = 'sparse_categorical_crossentropy',\n    metrics=['sparse_categorical_accuracy']\n)\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T08:12:19.468631Z","iopub.execute_input":"2025-12-02T08:12:19.468960Z","iopub.status.idle":"2025-12-02T08:12:19.667343Z","shell.execute_reply.started":"2025-12-02T08:12:19.468938Z","shell.execute_reply":"2025-12-02T08:12:19.666481Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"STEPS_PER_EPOCH = 12753 // BATCH_SIZE\n\nprint(\"Mulai Training...\")\nhistory = model.fit(\n    train_dataset, \n    steps_per_epoch=STEPS_PER_EPOCH,\n    epochs=25, # Naikkan epoch karena gambar lebih kecil & ada augmentasi\n    validation_data=val_dataset\n)\n\n# --- PREDIKSI & SUBMIT ---\nprint(\"Prediksi...\")\ntest_images_ds = test_dataset.map(lambda image, idnum: image)\nprobabilities = model.predict(test_images_ds)\npredictions = np.argmax(probabilities, axis=-1)\n\ntest_ids_ds = test_dataset.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(8000))).numpy().astype('U')\n\nnp.savetxt('submission.csv', np.rec.fromarrays([test_ids, predictions]), fmt=['%s', '%d'], delimiter=',', header='id,label', comments='')\nprint(\"Selesai! File submission.csv siap.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T08:12:27.382436Z","iopub.execute_input":"2025-12-02T08:12:27.382790Z","iopub.status.idle":"2025-12-02T08:22:32.684427Z","shell.execute_reply.started":"2025-12-02T08:12:27.382765Z","shell.execute_reply":"2025-12-02T08:22:32.683370Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Mengambil data history dari proses training tadi\nacc = history.history['sparse_categorical_accuracy']\nval_acc = history.history['val_sparse_categorical_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs_range = range(len(acc))\n\nplt.figure(figsize=(15, 5))\n\n# Plot Akurasi\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, acc, label='Training Accuracy')\nplt.plot(epochs_range, val_acc, label='Validation Accuracy')\nplt.legend(loc='lower right')\nplt.title('Training and Validation Accuracy')\nplt.grid(True)\n\n# Plot Loss (Error)\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label='Training Loss')\nplt.plot(epochs_range, val_loss, label='Validation Loss')\nplt.legend(loc='upper right')\nplt.title('Training and Validation Loss')\nplt.grid(True)\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T08:22:38.038234Z","iopub.execute_input":"2025-12-02T08:22:38.038569Z","iopub.status.idle":"2025-12-02T08:22:38.587493Z","shell.execute_reply.started":"2025-12-02T08:22:38.038541Z","shell.execute_reply":"2025-12-02T08:22:38.586424Z"}},"outputs":[],"execution_count":null}]}