{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-02T03:58:12.501792Z","iopub.execute_input":"2025-12-02T03:58:12.502075Z","iopub.status.idle":"2025-12-02T03:58:15.053269Z","shell.execute_reply.started":"2025-12-02T03:58:12.502051Z","shell.execute_reply":"2025-12-02T03:58:15.052111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 1 - Environment / imports (JALANKAN PERTAMA)\nimport os\n# Paksa CPU (matikan GPU/CUDA)\nos.environ[\"CUDA_VISIBLE_DEVICES\"] = \"-1\"\n# Kurangi logging TF\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '2'\n# Kadang protobuf modern bikin masalah GetPrototype, paksa implementasi python protobuf\nos.environ['PROTOCOL_BUFFERS_PYTHON_IMPLEMENTATION'] = 'python'\n\nimport math, glob, sys\nimport numpy as np\nimport tensorflow as tf\nfrom matplotlib import pyplot as plt\n\nprint(\"TensorFlow Version:\", tf.__version__)\nprint(\"Logical devices:\", tf.config.list_logical_devices())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T04:12:46.041624Z","iopub.execute_input":"2025-12-02T04:12:46.041977Z","iopub.status.idle":"2025-12-02T04:12:46.048107Z","shell.execute_reply.started":"2025-12-02T04:12:46.041951Z","shell.execute_reply":"2025-12-02T04:12:46.047282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 2 - Path detection dan checks\n# Ganti jika path berbeda, tapi default Kaggle dataset path:\nGCS_DS_PATH = '/kaggle/input/tpu-getting-started'\n\n# Pastikan file tfrec ada\ntrain_files = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-224x224/train/*.tfrec')\nval_files   = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-224x224/val/*.tfrec')\ntest_files  = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-224x224/test/*.tfrec')\n\nprint(\"Train tfrec count:\", len(train_files))\nprint(\"Val   tfrec count:\", len(val_files))\nprint(\"Test  tfrec count:\", len(test_files))\n\nif len(train_files)==0:\n    raise RuntimeError(\"Tidak menemukan file training .tfrec di path default. Cek dataset path.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T04:12:52.842747Z","iopub.execute_input":"2025-12-02T04:12:52.843062Z","iopub.status.idle":"2025-12-02T04:12:52.865758Z","shell.execute_reply.started":"2025-12-02T04:12:52.843037Z","shell.execute_reply":"2025-12-02T04:12:52.864943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 3 - Fungsi decode & load dataset\nIMAGE_SIZE = [224,224]\n\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.image.resize(image, IMAGE_SIZE)   # aman jika ukuran berbeda\n    image = tf.cast(image, tf.float32) / 255.0\n    return image\n\ndef read_labeled_tfrecord(example):\n    FEATURES = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    ex = tf.io.parse_single_example(example, FEATURES)\n    return decode_image(ex['image']), tf.cast(ex['class'], tf.int32)\n\ndef read_test_tfrecord(example):\n    FEATURES = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    ex = tf.io.parse_single_example(example, FEATURES)\n    return decode_image(ex['image']), ex['id']\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    # options for performance\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False\n    ds = tf.data.TFRecordDataset(filenames, num_parallel_reads=tf.data.AUTOTUNE)\n    ds = ds.with_options(ignore_order)\n    if labeled:\n        ds = ds.map(read_labeled_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)\n    else:\n        ds = ds.map(read_test_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)\n    return ds\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T04:13:02.916817Z","iopub.execute_input":"2025-12-02T04:13:02.917131Z","iopub.status.idle":"2025-12-02T04:13:02.925753Z","shell.execute_reply.started":"2025-12-02T04:13:02.917107Z","shell.execute_reply":"2025-12-02T04:13:02.924849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 4 - Setup pipeline (ubah BATCH_SIZE jika perlu)\n# Karena CPU, gunakan batch lebih kecil\nBATCH_SIZE = 32\nEPOCHS = 3   # coba 3 dulu; naikan kalau ingin\nSTEPS_PER_EPOCH = 0\n\ntrain_ds = load_dataset(train_files, labeled=True)\ntrain_ds = train_ds.repeat().shuffle(2048).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\nval_ds = load_dataset(val_files, labeled=True, ordered=True)\nval_ds = val_ds.batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\ntest_ds = load_dataset(test_files, labeled=False)\ntest_ds = test_ds.batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\n# optional: approximate steps per epoch (gunakan jumlah gambar dataset jika tahu)\n# Jika kamu ingin memaksa: NUM_TRAINING_IMAGES = 12753 (seperti contoh temanmu)\nNUM_TRAINING_IMAGES = 12753\nNUM_VALIDATION_IMAGES = 3712\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\n\nprint(\"Batch size:\", BATCH_SIZE)\nprint(\"Steps per epoch (approx):\", STEPS_PER_EPOCH)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T04:13:10.981899Z","iopub.execute_input":"2025-12-02T04:13:10.982222Z","iopub.status.idle":"2025-12-02T04:13:11.219692Z","shell.execute_reply.started":"2025-12-02T04:13:10.982199Z","shell.execute_reply":"2025-12-02T04:13:11.21892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport keras\nfrom keras import layers, models\n\nIMAGE_SIZE = (224, 224)\nNUM_CLASSES = 104\n\ndef build_mlp(input_shape=(*IMAGE_SIZE, 3), num_classes=NUM_CLASSES):\n    model = keras.Sequential([\n        layers.Input(shape=input_shape),\n        layers.Flatten(),\n\n        layers.Dense(512, activation='relu'),\n        layers.Dropout(0.3),\n\n        layers.Dense(256, activation='relu'),\n        layers.Dropout(0.3),\n\n        layers.Dense(num_classes, activation='softmax')\n    ])\n\n    return model\n\nmodel = build_mlp()\nmodel.compile(\n    optimizer='adam',\n    loss='sparse_categorical_crossentropy',\n    metrics=['accuracy']\n)\n\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T04:17:40.5015Z","iopub.execute_input":"2025-12-02T04:17:40.501898Z","iopub.status.idle":"2025-12-02T04:18:03.132973Z","shell.execute_reply.started":"2025-12-02T04:17:40.501857Z","shell.execute_reply":"2025-12-02T04:18:03.132055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ========== KODE 6: LOAD DATASET TANPA TPU ==========\nimport tensorflow as tf\nimport numpy as np\nimport os\n\nprint(\"TensorFlow Version:\", tf.__version__)\n\n# Path dataset di Kaggle\nDATASET_PATH = \"/kaggle/input/tpu-getting-started\"\n\nIMAGE_SIZE = [224, 224]\nBATCH_SIZE = 32\nEPOCHS = 20\n\n# ----------- Fungsi Decode -----------\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0\n    image = tf.reshape(image, [*IMAGE_SIZE, 3])\n    return image\n\n# ----------- Parse Train/Val -----------\ndef read_labeled_tfrecord(example):\n    lab = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, lab)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label\n\n# ----------- Parse Test -----------\ndef read_test_tfrecord(example):\n    lab = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, lab)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\n# ----------- Dataset Loader -----------\ndef load_dataset(filenames, labeled=True):\n    ignore_order = tf.data.Options()\n    ignore_order.experimental_deterministic = False\n    \n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=tf.data.AUTOTUNE)\n    dataset = dataset.with_options(ignore_order)\n    \n    if labeled:\n        dataset = dataset.map(read_labeled_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)\n    else:\n        dataset = dataset.map(read_test_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)\n    return dataset\n\n# ----------- Ambil File TFRecord -----------\nFILENAMES_TRAIN = tf.io.gfile.glob(DATASET_PATH + '/tfrecords-jpeg-224x224/train/*.tfrec')\nFILENAMES_VAL   = tf.io.gfile.glob(DATASET_PATH + '/tfrecords-jpeg-224x224/val/*.tfrec')\n\nprint(\"Train files:\", len(FILENAMES_TRAIN))\nprint(\"Val files:\",   len(FILENAMES_VAL))\n\n# ----------- Final Dataset TensorFlow -----------\ntraining_dataset = load_dataset(FILENAMES_TRAIN, labeled=True) \\\n    .shuffle(2048) \\\n    .batch(BATCH_SIZE) \\\n    .prefetch(tf.data.AUTOTUNE)\n\nvalidation_dataset = load_dataset(FILENAMES_VAL, labeled=True) \\\n    .batch(BATCH_SIZE) \\\n    .prefetch(tf.data.AUTOTUNE)\n\nprint(\"Dataset siap dipakai.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T04:21:21.35377Z","iopub.execute_input":"2025-12-02T04:21:21.354101Z","iopub.status.idle":"2025-12-02T04:21:21.552302Z","shell.execute_reply.started":"2025-12-02T04:21:21.354076Z","shell.execute_reply":"2025-12-02T04:21:21.55102Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ========== KODE 7: TRAIN MODEL ==========\nhistory = model.fit(\n    training_dataset,\n    validation_data=validation_dataset,\n    epochs=EPOCHS\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T04:21:41.30867Z","iopub.execute_input":"2025-12-02T04:21:41.308997Z","iopub.status.idle":"2025-12-02T04:22:59.14229Z","shell.execute_reply.started":"2025-12-02T04:21:41.308973Z","shell.execute_reply":"2025-12-02T04:22:59.140826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ========== KODE 8: VISUALISASI ==========\nimport matplotlib.pyplot as plt\n\ndef plot_training(history):\n    acc = history.history.get('accuracy', [])\n    val_acc = history.history.get('val_accuracy', [])\n    loss = history.history.get('loss', [])\n    val_loss = history.history.get('val_loss', [])\n\n    plt.figure(figsize=(12,5))\n\n    plt.subplot(1,2,1)\n    plt.plot(acc)\n    plt.plot(val_acc)\n    plt.title(\"Accuracy\")\n    plt.legend([\"Train\", \"Val\"])\n\n    plt.subplot(1,2,2)\n    plt.plot(loss)\n    plt.plot(val_loss)\n    plt.title(\"Loss\")\n    plt.legend([\"Train\", \"Val\"])\n\n    plt.show()\n\nplot_training(history)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T04:23:09.055302Z","iopub.execute_input":"2025-12-02T04:23:09.055648Z","iopub.status.idle":"2025-12-02T04:23:09.07732Z","shell.execute_reply.started":"2025-12-02T04:23:09.055626Z","shell.execute_reply":"2025-12-02T04:23:09.076005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ========== KODE 9: EVALUASI ==========\nval_loss, val_acc = model.evaluate(validation_dataset)\nprint(\"Validation Accuracy:\", val_acc)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T04:23:16.741505Z","iopub.execute_input":"2025-12-02T04:23:16.741951Z","iopub.status.idle":"2025-12-02T04:23:28.152783Z","shell.execute_reply.started":"2025-12-02T04:23:16.741925Z","shell.execute_reply":"2025-12-02T04:23:28.151747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ========== KODE 10: SUBMISSION ==========\nimport csv\n\n# test dataset\nFILENAMES_TEST = tf.io.gfile.glob(DATASET_PATH + '/tfrecords-jpeg-224x224/test/*.tfrec')\ntest_dataset = load_dataset(FILENAMES_TEST, labeled=False) \\\n                   .batch(BATCH_SIZE) \\\n                   .prefetch(tf.data.AUTOTUNE)\n\ntest_ids = []\ntest_preds = []\n\nfor images, ids in test_dataset:\n    probs = model.predict(images, verbose=0)\n    preds = np.argmax(probs, axis=1)\n    test_ids.extend([x.decode('utf-8') for x in ids.numpy()])\n    test_preds.extend(preds)\n\n# simpan file submission\nwith open(\"submission.csv\", \"w\", newline=\"\") as f:\n    writer = csv.writer(f)\n    writer.writerow([\"id\", \"label\"])\n    for id_, pred in zip(test_ids, test_preds):\n        writer.writerow([id_, pred])\n\nprint(\"submission.csv berhasil dibuat.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-02T04:23:52.311836Z","iopub.execute_input":"2025-12-02T04:23:52.312215Z","iopub.status.idle":"2025-12-02T04:24:32.526629Z","shell.execute_reply.started":"2025-12-02T04:23:52.312189Z","shell.execute_reply":"2025-12-02T04:24:32.525791Z"}},"outputs":[],"execution_count":null}]}