{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31194,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, math, random\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom sklearn.metrics import (\n    accuracy_score, precision_score, recall_score, f1_score,\n    classification_report\n)\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n\nprint(\"TensorFlow version:\", tf.__version__)\n\n# =============== TPU STRATEGY ====================\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print(\"Running on TPU:\", tpu.master())\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\nexcept ValueError:\n    print(\"No TPU found, using CPU/GPU instead.\")\n    strategy = tf.distribute.get_strategy()\n\nREPLICAS = strategy.num_replicas_in_sync\nprint(\"REPLICAS:\", REPLICAS)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T03:56:54.151521Z","iopub.execute_input":"2025-12-03T03:56:54.151813Z","iopub.status.idle":"2025-12-03T03:56:54.156379Z","shell.execute_reply.started":"2025-12-03T03:56:54.151792Z","shell.execute_reply":"2025-12-03T03:56:54.155751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============== PATH DATASET ====================\nGCS_PATH = \"/kaggle/input/tpu-getting-started\"\n\n# lihat dulu isi folder ini kalau mau memastikan\nprint(os.listdir(GCS_PATH))\n\n# subfolder tfrecord di dalamnya, biasanya 'tfrecords-jpeg-224x224'\nTFREC_SUBDIR = \"tfrecords-jpeg-224x224\"  # GANTI kalau namanya beda\n\nTRAIN_TFREC = tf.io.gfile.glob(f\"{GCS_PATH}/{TFREC_SUBDIR}/train/*.tfrec\")\nVAL_TFREC   = tf.io.gfile.glob(f\"{GCS_PATH}/{TFREC_SUBDIR}/val/*.tfrec\")\nTEST_TFREC  = tf.io.gfile.glob(f\"{GCS_PATH}/{TFREC_SUBDIR}/test/*.tfrec\")\n\nprint(\"Train files:\", len(TRAIN_TFREC))\nprint(\"Val files  :\", len(VAL_TFREC))\nprint(\"Test files :\", len(TEST_TFREC))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T03:56:54.157676Z","iopub.execute_input":"2025-12-03T03:56:54.157827Z","iopub.status.idle":"2025-12-03T03:56:54.199043Z","shell.execute_reply.started":"2025-12-03T03:56:54.157813Z","shell.execute_reply":"2025-12-03T03:56:54.198363Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMAGE_SIZE  = 224\nNUM_CLASSES = 104\nAUTO       = tf.data.AUTOTUNE\nBATCH_SIZE = 64 * REPLICAS\n\nprint(\"Batch size:\", BATCH_SIZE)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T03:56:54.199284Z","iopub.execute_input":"2025-12-03T03:56:54.199441Z","iopub.status.idle":"2025-12-03T03:56:54.202289Z","shell.execute_reply.started":"2025-12-03T03:56:54.199426Z","shell.execute_reply":"2025-12-03T03:56:54.201680Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.image.resize(image, [IMAGE_SIZE, IMAGE_SIZE])\n    image = tf.cast(image, tf.float32) / 255.0\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, LABELED_FORMAT)\n    image = decode_image(example[\"image\"])\n    label = tf.cast(example[\"class\"], tf.int32)\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\":    tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_FORMAT)\n    image = decode_image(example[\"image\"])\n    idnum = example[\"id\"]\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    options = tf.data.Options()\n    if not ordered:\n        options.experimental_deterministic = False\n\n    ds = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    ds = ds.with_options(options)\n    ds = ds.map(\n        read_labeled_tfrecord if labeled else read_unlabeled_tfrecord,\n        num_parallel_calls=AUTO\n    )\n    return ds\n\ndef get_training_dataset():\n    ds = load_dataset(TRAIN_TFREC, labeled=True)\n    ds = ds.shuffle(2048)\n    ds = ds.batch(BATCH_SIZE)\n    ds = ds.prefetch(AUTO)\n    return ds\n\ndef get_validation_dataset():\n    ds = load_dataset(VAL_TFREC, labeled=True, ordered=True)\n    ds = ds.batch(BATCH_SIZE)\n    ds = ds.cache()\n    ds = ds.prefetch(AUTO)\n    return ds\n\ndef get_test_dataset():\n    ds = load_dataset(TEST_TFREC, labeled=False, ordered=True)\n    ds = ds.batch(BATCH_SIZE)\n    ds = ds.prefetch(AUTO)\n    return ds\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T03:56:54.202801Z","iopub.execute_input":"2025-12-03T03:56:54.202955Z","iopub.status.idle":"2025-12-03T03:56:54.215254Z","shell.execute_reply.started":"2025-12-03T03:56:54.202941Z","shell.execute_reply":"2025-12-03T03:56:54.214625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ds = get_training_dataset()\nval_ds   = get_validation_dataset()\ntest_ds  = get_test_dataset()\n\nprint(train_ds)\nprint(val_ds)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T03:56:54.215639Z","iopub.execute_input":"2025-12-03T03:56:54.215816Z","iopub.status.idle":"2025-12-03T03:56:54.405569Z","shell.execute_reply.started":"2025-12-03T03:56:54.215802Z","shell.execute_reply":"2025-12-03T03:56:54.404774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_mlp(\n    num_hidden_layers=2,\n    hidden_units=512,\n    dropout_rate=0.4,\n    learning_rate=1e-3\n):\n    inputs = keras.Input(shape=(IMAGE_SIZE, IMAGE_SIZE, 3))\n    x = layers.Flatten()(inputs)\n\n    for _ in range(num_hidden_layers):\n        x = layers.Dense(hidden_units, activation=\"relu\")(x)\n        x = layers.Dropout(dropout_rate)(x)\n\n    outputs = layers.Dense(NUM_CLASSES, activation=\"softmax\")(x)\n    model = keras.Model(inputs, outputs)\n\n    model.compile(\n        optimizer=keras.optimizers.Adam(learning_rate),\n        loss=\"sparse_categorical_crossentropy\",\n        metrics=[\"accuracy\"]\n    )\n    return model\n\nwith strategy.scope():\n    model = build_mlp()\n    model.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T03:56:54.406007Z","iopub.execute_input":"2025-12-03T03:56:54.406160Z","iopub.status.idle":"2025-12-03T03:56:54.520937Z","shell.execute_reply.started":"2025-12-03T03:56:54.406146Z","shell.execute_reply":"2025-12-03T03:56:54.520177Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"EPOCHS = 15\n\ncheckpoint_cb = keras.callbacks.ModelCheckpoint(\n    \"best_mlp.h5\",\n    save_best_only=True,\n    monitor=\"val_accuracy\",\n    mode=\"max\"\n)\n\nearlystop_cb = keras.callbacks.EarlyStopping(\n    monitor=\"val_loss\",\n    patience=3,\n    restore_best_weights=True\n)\n\nhistory = model.fit(\n    train_ds,\n    epochs=EPOCHS,\n    validation_data=val_ds,\n    callbacks=[checkpoint_cb, earlystop_cb]\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T03:56:54.521318Z","iopub.execute_input":"2025-12-03T03:56:54.521484Z","iopub.status.idle":"2025-12-03T04:02:23.637949Z","shell.execute_reply.started":"2025-12-03T03:56:54.521470Z","shell.execute_reply":"2025-12-03T04:02:23.636712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_history(history):\n    hist = history.history\n    epochs = range(1, len(hist[\"loss\"]) + 1)\n\n    plt.figure(figsize=(12,5))\n\n    plt.subplot(1,2,1)\n    plt.plot(epochs, hist[\"loss\"], label=\"train\")\n    plt.plot(epochs, hist[\"val_loss\"], label=\"val\")\n    plt.title(\"Loss vs Epoch\")\n    plt.legend()\n\n    plt.subplot(1,2,2)\n    plt.plot(epochs, hist[\"accuracy\"], label=\"train\")\n    plt.plot(epochs, hist[\"val_accuracy\"], label=\"val\")\n    plt.title(\"Accuracy vs Epoch\")\n    plt.legend()\n\n    plt.show()\n\nplot_history(history)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T04:02:23.638546Z","iopub.execute_input":"2025-12-03T04:02:23.638752Z","iopub.status.idle":"2025-12-03T04:02:24.043644Z","shell.execute_reply.started":"2025-12-03T04:02:23.638726Z","shell.execute_reply":"2025-12-03T04:02:24.042563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_images = []\nval_labels = []\n\nfor imgs, labels in val_ds:\n    val_images.append(imgs.numpy())\n    val_labels.append(labels.numpy())\n\nval_images = np.concatenate(val_images)\nval_labels = np.concatenate(val_labels)\n\nval_proba = model.predict(val_images, batch_size=BATCH_SIZE)\nval_pred  = np.argmax(val_proba, axis=1)\n\nacc  = accuracy_score(val_labels, val_pred)\nprec = precision_score(val_labels, val_pred, average=\"macro\", zero_division=0)\nrec  = recall_score(val_labels, val_pred, average=\"macro\", zero_division=0)\nf1   = f1_score(val_labels, val_pred, average=\"macro\", zero_division=0)\n\nprint(\"Accuracy :\", acc)\nprint(\"Precision:\", prec)\nprint(\"Recall   :\", rec)\nprint(\"F1 Score :\", f1)\n\nprint(\"\\nClassification report:\")\nprint(classification_report(val_labels, val_pred, digits=3))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T04:02:24.044383Z","iopub.execute_input":"2025-12-03T04:02:24.044562Z","iopub.status.idle":"2025-12-03T04:02:27.825678Z","shell.execute_reply.started":"2025-12-03T04:02:24.044548Z","shell.execute_reply":"2025-12-03T04:02:27.824360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_ids = []\ntest_preds = []\n\nfor imgs, ids in test_ds:\n    proba = model.predict(imgs, verbose=0)\n    preds = np.argmax(proba, axis=1)\n    test_preds.extend(preds)\n    test_ids.extend(ids.numpy().astype(str))\n\nsubmission = pd.DataFrame({\n    \"id\": test_ids,\n    \"label\": test_preds\n})\n\nprint(submission.head())\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(\"Saved submission.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T04:02:27.826087Z","iopub.execute_input":"2025-12-03T04:02:27.826267Z","iopub.status.idle":"2025-12-03T04:02:39.363545Z","shell.execute_reply.started":"2025-12-03T04:02:27.826250Z","shell.execute_reply":"2025-12-03T04:02:39.362321Z"}},"outputs":[],"execution_count":null}]}