{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:49:07.157287Z","iopub.execute_input":"2025-12-03T09:49:07.157627Z","iopub.status.idle":"2025-12-03T09:49:07.226452Z","shell.execute_reply.started":"2025-12-03T09:49:07.157607Z","shell.execute_reply":"2025-12-03T09:49:07.225398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 1: Import dasar & daftar file (bawaan Kaggle)\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Library tambahan\nimport os\nimport warnings\nimport tensorflow as tf\nfrom kaggle_datasets import KaggleDatasets\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport pandas as pd\nimport seaborn as sns\nimport time\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:49:07.227061Z","iopub.execute_input":"2025-12-03T09:49:07.227212Z","iopub.status.idle":"2025-12-03T09:49:07.870338Z","shell.execute_reply.started":"2025-12-03T09:49:07.227198Z","shell.execute_reply":"2025-12-03T09:49:07.869181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 2: SUPPRESS & DETEKSI HARDWARE\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'\nwarnings.filterwarnings(\"ignore\")\nprint(\"TensorFlow:\", tf.__version__)\n\n# Deteksi hardware (TPU -> MirroredStrategy -> CPU)\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\n    print(\">> USING TPU\")\nexcept Exception:\n    try:\n        strategy = tf.distribute.MirroredStrategy()\n        print(\">> USING MirroredStrategy (GPU if available)\")\n    except Exception:\n        strategy = tf.distribute.get_strategy()\n        print(\">> USING CPU\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:49:07.871051Z","iopub.execute_input":"2025-12-03T09:49:07.871240Z","iopub.status.idle":"2025-12-03T09:49:07.878636Z","shell.execute_reply.started":"2025-12-03T09:49:07.871222Z","shell.execute_reply":"2025-12-03T09:49:07.877482Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 3: HYPERPARAMS & PATHS\nIMAGE_SIZE = [64, 64]\nEPOCHS = 25\nBATCH_SIZE = 128 * max(1, strategy.num_replicas_in_sync)\nLEARNING_RATE = 0.001\nprint(f\"Batch size total: {BATCH_SIZE}\")\n\n# Ambil path dataset dari KaggleDatasets (jika tersedia)\ntry:\n    GCS_DS_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started')\nexcept Exception:\n    GCS_DS_PATH = None\n    print(\"WARNING: GCS dataset not available. Check Kaggle dataset name or Internet.\")\n\nif GCS_DS_PATH:\n    FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/train/*.tfrec')\n    VAL_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/val/*.tfrec')\n    TEST_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/test/*.tfrec')\nelse:\n    FILENAMES = VAL_FILENAMES = TEST_FILENAMES = []\n\n# Angka dataset (tetap sesuai kompetisi)\nNUM_TRAINING_IMAGES = 12753\nNUM_VALIDATION_IMAGES = 3712\nSTEPS_PER_EPOCH = max(1, NUM_TRAINING_IMAGES // BATCH_SIZE)\nprint(\"Langkah per epoch (estimasi):\", STEPS_PER_EPOCH)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:49:07.879270Z","iopub.execute_input":"2025-12-03T09:49:07.879466Z","iopub.status.idle":"2025-12-03T09:49:07.917658Z","shell.execute_reply.started":"2025-12-03T09:49:07.879449Z","shell.execute_reply":"2025-12-03T09:49:07.916717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 4: DATA PIPELINE HELPERS\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.image.resize(image, IMAGE_SIZE)\n    image = tf.cast(image, tf.float32) / 255.0\n    image = tf.reshape(image, [-1])  # flatten untuk MLP\n    return image\n\ndef read_labeled_tfrecord(example):\n    fmt = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, fmt)\n    image = decode_image(example['image'])\n    label = example['class']\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    fmt = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, fmt)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    if not filenames:\n        empty_img = tf.zeros([IMAGE_SIZE[0]*IMAGE_SIZE[1]*3], dtype=tf.float32)\n        empty_label = tf.constant(0, dtype=tf.int64)\n        return tf.data.Dataset.from_tensors((empty_img, empty_label)).take(0)\n    options = tf.data.Options()\n    if not ordered:\n        options.experimental_deterministic = False\n    ds = tf.data.TFRecordDataset(filenames, num_parallel_reads=tf.data.AUTOTUNE)\n    ds = ds.with_options(options)\n    if labeled:\n        ds = ds.map(read_labeled_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)\n    else:\n        ds = ds.map(read_unlabeled_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)\n    return ds\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:49:07.918103Z","iopub.execute_input":"2025-12-03T09:49:07.918266Z","iopub.status.idle":"2025-12-03T09:49:07.924337Z","shell.execute_reply.started":"2025-12-03T09:49:07.918250Z","shell.execute_reply":"2025-12-03T09:49:07.923460Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 5: BUILD PIPELINES\nprint(\"Preparing datasets...\")\ntrain_ds = load_dataset(FILENAMES, labeled=True).repeat().shuffle(2048).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\nval_ds = load_dataset(VAL_FILENAMES, labeled=True, ordered=True).batch(BATCH_SIZE).cache().prefetch(tf.data.AUTOTUNE)\nprint(\"Datasets ready.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:49:07.924909Z","iopub.execute_input":"2025-12-03T09:49:07.925076Z","iopub.status.idle":"2025-12-03T09:49:08.045285Z","shell.execute_reply.started":"2025-12-03T09:49:07.925061Z","shell.execute_reply":"2025-12-03T09:49:08.044315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 6: MODEL (MLP)\nwith strategy.scope():\n    input_dim = IMAGE_SIZE[0] * IMAGE_SIZE[1] * 3\n    model = tf.keras.Sequential([\n        tf.keras.layers.InputLayer(shape=(input_dim,)),\n        tf.keras.layers.Dense(2048, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.3),\n\n        tf.keras.layers.Dense(1024, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.3),\n\n        tf.keras.layers.Dense(512, activation='relu'),\n        tf.keras.layers.Dense(104, activation='softmax')\n    ])\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=LEARNING_RATE),\n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\nprint(\"\\nMODEL SUMMARY\")\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:49:08.045828Z","iopub.execute_input":"2025-12-03T09:49:08.046015Z","iopub.status.idle":"2025-12-03T09:49:08.130282Z","shell.execute_reply.started":"2025-12-03T09:49:08.045995Z","shell.execute_reply":"2025-12-03T09:49:08.129279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 7: CALLBACKS (tanpa ModelCheckpoint) & TRAINING\nearly_stopping = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True, verbose=1)\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=3, min_lr=1e-6, verbose=1)\n\nprint(\"Starting training (if you want to skip training, interrupt after this message).\")\nhistory = model.fit(\n    train_ds,\n    steps_per_epoch=STEPS_PER_EPOCH,\n    epochs=EPOCHS,\n    validation_data=val_ds,\n    callbacks=[early_stopping, reduce_lr],\n    verbose=1\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:49:08.130877Z","iopub.execute_input":"2025-12-03T09:49:08.131041Z","iopub.status.idle":"2025-12-03T09:54:08.988132Z","shell.execute_reply.started":"2025-12-03T09:49:08.131025Z","shell.execute_reply":"2025-12-03T09:54:08.987010Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 8: PLOT RESULTS\ndef plot_results(history):\n    acc = history.history.get('sparse_categorical_accuracy', [])\n    val_acc = history.history.get('val_sparse_categorical_accuracy', [])\n    loss = history.history.get('loss', [])\n    val_loss = history.history.get('val_loss', [])\n    epochs_r = range(1, len(acc)+1)\n    plt.figure(figsize=(14,5))\n    plt.subplot(1,2,1)\n    plt.plot(epochs_r, acc, label='train acc')\n    plt.plot(epochs_r, val_acc, label='val acc')\n    plt.legend()\n    plt.title('Accuracy')\n    plt.subplot(1,2,2)\n    plt.plot(epochs_r, loss, label='train loss')\n    plt.plot(epochs_r, val_loss, label='val loss')\n    plt.legend()\n    plt.title('Loss')\n    plt.show()\n\nplot_results(history)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:54:08.988661Z","iopub.execute_input":"2025-12-03T09:54:08.988838Z","iopub.status.idle":"2025-12-03T09:54:09.272285Z","shell.execute_reply.started":"2025-12-03T09:54:08.988823Z","shell.execute_reply":"2025-12-03T09:54:09.271232Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 9: METRICS ON VALIDATION\nprint(\"\\nCalculating classification report on validation set...\")\ny_true = []\ny_pred = []\nfor batch_images, batch_labels in val_ds:\n    probs = model.predict(batch_images, verbose=0)\n    preds = np.argmax(probs, axis=1)\n    try:\n        y_true.extend(batch_labels.numpy().astype(int).tolist())\n    except Exception:\n        y_true.extend([int(x) for x in batch_labels])\n    y_pred.extend(preds.tolist())\n\nif y_true:\n    print(classification_report(y_true, y_pred, digits=4))\n    try:\n        cm = confusion_matrix(y_true, y_pred)\n        plt.figure(figsize=(8,6))\n        sns.heatmap(cm, cmap='Blues')\n        plt.title('Confusion Matrix (val)')\n        plt.xlabel('Predicted')\n        plt.ylabel('True')\n        plt.show()\n    except Exception as e:\n        print(\"Could not plot confusion matrix:\", e)\nelse:\n    print(\"No validation samples found for metrics.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:54:09.272742Z","iopub.execute_input":"2025-12-03T09:54:09.272896Z","iopub.status.idle":"2025-12-03T09:54:15.144579Z","shell.execute_reply.started":"2025-12-03T09:54:09.272881Z","shell.execute_reply":"2025-12-03T09:54:15.143339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 10: PREDICT TEST & SAVE SUBMISSION\nprint(\"\\nPredicting test set and preparing submission...\")\nout_path = '/kaggle/working/submission.csv'\n\nif not TEST_FILENAMES:\n    print(\"No TEST tfrecords found; skipping submission creation.\")\nelse:\n    test_ds = tf.data.TFRecordDataset(TEST_FILENAMES, num_parallel_reads=tf.data.AUTOTUNE)\n    test_ds = test_ds.map(read_unlabeled_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)\n    test_ds = test_ds.batch(BATCH_SIZE)\n\n    all_preds = []\n    all_ids = []\n    for batch_images, batch_ids in test_ds:\n        probs = model.predict(batch_images, verbose=0)\n        preds = np.argmax(probs, axis=1)\n        all_preds.extend(preds.tolist())\n\n        # decode batch IDs robustly\n        try:\n            ids_np = batch_ids.numpy()\n        except Exception:\n            ids_np = np.array(batch_ids)\n        for id_item in ids_np:\n            if isinstance(id_item, (bytes, bytearray)):\n                all_ids.append(id_item.decode('utf-8'))\n            else:\n                try:\n                    all_ids.append(id_item.astype(str))\n                except Exception:\n                    all_ids.append(str(id_item))\n\n    all_preds = np.array(all_preds)\n    all_ids = np.array(all_ids, dtype='U')\n\n    print(\"Collected IDs:\", len(all_ids), \"Collected preds:\", len(all_preds))\n    if len(all_ids) != len(all_preds):\n        print(\"ERROR: ID and prediction counts mismatch!\")\n        print(\"IDs sample:\", all_ids[:10])\n        print(\"Preds sample:\", all_preds[:10])\n    else:\n        submission = pd.DataFrame({'id': all_ids, 'label': all_preds})\n        submission.to_csv(out_path, index=False)\n        submission.to_csv('submission_backup.csv', index=False)\n        print(\"Saved submission to\", out_path)\n        print(submission.head(6))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:54:15.145265Z","iopub.execute_input":"2025-12-03T09:54:15.145450Z","iopub.status.idle":"2025-12-03T09:54:26.109094Z","shell.execute_reply.started":"2025-12-03T09:54:15.145433Z","shell.execute_reply":"2025-12-03T09:54:26.108155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 11: FINAL SAFETY CELL (run last)\nprint(\"\\nFinal check before saving notebook version:\")\np = '/kaggle/working/submission.csv'\nif not os.path.exists(p) or os.path.getsize(p) == 0:\n    raise RuntimeError(f\"submission.csv is missing or empty ({p}). Run the cell that creates it BEFORE saving notebook version.\")\nelse:\n    print(\"OK: submission.csv exists and is non-empty.\")\n    print(\"Size (bytes):\", os.path.getsize(p))\n    print(\"File list sample:\", os.listdir('/kaggle/working')[:20])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:54:26.109544Z","iopub.execute_input":"2025-12-03T09:54:26.109719Z","iopub.status.idle":"2025-12-03T09:54:26.114014Z","shell.execute_reply.started":"2025-12-03T09:54:26.109702Z","shell.execute_reply":"2025-12-03T09:54:26.113239Z"}},"outputs":[],"execution_count":null}]}