{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31194,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# FINAL: PETALS TO THE METAL - MLP IMPLEMENTATION (PERBAIKAN LENGKAP)\nimport os\nimport warnings\nimport tensorflow as tf\nfrom kaggle_datasets import KaggleDatasets\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport pandas as pd\nimport seaborn as sns\nimport time\n\n# ---------------------------\n# SUPPRESS & ENV\n# ---------------------------\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'\nwarnings.filterwarnings(\"ignore\")\nprint(\"TensorFlow:\", tf.__version__)\n\n# ---------------------------\n# HARDWARE DETECTION\n# ---------------------------\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\n    print(\">> USING TPU\")\nexcept Exception:\n    try:\n        strategy = tf.distribute.MirroredStrategy()\n        print(\">> USING MirroredStrategy (GPU if available)\")\n    except Exception:\n        strategy = tf.distribute.get_strategy()\n        print(\">> USING CPU\")\n\n# ---------------------------\n# HYPERPARAMS & PATHS\n# ---------------------------\nIMAGE_SIZE = [64, 64]\nEPOCHS = 25\nBATCH_SIZE = 128 * max(1, strategy.num_replicas_in_sync)\nLEARNING_RATE = 0.001\nprint(f\"Batch size total: {BATCH_SIZE}\")\n\n# GCS (ubah nama dataset jika perlu)\ntry:\n    GCS_DS_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started')\nexcept Exception:\n    GCS_DS_PATH = None\n    print(\"WARNING: GCS dataset not available. Check Kaggle dataset name or Internet.\")\n\nif GCS_DS_PATH:\n    FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/train/*.tfrec')\n    VAL_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/val/*.tfrec')\n    TEST_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/test/*.tfrec')\nelse:\n    FILENAMES = VAL_FILENAMES = TEST_FILENAMES = []\n\nNUM_TRAINING_IMAGES = 12753\nNUM_VALIDATION_IMAGES = 3712\nSTEPS_PER_EPOCH = max(1, NUM_TRAINING_IMAGES // BATCH_SIZE)\n\n# ---------------------------\n# DATA PIPELINE HELPERS\n# ---------------------------\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.image.resize(image, IMAGE_SIZE)\n    image = tf.cast(image, tf.float32) / 255.0\n    image = tf.reshape(image, [-1])  # flatten for MLP\n    return image\n\ndef read_labeled_tfrecord(example):\n    fmt = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, fmt)\n    image = decode_image(example['image'])\n    label = example['class']\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    fmt = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, fmt)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    if not filenames:\n        empty_img = tf.zeros([IMAGE_SIZE[0]*IMAGE_SIZE[1]*3], dtype=tf.float32)\n        empty_label = tf.constant(0, dtype=tf.int64)\n        return tf.data.Dataset.from_tensors((empty_img, empty_label)).take(0)\n    options = tf.data.Options()\n    if not ordered:\n        options.experimental_deterministic = False\n    ds = tf.data.TFRecordDataset(filenames, num_parallel_reads=tf.data.AUTOTUNE)\n    ds = ds.with_options(options)\n    if labeled:\n        ds = ds.map(read_labeled_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)\n    else:\n        ds = ds.map(read_unlabeled_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)\n    return ds\n\n# ---------------------------\n# BUILD PIPELINES\n# ---------------------------\nprint(\"Preparing datasets...\")\ntrain_ds = load_dataset(FILENAMES, labeled=True).repeat().shuffle(2048).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\nval_ds = load_dataset(VAL_FILENAMES, labeled=True, ordered=True).batch(BATCH_SIZE).cache().prefetch(tf.data.AUTOTUNE)\nprint(\"Datasets ready.\")\n\n# ---------------------------\n# MODEL (MLP)\n# ---------------------------\nwith strategy.scope():\n    input_dim = IMAGE_SIZE[0] * IMAGE_SIZE[1] * 3\n    model = tf.keras.Sequential([\n        tf.keras.layers.InputLayer(shape=(input_dim,)),\n        tf.keras.layers.Dense(2048, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.3),\n\n        tf.keras.layers.Dense(1024, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.3),\n\n        tf.keras.layers.Dense(512, activation='relu'),\n        tf.keras.layers.Dense(104, activation='softmax')\n    ])\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=LEARNING_RATE),\n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\nprint(\"\\nMODEL SUMMARY\")\nmodel.summary()\n\n# ---------------------------\n# CALLBACKS (NOTE: NO ModelCheckpoint to avoid requiring file)\n# ---------------------------\nearly_stopping = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True, verbose=1)\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=3, min_lr=1e-6, verbose=1)\n\n# ---------------------------\n# TRAIN (optional)\n# ---------------------------\nprint(\"Starting training (if you want to skip training, interrupt after this message).\")\nhistory = model.fit(\n    train_ds,\n    steps_per_epoch=STEPS_PER_EPOCH,\n    epochs=EPOCHS,\n    validation_data=val_ds,\n    callbacks=[early_stopping, reduce_lr],\n    verbose=1\n)\n\n# ---------------------------\n# PLOT RESULTS\n# ---------------------------\ndef plot_results(history):\n    acc = history.history.get('sparse_categorical_accuracy', [])\n    val_acc = history.history.get('val_sparse_categorical_accuracy', [])\n    loss = history.history.get('loss', [])\n    val_loss = history.history.get('val_loss', [])\n    epochs_r = range(1, len(acc)+1)\n    plt.figure(figsize=(14,5))\n    plt.subplot(1,2,1)\n    plt.plot(epochs_r, acc, label='train acc')\n    plt.plot(epochs_r, val_acc, label='val acc')\n    plt.legend()\n    plt.title('Accuracy')\n    plt.subplot(1,2,2)\n    plt.plot(epochs_r, loss, label='train loss')\n    plt.plot(epochs_r, val_loss, label='val loss')\n    plt.legend()\n    plt.title('Loss')\n    plt.show()\n\nplot_results(history)\n\n# ---------------------------\n# METRICS ON VALIDATION\n# ---------------------------\nprint(\"\\nCalculating classification report on validation set...\")\ny_true = []\ny_pred = []\nfor batch_images, batch_labels in val_ds:\n    probs = model.predict(batch_images, verbose=0)\n    preds = np.argmax(probs, axis=1)\n    try:\n        y_true.extend(batch_labels.numpy().astype(int).tolist())\n    except Exception:\n        y_true.extend([int(x) for x in batch_labels])\n    y_pred.extend(preds.tolist())\n\nif y_true:\n    print(classification_report(y_true, y_pred, digits=4))\n    try:\n        cm = confusion_matrix(y_true, y_pred)\n        plt.figure(figsize=(8,6))\n        sns.heatmap(cm, cmap='Blues')\n        plt.title('Confusion Matrix (val)')\n        plt.xlabel('Predicted')\n        plt.ylabel('True')\n        plt.show()\n    except Exception as e:\n        print(\"Could not plot confusion matrix:\", e)\nelse:\n    print(\"No validation samples found for metrics.\")\n\n# ---------------------------\n# PREDICT TEST & SAVE SUBMISSION (BATCH-SAFE)\n# ---------------------------\nprint(\"\\nPredicting test set and preparing submission...\")\nout_path = '/kaggle/working/submission.csv'\n\nif not TEST_FILENAMES:\n    print(\"No TEST tfrecords found; skipping submission creation.\")\nelse:\n    test_ds = tf.data.TFRecordDataset(TEST_FILENAMES, num_parallel_reads=tf.data.AUTOTUNE)\n    test_ds = test_ds.map(read_unlabeled_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)\n    test_ds = test_ds.batch(BATCH_SIZE)\n\n    all_preds = []\n    all_ids = []\n    for batch_images, batch_ids in test_ds:\n        probs = model.predict(batch_images, verbose=0)\n        preds = np.argmax(probs, axis=1)\n        all_preds.extend(preds.tolist())\n\n        # decode batch IDs robustly\n        try:\n            ids_np = batch_ids.numpy()\n        except Exception:\n            ids_np = np.array(batch_ids)\n        for id_item in ids_np:\n            if isinstance(id_item, (bytes, bytearray)):\n                all_ids.append(id_item.decode('utf-8'))\n            else:\n                try:\n                    all_ids.append(id_item.astype(str))\n                except Exception:\n                    all_ids.append(str(id_item))\n\n    all_preds = np.array(all_preds)\n    all_ids = np.array(all_ids, dtype='U')\n\n    print(\"Collected IDs:\", len(all_ids), \"Collected preds:\", len(all_preds))\n    if len(all_ids) != len(all_preds):\n        print(\"ERROR: ID and prediction counts mismatch!\")\n        print(\"IDs sample:\", all_ids[:10])\n        print(\"Preds sample:\", all_preds[:10])\n    else:\n        submission = pd.DataFrame({'id': all_ids, 'label': all_preds})\n        submission.to_csv(out_path, index=False)\n        submission.to_csv('submission_backup.csv', index=False)\n        print(\"Saved submission to\", out_path)\n        print(submission.head(6))\n\n# ---------------------------\n# FINAL SAFETY CELL (run last; prevents saving bad version)\n# ---------------------------\nprint(\"\\nFinal check before saving notebook version:\")\np = '/kaggle/working/submission.csv'\nif not os.path.exists(p) or os.path.getsize(p) == 0:\n    raise RuntimeError(f\"submission.csv is missing or empty ({p}). Run the cell that creates it BEFORE saving notebook version.\")\nelse:\n    print(\"OK: submission.csv exists and is non-empty.\")\n    print(\"Size (bytes):\", os.path.getsize(p))\n    print(\"File list sample:\", os.listdir('/kaggle/working')[:20])\n\n# End of script\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-01T11:28:13.531222Z","iopub.execute_input":"2025-12-01T11:28:13.531411Z","iopub.status.idle":"2025-12-01T11:32:34.152114Z","shell.execute_reply.started":"2025-12-01T11:28:13.531394Z","shell.execute_reply":"2025-12-01T11:32:34.150889Z"}},"outputs":[],"execution_count":null}]}