{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# CELL 1: GPU-ONLY SETUP (KARENA TPU LAGI BERMASALAH HARI INI)\n\nimport os, warnings, tensorflow as tf\nwarnings.filterwarnings(\"ignore\")\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'\n\nprint(\"TensorFlow:\", tf.__version__)\n\n# Paksa pakai GPU (Kaggle lagi error TPU)\nif tf.config.list_physical_devices('GPU'):\n    strategy = tf.distribute.MirroredStrategy()\n    print(\"TPU TERDETEKSI! Menggunakan MirroredStrategy\")\nelse:\n    strategy = tf.distribute.get_strategy()\n    print(\"Hanya TPU tersedia\")\n\nprint(f\"Replica count: {strategy.num_replicas_in_sync}\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T07:29:22.054679Z","iopub.execute_input":"2025-12-03T07:29:22.055043Z","iopub.status.idle":"2025-12-03T07:29:22.061979Z","shell.execute_reply.started":"2025-12-03T07:29:22.055016Z","shell.execute_reply":"2025-12-03T07:29:22.061013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 2: HYPERPARAMETER GPU-FRIENDLY\n\nIMG_SIZE = [160, 160]                                   # 160×160 cukup untuk MLP di GPU\nBATCH_SIZE = 512 * strategy.num_replicas_in_sync        # GPU kuat, kita geber batch besar\nEPOCHS = 40\nLEARNING_RATE = 0.001\n\nprint(f\"Batch size efektif: {BATCH_SIZE}\")\nprint(f\"Image akan di-flatten jadi: {160*160*3:,} neuron\")\n\n# Dataset path (pakai yang 224x224 tetap, tapi kita resize ke 160x160)\nfrom kaggle_datasets import KaggleDatasets\nGCS_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started')\n\nTRAIN_TFREC = tf.io.gfile.glob(GCS_PATH + '/tfrecords-jpeg-224x224/train/*.tfrec')\nVAL_TFREC   = tf.io.gfile.glob(GCS_PATH + '/tfrecords-jpeg-224x224/val/*.tfrec')\nTEST_TFREC  = tf.io.gfile.glob(GCS_PATH + '/tfrecords-jpeg-224x224/test/*.tfrec')\n\nprint(f\"File ditemukan → Train: {len(TRAIN_TFREC)} | Val: {len(VAL_TFREC)} | Test: {len(TEST_TFREC)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T07:29:33.834283Z","iopub.execute_input":"2025-12-03T07:29:33.834601Z","iopub.status.idle":"2025-12-03T07:29:36.681639Z","shell.execute_reply.started":"2025-12-03T07:29:33.834582Z","shell.execute_reply":"2025-12-03T07:29:36.680573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 3: PARSING & DATASET\ndef decode(img):\n    img = tf.image.decode_jpeg(img, channels=3)\n    img = tf.image.resize(img, IMG_SIZE)\n    img = tf.cast(img, tf.float32) / 255.0\n    return tf.reshape(img, [-1])                     # FLATTEN untuk MLP\n\ndef parse_train(example):\n    feature = {\"image\": tf.io.FixedLenFeature([], tf.string),\n               \"class\": tf.io.FixedLenFeature([], tf.int64)}\n    ex = tf.io.parse_single_example(example, feature)\n    return decode(ex['image']), ex['class']\n\ndef parse_test(example):\n    feature = {\"image\": tf.io.FixedLenFeature([], tf.string),\n               \"id\":    tf.io.FixedLenFeature([], tf.string)}\n    ex = tf.io.parse_single_example(example, feature)\n    return decode(ex['image']), ex['id']\n\n# Dataset\ntrain_ds = tf.data.TFRecordDataset(TRAIN_TFREC, num_parallel_reads=tf.data.AUTOTUNE)\ntrain_ds = train_ds.map(parse_train, tf.data.AUTOTUNE)\ntrain_ds = train_ds.shuffle(5000).repeat().batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\nval_ds = tf.data.TFRecordDataset(VAL_TFREC, num_parallel_reads=tf.data.AUTOTUNE)\nval_ds = val_ds.map(parse_train, tf.data.AUTOTUNE).batch(BATCH_SIZE).cache().prefetch(tf.data.AUTOTUNE)\n\nsteps_per_epoch = 12753 // BATCH_SIZE\nprint(\"Dataset siap!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T07:29:57.238443Z","iopub.execute_input":"2025-12-03T07:29:57.238721Z","iopub.status.idle":"2025-12-03T07:29:57.425359Z","shell.execute_reply.started":"2025-12-03T07:29:57.238705Z","shell.execute_reply":"2025-12-03T07:29:57.423978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 4: BANGUN MODEL MLP (murni Dense layers)\nwith strategy.scope():\n    model = tf.keras.Sequential([\n        tf.keras.layers.Input(shape=(IMG_SIZE[0]*IMG_SIZE[1]*3,)),\n        \n        tf.keras.layers.Dense(1536, activation='swish'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.35),\n        \n        tf.keras.layers.Dense(768, activation='swish'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.35),\n        \n        tf.keras.layers.Dense(384, activation='swish'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.25),\n        \n        tf.keras.layers.Dense(104, activation='softmax')\n    ])\n    \n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(LEARNING_RATE),\n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T07:30:06.44654Z","iopub.execute_input":"2025-12-03T07:30:06.446853Z","iopub.status.idle":"2025-12-03T07:30:07.340586Z","shell.execute_reply.started":"2025-12-03T07:30:06.446809Z","shell.execute_reply":"2025-12-03T07:30:07.33959Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 5: CALLBACK & TRAINING\ncallbacks = [\n    tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True),\n    tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.3, patience=4, min_lr=1e-7)\n]\n\nprint(\"Mulai training...\")\nhistory = model.fit(\n    train_ds,\n    steps_per_epoch=steps_per_epoch,\n    epochs=EPOCHS,\n    validation_data=val_ds,\n    callbacks=callbacks,\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T07:30:16.952476Z","iopub.execute_input":"2025-12-03T07:30:16.953253Z","iopub.status.idle":"2025-12-03T07:51:39.854699Z","shell.execute_reply.started":"2025-12-03T07:30:16.953234Z","shell.execute_reply":"2025-12-03T07:51:39.85401Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 6: PLOT TRAINING HISTORY\nimport matplotlib.pyplot as plt\nplt.figure(figsize=(12,4))\n\nplt.subplot(1,2,1)\nplt.plot(history.history['sparse_categorical_accuracy'], label='Train Acc')\nplt.plot(history.history['val_sparse_categorical_accuracy'], label='Val Acc')\nplt.legend(); plt.title('Accuracy')\n\nplt.subplot(1,2,2)\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Val Loss')\nplt.legend(); plt.title('Loss')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T07:52:49.947594Z","iopub.execute_input":"2025-12-03T07:52:49.94792Z","iopub.status.idle":"2025-12-03T07:52:50.300822Z","shell.execute_reply.started":"2025-12-03T07:52:49.947903Z","shell.execute_reply":"2025-12-03T07:52:50.299916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 7: EVALUASI VALIDATION + CONFUSION MATRIX (FIXED & LENGKAP)\n\nimport numpy as np                       # <--- INI YANG KURANG!!!\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nprint(\"Sedang memprediksi seluruh validation set... (sabar ya, ±30 detik)\")\n\n# Prediksi semua data validasi\nval_pred_probs = model.predict(val_ds, verbose=0)\nval_pred = np.argmax(val_pred_probs, axis=1)\n\n# Ambil label sebenarnya\nval_true = np.concatenate([y.numpy() for x, y in val_ds])\n\nprint(\"Selesai prediksi!\\n\")\nprint(classification_report(val_true, val_pred, digits=4))\n\n# Confusion Matrix\ncm = confusion_matrix(val_true, val_pred)\n\nplt.figure(figsize=(12, 10))\nsns.heatmap(cm, \n            annot=False, \n            fmt='d', \n            cmap='Blues', \n            cbar=True,\n            linewidths=0.5,\n            linecolor='gray')\nplt.title('Confusion Matrix – Validation Set', fontsize=16, pad=20)\nplt.xlabel('Predicted Label', fontsize=12)\nplt.ylabel('True Label', fontsize=12)\nplt.show()\n\nprint(f\"Validation Accuracy: {np.mean(val_true == val_pred):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T07:53:43.032584Z","iopub.execute_input":"2025-12-03T07:53:43.032907Z","iopub.status.idle":"2025-12-03T07:53:54.796007Z","shell.execute_reply.started":"2025-12-03T07:53:43.032884Z","shell.execute_reply":"2025-12-03T07:53:54.795008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 8: PREDIKSI TEST & BUAT SUBMISSION (FIXED & SUPER AMAN)\n\nimport numpy as np\nimport pandas as pd                     # <--- INI YANG KURANG!!!\nimport tensorflow as tf\n\nprint(\"Memulai prediksi test set... (sekitar 1–2 menit)\")\n\n# Dataset test\ntest_ds = tf.data.TFRecordDataset(TEST_TFREC, num_parallel_reads=tf.data.AUTOTUNE)\ntest_ds = test_ds.map(parse_test, num_parallel_calls=tf.data.AUTOTUNE)\ntest_ds = test_ds.batch(BATCH_SIZE)\n\nids = []\npreds = []\n\nprint(\"Processing batches...\")\nfor batch_imgs, batch_ids in test_ds:\n    # Prediksi\n    batch_pred = model.predict(batch_imgs, verbose=0)\n    batch_labels = np.argmax(batch_pred, axis=1)\n    preds.extend(batch_labels.tolist())\n    \n    # Ambil ID (aman dari bytes → string)\n    for img_id in batch_ids.numpy():\n        if isinstance(img_id, bytes):\n            ids.append(img_id.decode('utf-8'))\n        else:\n            ids.append(str(img_id))\n\n# Buat submission\nsubmission = pd.DataFrame({\n    'id': ids,\n    'label': preds\n})\n\n# Simpan\nsubmission.to_csv('submission.csv', index=False)\nprint(\"SUBMISSION BERHASIL DISIMPAN!\")\nprint(f\"Total baris: {len(submission)}\")\nprint(\"\\nPreview:\")\ndisplay(submission.head(10))   # pakai display() biar lebih cantik di Kaggle","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T07:56:52.828237Z","iopub.execute_input":"2025-12-03T07:56:52.828532Z","iopub.status.idle":"2025-12-03T07:57:23.504205Z","shell.execute_reply.started":"2025-12-03T07:56:52.828516Z","shell.execute_reply":"2025-12-03T07:57:23.503285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 9: FINAL CHECK (anti kosong)\nimport os\npath = 'submission.csv'\nif not os.path.exists(path):\n    raise FileNotFoundError(\"submission.csv TIDAK ADA!\")\nif os.path.getsize(path) < 1000:\n    raise ValueError(\"submission.csv terlalu kecil/kosong!\")\n\nprint(f\"SELESAI! submission.csv siap → {os.path.getsize(path):,} bytes\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T07:57:28.499588Z","iopub.execute_input":"2025-12-03T07:57:28.499976Z","iopub.status.idle":"2025-12-03T07:57:28.508005Z","shell.execute_reply.started":"2025-12-03T07:57:28.49995Z","shell.execute_reply":"2025-12-03T07:57:28.506241Z"}},"outputs":[],"execution_count":null}]}