{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31194,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:50:55.465671Z","iopub.execute_input":"2025-12-03T09:50:55.465929Z","iopub.status.idle":"2025-12-03T09:50:55.923948Z","shell.execute_reply.started":"2025-12-03T09:50:55.465912Z","shell.execute_reply":"2025-12-03T09:50:55.923154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =================================================================================\n# FINAL REVISION: DEEP MLP FOR FLOWER CLASSIFICATION (VARIABLE NAME FIX)\n# =================================================================================\n\n# 1. SYSTEM FIX (WAJIB ADA UNTUK MENGHINDARI ERROR MESSAGEFACTORY)\nimport sys\nimport subprocess\ntry:\n    subprocess.check_call([sys.executable, \"-m\", \"pip\", \"install\", \"-q\", \"protobuf==3.20.3\"])\nexcept:\n    pass\n\nimport os\nimport re\nimport warnings\nimport tensorflow as tf\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.metrics import classification_report\n\n# Konfigurasi Log\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'\nwarnings.filterwarnings(\"ignore\")\nprint(f\"TensorFlow Version: {tf.__version__}\")\n\n# ==========================================\n# 2. HARDWARE SETUP\n# ==========================================\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\n    print(\">> HARDWARE: TPU DETECTED\")\nexcept ValueError:\n    strategy = tf.distribute.MirroredStrategy()\n    print(\">> HARDWARE: GPU DETECTED\")\n\n# ==========================================\n# 3. HYPERPARAMETERS\n# ==========================================\nIMAGE_SIZE = [64, 64] \nEPOCHS = 30  \nBATCH_SIZE = 128 * strategy.num_replicas_in_sync \nLEARNING_RATE = 0.0005 \n\n# Akses GCS Path\ntry:\n    GCS_DS_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started')\nexcept:\n    GCS_DS_PATH = \"/kaggle/input/tpu-getting-started\"\n\n# --- PERBAIKAN VARIABEL DI SINI (KONSISTEN MENGGUNAKAN 'TRAINING_FILENAMES') ---\nTRAINING_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/train/*.tfrec')\nVALIDATION_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/val/*.tfrec')\nTEST_FILENAMES = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/test/*.tfrec')\n\nCLASSES = 104\nAUTO = tf.data.AUTOTUNE\n\ndef count_data_items(filenames):\n    n = [int(re.search(r'-(\\d+)\\.', f).group(1)) for f in filenames]\n    return np.sum(n)\n\n# Variabel ini sekarang cocok dengan definisi di atas\nNUM_TRAINING_IMAGES = count_data_items(TRAINING_FILENAMES)\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\n\nprint(f\"Training Images: {NUM_TRAINING_IMAGES}\")\n\n# ==========================================\n# 4. PREPROCESSING & AUGMENTATION\n# ==========================================\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0\n    image = tf.image.resize(image, IMAGE_SIZE)\n    return image\n\ndef read_labeled_tfrecord(example):\n    format = {\"image\": tf.io.FixedLenFeature([], tf.string),\n              \"class\": tf.io.FixedLenFeature([], tf.int64)}\n    example = tf.io.parse_single_example(example, format)\n    return decode_image(example['image']), example['class']\n\ndef read_unlabeled_tfrecord(example):\n    format = {\"image\": tf.io.FixedLenFeature([], tf.string),\n              \"id\": tf.io.FixedLenFeature([], tf.string)}\n    example = tf.io.parse_single_example(example, format)\n    return decode_image(example['image']), example['id']\n\ndef augment(image, label):\n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_contrast(image, 0.8, 1.2)\n    image = tf.image.random_brightness(image, 0.1)\n    return image, label\n\ndef load_dataset(filenames, labeled=True, ordered=False, do_aug=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False\n    \n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, num_parallel_calls=AUTO)\n    \n    if do_aug:\n        dataset = dataset.map(augment, num_parallel_calls=AUTO)\n        \n    return dataset\n\n# Setup Pipelines (Pastikan variabel filenames benar)\ntrain_ds = load_dataset(TRAINING_FILENAMES, labeled=True, do_aug=True).repeat().shuffle(2048).batch(BATCH_SIZE).prefetch(AUTO)\nval_ds = load_dataset(VALIDATION_FILENAMES, labeled=True, ordered=True).batch(BATCH_SIZE).cache().prefetch(AUTO)\ntest_ds = load_dataset(TEST_FILENAMES, labeled=False, ordered=True).batch(BATCH_SIZE).prefetch(AUTO)\n\n# ==========================================\n# 5. MODEL ARCHITECTURE\n# ==========================================\nwith strategy.scope():\n    model = tf.keras.Sequential([\n        tf.keras.layers.InputLayer(shape=(IMAGE_SIZE[0], IMAGE_SIZE[1], 3)),\n        tf.keras.layers.Flatten(),\n        \n        # Layer 1\n        tf.keras.layers.Dense(2048, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.4),\n        \n        # Layer 2\n        tf.keras.layers.Dense(1024, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.3),\n        \n        # Layer 3\n        tf.keras.layers.Dense(512, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.3),\n        \n        # Layer 4\n        tf.keras.layers.Dense(256, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        \n        # Output\n        tf.keras.layers.Dense(CLASSES, activation='softmax')\n    ])\n    \n    model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=LEARNING_RATE),\n                  loss='sparse_categorical_crossentropy',\n                  metrics=['sparse_categorical_accuracy'])\n\nprint(\"\\n>>> MODEL SUMMARY:\")\nmodel.summary()\n\n# ==========================================\n# 6. TRAINING\n# ==========================================\nprint(\"\\n>>> STARTING TRAINING (30 EPOCHS)...\")\nearly_stopping = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=8, restore_best_weights=True)\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=3, min_lr=1e-6)\n\nhistory = model.fit(\n    train_ds,\n    steps_per_epoch=STEPS_PER_EPOCH,\n    epochs=EPOCHS,\n    validation_data=val_ds,\n    callbacks=[early_stopping, reduce_lr],\n    verbose=1\n)\n\n# ==========================================\n# 7. EVALUASI & VISUALISASI\n# ==========================================\ndef plot_history(history):\n    acc = history.history['sparse_categorical_accuracy']\n    val_acc = history.history['val_sparse_categorical_accuracy']\n    loss = history.history['loss']\n    val_loss = history.history['val_loss']\n    epochs = range(1, len(acc) + 1)\n\n    plt.figure(figsize=(15, 5))\n    \n    # Accuracy\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs, acc, label='Training Accuracy')\n    plt.plot(epochs, val_acc, label='Validation Accuracy')\n    plt.title('Training & Validation Accuracy')\n    plt.legend()\n    \n    # Loss\n    plt.subplot(1, 2, 2)\n    plt.plot(epochs, loss, label='Training Loss')\n    plt.plot(epochs, val_loss, label='Validation Loss')\n    plt.title('Training & Validation Loss')\n    plt.legend()\n    plt.show()\n\nprint(\"\\n>>> HASIL EVALUASI:\")\nplot_history(history)\n\n# Classification Report\nprint(\">> Menghitung Classification Report...\")\nval_labels_ds = val_ds.map(lambda image, label: label).unbatch()\ny_true = next(iter(val_labels_ds.batch(count_data_items(VALIDATION_FILENAMES)))).numpy()\n\nval_images_ds = val_ds.map(lambda image, label: image)\nval_probs = model.predict(val_images_ds, verbose=1)\ny_pred = np.argmax(val_probs, axis=-1)\n\nprint(classification_report(y_true, y_pred))\n\n# ==========================================\n# 8. SUBMISSION\n# ==========================================\nprint(\"\\n>>> MEMBUAT SUBMISSION FILE...\")\n# Ambil ID\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(count_data_items(TEST_FILENAMES)))).numpy().astype('U')\n\n# Ambil Prediksi\ntest_images_ds = test_ds.map(lambda image, idnum: image)\ntest_probs = model.predict(test_images_ds, verbose=1)\ntest_preds = np.argmax(test_probs, axis=-1)\n\nif len(test_ids) == len(test_preds):\n    submission = pd.DataFrame({'id': test_ids, 'label': test_preds})\n    submission.to_csv('submission.csv', index=False)\n    print(f\">> BERHASIL: submission.csv dibuat dengan {len(submission)} baris.\")\nelse:\n    print(f\">> ERROR: Jumlah ID ({len(test_ids)}) tidak sama dengan Prediksi ({len(test_preds)})\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:50:55.925635Z","iopub.execute_input":"2025-12-03T09:50:55.925986Z","iopub.status.idle":"2025-12-03T09:54:21.336880Z","shell.execute_reply.started":"2025-12-03T09:50:55.925965Z","shell.execute_reply":"2025-12-03T09:54:21.336180Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}