{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31194,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom kaggle_datasets import KaggleDatasets\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport re, os\n\nprint(f\"TensorFlow Version: {tf.__version__}\")\n\n# 1. Deteksi Hardware TPU\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() \n    print('Device:', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy()\n\nprint(f\"REPLICAS: {strategy.num_replicas_in_sync}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T03:36:55.882279Z","iopub.execute_input":"2025-12-03T03:36:55.882833Z","iopub.status.idle":"2025-12-03T03:36:55.887877Z","shell.execute_reply.started":"2025-12-03T03:36:55.882814Z","shell.execute_reply":"2025-12-03T03:36:55.886951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. Konfigurasi\nIMAGE_SIZE = [192, 192] \nEPOCHS = 10\n# Batch size disesuaikan dengan jumlah core TPU (8 core x 16 = 128)\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync \n\n# Path Dataset di Kaggle\nGCS_PATH = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-192x192'\n\n# Mengambil daftar nama file\nTRAINING_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/train/*.tfrec')\nVALIDATION_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/val/*.tfrec')\nTEST_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/test/*.tfrec')\n\n# Daftar Kelas (104 Kelas Bunga)\nCLASSES = ['pink primrose', 'hard-leaved pocket orchid', 'canterbury bells', 'sweet pea', 'wild geranium', 'tiger lily', 'moon orchid', 'bird of paradise', 'monkshood', 'globe thistle', 'snapdragon', \"colt's foot\", 'king protea', 'spear thistle', 'yellow iris', 'globe-flower', 'purple coneflower', 'peruvian lily', 'balloon flower', 'giant white arum lily', 'fire lily', 'pincushion flower', 'fritillary', 'red ginger', 'grape hyacinth', 'corn poppy', 'prince of wales feathers', 'stemless gentian', 'artichoke', 'sweet william', 'carnation', 'garden phlox', 'love in the mist', 'mexican aster', 'alpine sea holly', 'ruby-lipped cattleya', 'cape flower', 'great masterwort', 'siam tulip', 'lenten rose', 'barbeton daisy', 'daffodil', 'sword lily', 'poinsettia', 'bolero deep blue', 'wallflower', 'marigold', 'buttercup', 'daisy', 'common dandelion', 'petunia', 'wild pansy', 'primula', 'sunflower', 'lilac hibiscus', 'bishop of llandaff', 'gaillardia', 'gazania', 'azalea', 'water lily', 'rose', 'thorn apple', 'morning glory', 'passion flower', 'lotus', 'toad lily', 'anthurium', 'frangipani', 'clematis', 'hibiscus', 'columbine', 'desert-rose', 'tree mallow', 'magnolia', 'cyclamen ', 'watercress', 'canna lily', 'hippeastrum', 'bee balm', 'pink-yellow dahlia', 'bromelia', 'common tulip', 'wild rose', 'blanket flower', 'trumpet creeper', 'blackberry lily', 'common poppy', 'thorn apple']\n\nprint(f\"File Training ditemukan: {len(TRAINING_FILENAMES)}\")\nprint(f\"File Validation ditemukan: {len(VALIDATION_FILENAMES)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T03:36:58.476177Z","iopub.execute_input":"2025-12-03T03:36:58.476449Z","iopub.status.idle":"2025-12-03T03:36:58.497952Z","shell.execute_reply.started":"2025-12-03T03:36:58.476420Z","shell.execute_reply":"2025-12-03T03:36:58.497045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3. Helper Function untuk menghitung jumlah gambar dari nama file\ndef count_data_items(filenames):\n    # Format nama file: flowers00-230.tfrec (artinya ada 230 gambar di file itu)\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) for filename in filenames]\n    return np.sum(n)\n\nNUM_TRAINING_IMAGES = count_data_items(TRAINING_FILENAMES)\nNUM_VALIDATION_IMAGES = count_data_items(VALIDATION_FILENAMES)\n\n# Menghitung Steps per Epoch\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\n\nprint(f'Total Training Images: {NUM_TRAINING_IMAGES}')\nprint(f'Total Validation Images: {NUM_VALIDATION_IMAGES}')\nprint(f'Steps Per Epoch: {STEPS_PER_EPOCH}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T03:37:00.355596Z","iopub.execute_input":"2025-12-03T03:37:00.355820Z","iopub.status.idle":"2025-12-03T03:37:00.360640Z","shell.execute_reply.started":"2025-12-03T03:37:00.355801Z","shell.execute_reply":"2025-12-03T03:37:00.359663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4. Fungsi-fungsi Pembaca Data\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0  # Normalisasi ke 0-1\n    image = tf.reshape(image, [*IMAGE_SIZE, 3]) \n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), \n        \"class\": tf.io.FixedLenFeature([], tf.int64),  \n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label \n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), \n        \"id\": tf.io.FixedLenFeature([], tf.string), \n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False \n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=tf.data.AUTOTUNE)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)\n    return dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T03:37:03.326135Z","iopub.execute_input":"2025-12-03T03:37:03.326446Z","iopub.status.idle":"2025-12-03T03:37:03.331816Z","shell.execute_reply.started":"2025-12-03T03:37:03.326396Z","shell.execute_reply":"2025-12-03T03:37:03.331039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 5. Membuat Pipeline\ndef get_training_dataset():\n    dataset = load_dataset(TRAINING_FILENAMES, labeled=True)\n    dataset = dataset.repeat() # Wajib untuk TPU training loop\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(tf.data.AUTOTUNE)\n    return dataset\n\ndef get_validation_dataset():\n    dataset = load_dataset(VALIDATION_FILENAMES, labeled=True, ordered=True)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(tf.data.AUTOTUNE)\n    return dataset\n\ndef get_test_dataset(ordered=False):\n    dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(tf.data.AUTOTUNE)\n    return dataset\n\n# Inisialisasi dataset\ntrain_dataset = get_training_dataset()\nvalid_dataset = get_validation_dataset()\n\nprint(\"Pipeline dataset siap.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T03:37:16.979063Z","iopub.execute_input":"2025-12-03T03:37:16.979336Z","iopub.status.idle":"2025-12-03T03:37:17.019572Z","shell.execute_reply.started":"2025-12-03T03:37:16.979320Z","shell.execute_reply":"2025-12-03T03:37:17.018602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 6. Build Model\nwith strategy.scope():\n    model = tf.keras.Sequential([\n        # Input Layer\n        tf.keras.layers.Flatten(input_shape=[*IMAGE_SIZE, 3]),\n        \n        # Hidden Layers\n        tf.keras.layers.Dense(512, activation='relu'),\n        tf.keras.layers.Dropout(0.3), # Mencegah overfitting\n        tf.keras.layers.Dense(256, activation='relu'),\n        tf.keras.layers.Dropout(0.3),\n        \n        # Output Layer (104 Kelas)\n        tf.keras.layers.Dense(104, activation='softmax')\n    ])\n    \n    model.compile(\n        optimizer='adam',\n        loss = 'sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T03:37:20.358097Z","iopub.execute_input":"2025-12-03T03:37:20.358368Z","iopub.status.idle":"2025-12-03T03:37:20.454385Z","shell.execute_reply.started":"2025-12-03T03:37:20.358349Z","shell.execute_reply":"2025-12-03T03:37:20.453492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 7. Mulai Training\nprint(\"Mulai Training...\")\nhistory = model.fit(\n    train_dataset, \n    steps_per_epoch=STEPS_PER_EPOCH,\n    epochs=EPOCHS,\n    validation_data=valid_dataset\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T03:45:25.635874Z","iopub.execute_input":"2025-12-03T03:45:25.636136Z","iopub.status.idle":"2025-12-03T04:08:12.414360Z","shell.execute_reply.started":"2025-12-03T03:45:25.636121Z","shell.execute_reply":"2025-12-03T04:08:12.412953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 8. Plot Grafik\n# Grafik Loss (Semakin turun semakin baik)\nplt.figure(figsize=(10, 5))\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Loss Curve')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()\n\n# Grafik Akurasi (Semakin naik semakin baik)\nplt.figure(figsize=(10, 5))\nplt.plot(history.history['sparse_categorical_accuracy'], label='Training Acc')\nplt.plot(history.history['val_sparse_categorical_accuracy'], label='Validation Acc')\nplt.title('Accuracy Curve')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T04:08:50.857974Z","iopub.execute_input":"2025-12-03T04:08:50.858298Z","iopub.status.idle":"2025-12-03T04:08:51.118774Z","shell.execute_reply.started":"2025-12-03T04:08:50.858278Z","shell.execute_reply":"2025-12-03T04:08:51.117634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd # Kita gunakan pandas agar lebih mudah dan aman\n\n# --- 6. Membuat Submission (PERBAIKAN) ---\nprint(\"\\nMembuat Prediksi...\")\ntest_ds = get_test_dataset(ordered=True) # Ordered harus True agar urutan ID sesuai prediksi\n\n# Ambil gambar saja untuk prediksi\ntest_images_ds = test_ds.map(lambda image, idnum: image)\n\n# Lakukan prediksi\nprobabilities = model.predict(test_images_ds)\npredictions = np.argmax(probabilities, axis=-1)\nprint(f\"Jumlah Prediksi: {len(predictions)}\")\n\n# Ambil ID (PERBAIKAN DISINI)\n# Kita ambil ID secara terpisah dan pastikan mengambil SEMUANYA\nprint(\"Mengambil ID...\")\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\n# Kita batch dengan angka besar (misal 20000) untuk memastikan semua ID test (sekitar 7000an) terambil sekaligus\ntest_ids = next(iter(test_ids_ds.batch(20000))).numpy().astype('U') \nprint(f\"Jumlah ID: {len(test_ids)}\")\n\n# Cek apakah jumlahnya sama\nif len(test_ids) != len(predictions):\n    print(f\"ERROR: Masih ada mismatch! ID: {len(test_ids)}, Prediksi: {len(predictions)}\")\nelse:\n    # Simpan menggunakan Pandas (lebih aman dari np.savetxt untuk string/int mix)\n    submission = pd.DataFrame({'id': test_ids, 'label': predictions})\n    submission.to_csv('submission.csv', index=False)\n    print(\"Sukses! File submission.csv berhasil disimpan.\")\n    print(submission.head())","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}