{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !pip install -q \"protobuf==3.20.3\"\n# Tekan tombol \"Restart Session\" setelah instalasi ini selesai!\nimport os\nos.system('pip install -q \"protobuf==3.20.3\"')\nprint(\"Instalasi selesai. Silakan Restart Session jika belum.\")\nimport re\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom kaggle_datasets import KaggleDatasets\nimport matplotlib.pyplot as plt\n\nprint(f\"Menggunakan TensorFlow v{tf.__version__}\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:25:07.145239Z","iopub.execute_input":"2025-12-03T11:25:07.145566Z","iopub.status.idle":"2025-12-03T11:25:11.182597Z","shell.execute_reply.started":"2025-12-03T11:25:07.145544Z","shell.execute_reply":"2025-12-03T11:25:11.181360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def setup_accelerator():\n    try:\n        # Mencoba mendeteksi keberadaan TPU\n        resolver = tf.distribute.cluster_resolver.TPUClusterResolver()\n        print(f'Device terdeteksi: {resolver.master()}')\n        tf.config.experimental_connect_to_cluster(resolver)\n        tf.tpu.experimental.initialize_tpu_system(resolver)\n        return tf.distribute.experimental.TPUStrategy(resolver)\n    except ValueError:\n        # Fallback jika tidak ada TPU\n        return tf.distribute.get_strategy()\n\ntpu_strategy = setup_accelerator()\nNUM_REPLICAS = tpu_strategy.num_replicas_in_sync\nprint(f\"Jumlah Replicas: {NUM_REPLICAS}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:25:17.767246Z","iopub.execute_input":"2025-12-03T11:25:17.767561Z","iopub.status.idle":"2025-12-03T11:25:17.774989Z","shell.execute_reply.started":"2025-12-03T11:25:17.767540Z","shell.execute_reply":"2025-12-03T11:25:17.773943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"GCS_BUCKET = KaggleDatasets().get_gcs_path('tpu-getting-started')\n\n# Konfigurasi Ukuran & Training\n# Kita pakai ukuran kecil (64x64) agar MLP training-nya ngebut\nIMG_DIM = [64, 64] \nEPOCH_COUNT = 25\nGLOBAL_BATCH_SIZE = 16 * NUM_REPLICAS\n\n# Path Data TFRecord\nGCS_PATTERN = f'{GCS_BUCKET}/tfrecords-jpeg-192x192' \nTRAIN_FILES = tf.io.gfile.glob(f'{GCS_PATTERN}/train/*.tfrec')\nVAL_FILES   = tf.io.gfile.glob(f'{GCS_PATTERN}/val/*.tfrec')\nTEST_FILES  = tf.io.gfile.glob(f'{GCS_PATTERN}/test/*.tfrec')\n\n# Daftar Label Bunga\nFLOWER_LABELS = [\n    'pink primrose', 'hard-leaved pocket orchid', 'canterbury bells', 'sweet pea', \n    'wild geranium', 'tiger lily', 'moon orchid', 'bird of paradise', 'monkshood', \n    'globe thistle', 'snapdragon', \"colt's foot\", 'king protea', 'spear thistle', \n    'yellow iris', 'globe-flower', 'purple coneflower', 'peruvian lily', \n    'balloon flower', 'giant white arum lily', 'fire lily', 'pincushion flower', \n    'fritillary', 'red ginger', 'grape hyacinth', 'corn poppy', 'prince of wales feathers', \n    'stemless gentian', 'artichoke', 'sweet william', 'carnation', 'garden phlox', \n    'love in the mist', 'cosmos', 'alpine sea holly', 'ruby-lipped cattleya', \n    'cape flower', 'great masterwort', 'siam tulip', 'lenten rose', 'barbeton daisy', \n    'daffodil', 'sword lily', 'poinsettia', 'bolero deep blue', 'wallflower', 'marigold', \n    'buttercup', 'daisy', 'common dandelion', 'petunia', 'wild pansy', 'primula', \n    'sunflower', 'lilac hibiscus', 'bishop of llandaff', 'gaillardia', 'geranium', \n    'orange dahlia', 'pink-yellow dahlia', 'cautleya spicata', 'japanese anemone', \n    'black-eyed susan', 'silverbush', 'californian poppy', 'osteospermum', \n    'spring crocus', 'iris', 'windflower', 'tree poppy', 'gazania', 'azalea', \n    'water lily', 'rose', 'thorn apple', 'morning glory', 'passion flower', \n    'lotus', 'toad lily', 'anthurium', 'frangipani', 'clematis', 'hibiscus', \n    'columbine', 'desert-rose', 'tree mallow', 'magnolia', 'cyclamen ', \n    'watercress', 'canna lily', 'hippeastrum', 'bee balm', 'pink quill', 'foxglove', \n    'bougainvillea', 'camellia', 'mallow', 'mexican petunia', 'bromelia', \n    'blanket flower', 'trumpet creeper', 'blackberry lily', 'common tulip', \n    'wild rose'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:25:21.609892Z","iopub.execute_input":"2025-12-03T11:25:21.610180Z","iopub.status.idle":"2025-12-03T11:25:22.342433Z","shell.execute_reply.started":"2025-12-03T11:25:21.610160Z","shell.execute_reply":"2025-12-03T11:25:22.341466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"AUTO_TUNE = tf.data.experimental.AUTOTUNE\n\ndef process_img_data(img_bytes):\n    \"\"\"Mendecode JPEG dan resize agar ringan untuk MLP.\"\"\"\n    img = tf.image.decode_jpeg(img_bytes, channels=3)\n    img = tf.cast(img, tf.float32) / 255.0 # Normalisasi 0-1\n    img = tf.image.resize(img, IMG_DIM)     # Resize penting!\n    img = tf.reshape(img, [*IMG_DIM, 3])\n    return img\n\ndef parse_labeled_data(example_proto):\n    schema = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    parsed = tf.io.parse_single_example(example_proto, schema)\n    return process_img_data(parsed['image']), tf.cast(parsed['class'], tf.int32)\n\ndef parse_test_data(example_proto):\n    schema = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    parsed = tf.io.parse_single_example(example_proto, schema)\n    return process_img_data(parsed['image']), parsed['id']\n\ndef create_dataset(files, is_labeled=True, is_ordered=False):\n    options = tf.data.Options()\n    if not is_ordered:\n        options.experimental_deterministic = False\n        \n    ds = tf.data.TFRecordDataset(files, num_parallel_reads=AUTO_TUNE)\n    ds = ds.with_options(options)\n    \n    parser = parse_labeled_data if is_labeled else parse_test_data\n    ds = ds.map(parser, num_parallel_calls=AUTO_TUNE)\n    return ds\n\n# Menyiapkan generator data\ndef get_train_pipeline():\n    ds = create_dataset(TRAIN_FILES, is_labeled=True)\n    ds = ds.repeat().shuffle(2048)\n    ds = ds.batch(GLOBAL_BATCH_SIZE).prefetch(AUTO_TUNE)\n    return ds\n\ndef get_val_pipeline():\n    ds = create_dataset(VAL_FILES, is_labeled=True, is_ordered=False)\n    ds = ds.batch(GLOBAL_BATCH_SIZE).cache().prefetch(AUTO_TUNE)\n    return ds\n\ndef get_test_pipeline(ordered=False):\n    ds = create_dataset(TEST_FILES, is_labeled=False, is_ordered=ordered)\n    ds = ds.batch(GLOBAL_BATCH_SIZE).prefetch(AUTO_TUNE)\n    return ds\n\n# Menghitung total gambar\ndef count_imgs(filenames):\n    total = sum([int(re.compile(r\"-([0-9]*)\\.\").search(f).group(1)) for f in filenames])\n    return total\n\nNUM_TRAIN = count_imgs(TRAIN_FILES)\nNUM_TEST = count_imgs(TEST_FILES)\nSTEPS_PER_EPOCH = NUM_TRAIN // GLOBAL_BATCH_SIZE","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:25:25.928989Z","iopub.execute_input":"2025-12-03T11:25:25.929307Z","iopub.status.idle":"2025-12-03T11:25:25.942520Z","shell.execute_reply.started":"2025-12-03T11:25:25.929285Z","shell.execute_reply":"2025-12-03T11:25:25.941574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_mlp_model():\n    with tpu_strategy.scope():\n        # Definisi Layer secara berurutan\n        layers = [\n            tf.keras.Input(shape=[*IMG_DIM, 3]),\n            tf.keras.layers.Flatten(),\n            \n            # Hidden Layer 1\n            tf.keras.layers.Dense(512, activation='relu'),\n            tf.keras.layers.BatchNormalization(),\n            tf.keras.layers.Dropout(0.3),\n            \n            # Hidden Layer 2\n            tf.keras.layers.Dense(256, activation='relu'),\n            tf.keras.layers.BatchNormalization(),\n            tf.keras.layers.Dropout(0.3),\n            \n            # Hidden Layer 3\n            tf.keras.layers.Dense(128, activation='relu'),\n            tf.keras.layers.BatchNormalization(),\n            \n            # Output Layer\n            tf.keras.layers.Dense(len(FLOWER_LABELS), activation='softmax')\n        ]\n        \n        model = tf.keras.Sequential(layers)\n        \n        model.compile(\n            optimizer='adam',\n            loss='sparse_categorical_crossentropy',\n            metrics=['sparse_categorical_accuracy']\n        )\n    return model\n\nmodel = build_mlp_model()\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:25:30.471212Z","iopub.execute_input":"2025-12-03T11:25:30.471580Z","iopub.status.idle":"2025-12-03T11:25:30.600630Z","shell.execute_reply.started":"2025-12-03T11:25:30.471556Z","shell.execute_reply":"2025-12-03T11:25:30.599723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def lr_schedule(epoch):\n    start_lr = 0.00001\n    min_lr = 0.00001\n    max_lr = 0.00005 * NUM_REPLICAS\n    rampup = 5\n    sustain = 0\n    decay = 0.8\n\n    if epoch < rampup:\n        return (max_lr - start_lr) / rampup * epoch + start_lr\n    elif epoch < rampup + sustain:\n        return max_lr\n    else:\n        return (max_lr - min_lr) * decay**(epoch - rampup - sustain) + min_lr\n\nlr_controller = tf.keras.callbacks.LearningRateScheduler(lr_schedule, verbose=1)\n\nprint(\"\\n>>> MEMULAI TRAINING MODEL (MLP)...\")\nhistory = model.fit(\n    get_train_pipeline(),\n    steps_per_epoch=STEPS_PER_EPOCH,\n    epochs=EPOCH_COUNT,\n    validation_data=get_val_pipeline(),\n    callbacks=[lr_controller]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:25:35.198902Z","iopub.execute_input":"2025-12-03T11:25:35.199910Z","iopub.status.idle":"2025-12-03T11:42:58.611380Z","shell.execute_reply.started":"2025-12-03T11:25:35.199877Z","shell.execute_reply":"2025-12-03T11:42:58.609948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_history(hist):\n    acc = hist.history['sparse_categorical_accuracy']\n    val_acc = hist.history['val_sparse_categorical_accuracy']\n    loss = hist.history['loss']\n    val_loss = hist.history['val_loss']\n    epochs_range = range(len(acc))\n\n    plt.figure(figsize=(15, 5))\n    \n    plt.subplot(1, 2, 1)\n    plt.plot(epochs_range, acc, label='Training Accuracy')\n    plt.plot(epochs_range, val_acc, label='Validation Accuracy')\n    plt.legend(loc='lower right')\n    plt.title('Training and Validation Accuracy')\n\n    plt.subplot(1, 2, 2)\n    plt.plot(epochs_range, loss, label='Training Loss')\n    plt.plot(epochs_range, val_loss, label='Validation Loss')\n    plt.legend(loc='upper right')\n    plt.title('Training and Validation Loss')\n    plt.show()\n\nplot_history(history)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:43:47.053717Z","iopub.execute_input":"2025-12-03T11:43:47.054023Z","iopub.status.idle":"2025-12-03T11:43:47.514851Z","shell.execute_reply.started":"2025-12-03T11:43:47.054001Z","shell.execute_reply":"2025-12-03T11:43:47.513502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n>>> MEMPROSES DATA TEST & PREDIKSI...\")\n\n# Mengambil dataset test secara terurut\ntest_ds = get_test_pipeline(ordered=True)\nraw_images = test_ds.map(lambda img, idnum: img)\n\n# Prediksi probabilitas\nprobs = model.predict(raw_images)\npredicted_classes = np.argmax(probs, axis=-1)\n\n# Mengambil ID gambar\nid_ds = test_ds.map(lambda img, idnum: idnum).unbatch()\ntest_ids = next(iter(id_ds.batch(NUM_TEST))).numpy().astype('U')\n\n# Menyusun DataFrame\ndf_submission = pd.DataFrame({\n    'id': test_ids,\n    'label': predicted_classes\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:44:03.876240Z","iopub.execute_input":"2025-12-03T11:44:03.876527Z","iopub.status.idle":"2025-12-03T11:44:13.682035Z","shell.execute_reply.started":"2025-12-03T11:44:03.876509Z","shell.execute_reply":"2025-12-03T11:44:13.681228Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filename = 'sample_submission.csv'\ndf_submission.to_csv(filename, index=False)\nprint(f\"\\n>>> SUKSES! File '{filename}' berhasil disimpan.\")\nprint(df_submission.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T11:44:49.460282Z","iopub.execute_input":"2025-12-03T11:44:49.460674Z","iopub.status.idle":"2025-12-03T11:44:49.491816Z","shell.execute_reply.started":"2025-12-03T11:44:49.460622Z","shell.execute_reply":"2025-12-03T11:44:49.490607Z"}},"outputs":[],"execution_count":null}]}