{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Petals to the Metal - Flower Classification dengan Multi-Layer Perceptron\n# Kompetisi Kaggle: Klasifikasi 104 jenis bunga\n# Metrik: Macro F1 Score\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import f1_score, classification_report\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers, models, callbacks\nimport os\nimport re\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Set random seed untuk reproducibility\nnp.random.seed(42)\ntf.random.set_seed(42)\n\nprint(\"TensorFlow Version:\", tf.__version__)\nprint(\"GPU Available:\", len(tf.config.list_physical_devices('GPU')) > 0)\n\n# ==================== KONFIGURASI ====================\nclass Config:\n    # Pilih resolusi (gunakan 224x224 untuk keseimbangan speed vs accuracy)\n    RESOLUTION = 224\n    IMG_CHANNELS = 3\n    INPUT_DIM = RESOLUTION * RESOLUTION * IMG_CHANNELS  # 150,528\n    \n    # Model parameters\n    BATCH_SIZE = 32  # Reduce untuk menghindari memory issue\n    EPOCHS = 50\n    LEARNING_RATE = 0.001\n    \n    # MLP Architecture (lebih kecil untuk menghindari overfitting)\n    HIDDEN_LAYERS = [512, 256, 128]\n    DROPOUT_RATE = 0.3\n    \n    # Training\n    EARLY_STOPPING_PATIENCE = 15\n    \n    # Classes\n    NUM_CLASSES = 104\n\nconfig = Config()\n\nprint(f\"\\nConfiguration:\")\nprint(f\"  Image Size: {config.RESOLUTION}x{config.RESOLUTION}\")\nprint(f\"  Input Dimension: {config.INPUT_DIM:,}\")\nprint(f\"  Batch Size: {config.BATCH_SIZE}\")\nprint(f\"  Number of Classes: {config.NUM_CLASSES}\")\n\n# ==================== FIND DATA PATH ====================\nprint(\"\\n\" + \"=\"*60)\nprint(\"LOCATING DATA FILES\")\nprint(\"=\"*60)\n\ndef find_data_path():\n    \"\"\"Find the correct path to the competition data\"\"\"\n    possible_paths = [\n        '/kaggle/input/tpu-getting-started',\n        '/kaggle/input/petals-to-the-metal-flower-classification',\n        '/kaggle/input/petals-to-the-metal',\n        '../input/tpu-getting-started',\n        './data'\n    ]\n    \n    for path in possible_paths:\n        if os.path.exists(path):\n            print(f\"✓ Found data at: {path}\")\n            return path\n    \n    # List available inputs\n    if os.path.exists('/kaggle/input'):\n        print(\"\\nAvailable datasets in /kaggle/input:\")\n        for item in os.listdir('/kaggle/input'):\n            print(f\"  - {item}\")\n    \n    return None\n\nbase_path = find_data_path()\n\nif base_path is None:\n    print(\"\\n⚠ WARNING: Data path not found!\")\n    print(\"Please add the competition dataset in Kaggle Notebook:\")\n    print(\"  1. Click 'Add Data' button\")\n    print(\"  2. Search for 'Petals to the Metal' or 'TPU Getting Started'\")\n    print(\"  3. Add the dataset\")\n    raise FileNotFoundError(\"Competition dataset not found\")\n\n# Find TFRecord files with different resolutions\nprint(\"\\nSearching for TFRecord files...\")\ntfrecord_paths = []\nfor resolution in ['512x512', '331x331', '224x224', '192x192']:\n    path = os.path.join(base_path, f'tfrecords-jpeg-{resolution}')\n    if os.path.exists(path):\n        tfrecord_paths.append((resolution, path))\n        print(f\"  ✓ Found: tfrecords-jpeg-{resolution}\")\n\nif not tfrecord_paths:\n    print(\"  Searching in root directory...\")\n    tfrecord_paths = [(str(config.RESOLUTION), base_path)]\n\n# Use the first available resolution\nselected_resolution, data_path = tfrecord_paths[0] if tfrecord_paths else (str(config.RESOLUTION), base_path)\nprint(f\"\\n✓ Using resolution: {selected_resolution}\")\n\nTRAIN_PATH = os.path.join(data_path, 'train')\nVAL_PATH = os.path.join(data_path, 'val')\nTEST_PATH = os.path.join(data_path, 'test')\n\n# ==================== DATA LOADING FUNCTIONS ====================\nprint(\"\\n\" + \"=\"*60)\nprint(\"SETTING UP DATA PIPELINE\")\nprint(\"=\"*60)\n\nAUTO = tf.data.AUTOTUNE\n\ndef decode_image(image_data):\n    \"\"\"Decode and preprocess image\"\"\"\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0\n    image = tf.image.resize(image, [config.RESOLUTION, config.RESOLUTION])\n    return image\n\ndef read_labeled_tfrecord(example):\n    \"\"\"Parse labeled TFRecord\"\"\"\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    \"\"\"Parse unlabeled TFRecord (test set)\"\"\"\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\ndef flatten_for_mlp(image, label):\n    \"\"\"Flatten image untuk MLP\"\"\"\n    image = tf.reshape(image, [config.INPUT_DIM])\n    return image, label\n\ndef flatten_for_mlp_test(image, idnum):\n    \"\"\"Flatten image untuk MLP (test)\"\"\"\n    image = tf.reshape(image, [config.INPUT_DIM])\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    \"\"\"Load and parse TFRecord dataset\"\"\"\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False\n    \n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    dataset = dataset.with_options(ignore_order)\n    \n    if labeled:\n        dataset = dataset.map(read_labeled_tfrecord, num_parallel_calls=AUTO)\n        dataset = dataset.map(flatten_for_mlp, num_parallel_calls=AUTO)\n    else:\n        dataset = dataset.map(read_unlabeled_tfrecord, num_parallel_calls=AUTO)\n        dataset = dataset.map(flatten_for_mlp_test, num_parallel_calls=AUTO)\n    \n    return dataset\n\ndef get_dataset(files, labeled=True, ordered=False, repeated=False, batch_size=32):\n    \"\"\"Complete dataset pipeline\"\"\"\n    dataset = load_dataset(files, labeled=labeled, ordered=ordered)\n    \n    if repeated:\n        dataset = dataset.repeat()\n    \n    if not ordered:\n        dataset = dataset.shuffle(2048)\n    \n    dataset = dataset.batch(batch_size)\n    dataset = dataset.prefetch(AUTO)\n    \n    return dataset\n\n# ==================== LOAD DATA ====================\nprint(\"\\nLoading TFRecord files...\")\n\n# Get file paths\ntrain_files = tf.io.gfile.glob(TRAIN_PATH + '/*.tfrec')\nval_files = tf.io.gfile.glob(VAL_PATH + '/*.tfrec')\ntest_files = tf.io.gfile.glob(TEST_PATH + '/*.tfrec')\n\nprint(f\"Training files found: {len(train_files)}\")\nprint(f\"Validation files found: {len(val_files)}\")\nprint(f\"Test files found: {len(test_files)}\")\n\n# Check if files exist\nif len(train_files) == 0 and len(val_files) == 0:\n    print(\"\\n⚠ WARNING: No TFRecord files found!\")\n    print(\"Attempting alternative search...\")\n    \n    # Try to find files in subdirectories\n    for root, dirs, files in os.walk(data_path):\n        tfrec_files = [f for f in files if f.endswith('.tfrec')]\n        if tfrec_files:\n            print(f\"  Found {len(tfrec_files)} .tfrec files in: {root}\")\n            if 'train' in root:\n                train_files = [os.path.join(root, f) for f in tfrec_files]\n            elif 'val' in root:\n                val_files = [os.path.join(root, f) for f in tfrec_files]\n            elif 'test' in root:\n                test_files = [os.path.join(root, f) for f in tfrec_files]\n\nif len(train_files) == 0:\n    raise FileNotFoundError(\n        \"No training files found! Please ensure:\\n\"\n        \"1. The competition dataset is added to your notebook\\n\"\n        \"2. The data contains tfrecords-jpeg-{resolution}/train/*.tfrec files\\n\"\n        \"3. You're using the correct competition: 'Petals to the Metal' or 'TPU Getting Started'\"\n    )\n\n# Count samples\nprint(\"\\nCounting samples...\")\nnum_training_images = int(np.sum([int(re.compile(r\"-([0-9]*)\\.\").search(f).group(1)) \n                                   for f in train_files]))\nnum_validation_images = int(np.sum([int(re.compile(r\"-([0-9]*)\\.\").search(f).group(1)) \n                                     for f in val_files])) if val_files else 0\nnum_test_images = int(np.sum([int(re.compile(r\"-([0-9]*)\\.\").search(f).group(1)) \n                               for f in test_files])) if test_files else 0\n\nprint(f\"Training samples: {num_training_images:,}\")\nprint(f\"Validation samples: {num_validation_images:,}\")\nprint(f\"Test samples: {num_test_images:,}\")\n\n# ==================== PREPARE DATASETS ====================\nprint(\"\\n\" + \"=\"*60)\nprint(\"PREPARING DATASETS\")\nprint(\"=\"*60)\n\n# Calculate steps\nsteps_per_epoch = num_training_images // config.BATCH_SIZE\nvalidation_steps = num_validation_images // config.BATCH_SIZE if num_validation_images > 0 else None\n\nprint(f\"Steps per epoch: {steps_per_epoch}\")\nif validation_steps:\n    print(f\"Validation steps: {validation_steps}\")\n\n# Create datasets\ntrain_dataset = get_dataset(train_files, labeled=True, ordered=False, \n                           batch_size=config.BATCH_SIZE)\nval_dataset = get_dataset(val_files, labeled=True, ordered=True, \n                         batch_size=config.BATCH_SIZE) if val_files else None\ntest_dataset = get_dataset(test_files, labeled=False, ordered=True, \n                          batch_size=config.BATCH_SIZE) if test_files else None\n\nprint(\"✓ Datasets prepared successfully\")\n\n# ==================== VISUALIZE SAMPLE IMAGES ====================\nprint(\"\\n\" + \"=\"*60)\nprint(\"VISUALIZING SAMPLE IMAGES\")\nprint(\"=\"*60)\n\ndef display_batch_of_images(images_batch, labels_batch=None, predictions_batch=None):\n    \"\"\"Display a batch of images in a grid\"\"\"\n    batch_size = images_batch.shape[0]\n    \n    # Determine grid size\n    if batch_size <= 9:\n        rows, cols = 3, 3\n    elif batch_size <= 16:\n        rows, cols = 4, 4\n    elif batch_size <= 20:\n        rows, cols = 4, 5\n    else:\n        rows, cols = 5, 5\n    \n    fig, axes = plt.subplots(rows, cols, figsize=(cols * 3, rows * 3))\n    axes = axes.flatten()\n    \n    for idx in range(min(batch_size, rows * cols)):\n        ax = axes[idx]\n        \n        # Reshape from flattened to image if needed\n        if len(images_batch[idx].shape) == 1:\n            # Image is flattened, reshape it\n            image = images_batch[idx].numpy().reshape(config.RESOLUTION, config.RESOLUTION, 3)\n        else:\n            image = images_batch[idx].numpy()\n        \n        # Ensure image is in valid range [0, 1]\n        image = np.clip(image, 0, 1)\n        \n        ax.imshow(image)\n        ax.axis('off')\n        \n        # Add title with label/prediction\n        title = \"\"\n        if labels_batch is not None:\n            label = labels_batch[idx].numpy() if hasattr(labels_batch[idx], 'numpy') else labels_batch[idx]\n            title = f\"Class: {label}\"\n        \n        if predictions_batch is not None:\n            pred = predictions_batch[idx]\n            if labels_batch is not None:\n                title += f\"\\nPred: {pred}\"\n            else:\n                title = f\"Pred: {pred}\"\n        \n        if title:\n            ax.set_title(title, fontsize=10, fontweight='bold')\n    \n    # Hide empty subplots\n    for idx in range(batch_size, len(axes)):\n        axes[idx].axis('off')\n    \n    plt.tight_layout()\n    plt.savefig('sample_images.png', dpi=150, bbox_inches='tight')\n    plt.show()\n\n# Create dataset for visualization (before flattening)\nprint(\"\\nPreparing samples for visualization...\")\nviz_dataset = load_dataset(train_files[:1], labeled=True, ordered=True)\nviz_dataset = viz_dataset.batch(20)\n\ntry:\n    # Get one batch\n    ds_iter = iter(viz_dataset)\n    one_batch_images, one_batch_labels = next(ds_iter)\n    \n    print(f\"Displaying {len(one_batch_images)} sample images from training set...\")\n    display_batch_of_images(one_batch_images, one_batch_labels)\n    print(\"✓ Sample images saved as: sample_images.png\")\n    \nexcept Exception as e:\n    print(f\"⚠ Could not display images: {e}\")\n    print(\"  This is normal if there are very few samples or format issues\")\n\n# ==================== BUILD MLP MODEL ====================\nprint(\"\\n\" + \"=\"*60)\nprint(\"BUILDING MULTI-LAYER PERCEPTRON MODEL\")\nprint(\"=\"*60)\n\ndef create_mlp_model():\n    \"\"\"Create Multi-Layer Perceptron for flower classification\"\"\"\n    model = models.Sequential(name='MLP_Flower_Classifier')\n    \n    # Input layer\n    model.add(layers.Input(shape=(config.INPUT_DIM,), name='input'))\n    \n    # Hidden layers\n    for i, units in enumerate(config.HIDDEN_LAYERS, 1):\n        model.add(layers.Dense(units, name=f'dense_{i}'))\n        model.add(layers.BatchNormalization(name=f'bn_{i}'))\n        model.add(layers.Activation('relu', name=f'relu_{i}'))\n        model.add(layers.Dropout(config.DROPOUT_RATE, name=f'dropout_{i}'))\n    \n    # Output layer\n    model.add(layers.Dense(config.NUM_CLASSES, activation='softmax', name='output'))\n    \n    return model\n\nmodel = create_mlp_model()\n\nmodel.compile(\n    optimizer=keras.optimizers.Adam(learning_rate=config.LEARNING_RATE),\n    loss='sparse_categorical_crossentropy',\n    metrics=['accuracy']\n)\n\nprint(model.summary())\nprint(f\"\\nTotal Parameters: {model.count_params():,}\")\n\n# ==================== CALLBACKS ====================\nprint(\"\\n\" + \"=\"*60)\nprint(\"SETTING UP CALLBACKS\")\nprint(\"=\"*60)\n\ncallback_list = [\n    callbacks.EarlyStopping(\n        monitor='val_loss' if val_dataset else 'loss',\n        patience=config.EARLY_STOPPING_PATIENCE,\n        restore_best_weights=True,\n        verbose=1\n    ),\n    callbacks.ReduceLROnPlateau(\n        monitor='val_loss' if val_dataset else 'loss',\n        factor=0.5,\n        patience=5,\n        min_lr=1e-7,\n        verbose=1\n    ),\n    callbacks.ModelCheckpoint(\n        'best_mlp_model.h5',\n        monitor='val_loss' if val_dataset else 'loss',\n        save_best_only=True,\n        verbose=1\n    )\n]\n\nprint(\"✓ Callbacks configured\")\n\n# ==================== TRAINING ====================\nprint(\"\\n\" + \"=\"*60)\nprint(\"TRAINING MODEL\")\nprint(\"=\"*60)\n\nhistory = model.fit(\n    train_dataset,\n    steps_per_epoch=steps_per_epoch,\n    validation_data=val_dataset,\n    validation_steps=validation_steps,\n    epochs=config.EPOCHS,\n    callbacks=callback_list,\n    verbose=1\n)\n\n# ==================== EVALUATION ====================\nprint(\"\\n\" + \"=\"*60)\nprint(\"EVALUATING MODEL\")\nprint(\"=\"*60)\n\nif val_dataset:\n    val_loss, val_accuracy = model.evaluate(val_dataset, steps=validation_steps)\n    print(f\"\\nValidation Loss: {val_loss:.4f}\")\n    print(f\"Validation Accuracy: {val_accuracy:.4f}\")\n    \n    # Calculate Macro F1\n    print(\"\\nCalculating Macro F1 Score...\")\n    y_true, y_pred = [], []\n    \n    for images, labels in val_dataset.take(validation_steps):\n        predictions = model.predict(images, verbose=0)\n        y_pred.extend(np.argmax(predictions, axis=1))\n        y_true.extend(labels.numpy())\n    \n    macro_f1 = f1_score(y_true, y_pred, average='macro')\n    print(f\"Macro F1 Score: {macro_f1:.4f}\")\n    \n    # ==================== VISUALIZE PREDICTIONS ====================\n    print(\"\\n\" + \"=\"*60)\n    print(\"VISUALIZING PREDICTIONS ON VALIDATION SET\")\n    print(\"=\"*60)\n    \n    try:\n        # Get validation samples for visualization (before flattening)\n        viz_val_dataset = load_dataset(val_files[:1], labeled=True, ordered=True)\n        viz_val_dataset = viz_val_dataset.batch(20)\n        \n        viz_iter = iter(viz_val_dataset)\n        viz_images, viz_labels = next(viz_iter)\n        \n        # Flatten for prediction\n        viz_images_flat = tf.reshape(viz_images, [viz_images.shape[0], config.INPUT_DIM])\n        \n        # Make predictions\n        viz_predictions = model.predict(viz_images_flat, verbose=0)\n        viz_pred_classes = np.argmax(viz_predictions, axis=1)\n        \n        print(f\"Displaying predictions on {len(viz_images)} validation samples...\")\n        display_batch_of_images(viz_images, viz_labels, viz_pred_classes)\n        print(\"✓ Prediction visualization saved as: sample_images.png\")\n        \n    except Exception as e:\n        print(f\"⚠ Could not visualize predictions: {e}\")\n\n# ==================== VISUALIZATION ====================\nfig, axes = plt.subplots(1, 2, figsize=(15, 5))\n\naxes[0].plot(history.history['accuracy'], label='Train', linewidth=2)\nif 'val_accuracy' in history.history:\n    axes[0].plot(history.history['val_accuracy'], label='Validation', linewidth=2)\naxes[0].set_title('Model Accuracy', fontsize=14, fontweight='bold')\naxes[0].set_xlabel('Epoch')\naxes[0].set_ylabel('Accuracy')\naxes[0].legend()\naxes[0].grid(True, alpha=0.3)\n\naxes[1].plot(history.history['loss'], label='Train', linewidth=2)\nif 'val_loss' in history.history:\n    axes[1].plot(history.history['val_loss'], label='Validation', linewidth=2)\naxes[1].set_title('Model Loss', fontsize=14, fontweight='bold')\naxes[1].set_xlabel('Epoch')\naxes[1].set_ylabel('Loss')\naxes[1].legend()\naxes[1].grid(True, alpha=0.3)\n\nplt.tight_layout()\nplt.savefig('training_history.png', dpi=300, bbox_inches='tight')\nplt.show()\n\n# ==================== PREDICTIONS ====================\nprint(\"\\n\" + \"=\"*60)\nprint(\"GENERATING TEST PREDICTIONS\")\nprint(\"=\"*60)\n\nif test_dataset:\n    print(\"Making predictions on test set...\")\n    test_ids = []\n    test_preds = []\n    \n    for images, ids in test_dataset:\n        predictions = model.predict(images, verbose=0)\n        pred_classes = np.argmax(predictions, axis=1)\n        test_preds.extend(pred_classes)\n        test_ids.extend([id.numpy().decode('utf-8') for id in ids])\n    \n    # Create submission\n    submission = pd.DataFrame({\n        'id': test_ids,\n        'label': test_preds\n    })\n    \n    submission.to_csv('submission.csv', index=False)\n    print(f\"\\n✓ Submission file created: submission.csv\")\n    print(f\"  Total predictions: {len(submission)}\")\n    print(\"\\nFirst 10 predictions:\")\n    print(submission.head(10))\nelse:\n    print(\"⚠ No test files found, skipping prediction\")\n\n# ==================== SUMMARY ====================\nprint(\"\\n\" + \"=\"*60)\nprint(\"FINAL SUMMARY\")\nprint(\"=\"*60)\nprint(f\"Model: Multi-Layer Perceptron\")\nprint(f\"  Hidden Layers: {config.HIDDEN_LAYERS}\")\nprint(f\"  Total Parameters: {model.count_params():,}\")\nprint(f\"\\nDataset:\")\nprint(f\"  Training: {num_training_images:,} images\")\nprint(f\"  Validation: {num_validation_images:,} images\")\nprint(f\"  Test: {num_test_images:,} images\")\nprint(f\"\\nPerformance:\")\nprint(f\"  Best Training Accuracy: {max(history.history['accuracy']):.4f}\")\nif 'val_accuracy' in history.history:\n    print(f\"  Best Validation Accuracy: {max(history.history['val_accuracy']):.4f}\")\n    print(f\"  Macro F1 Score: {macro_f1:.4f}\")\nprint(\"\\n✓ Training completed!\")\nprint(\"=\"*60)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-02T13:32:22.931813Z","iopub.execute_input":"2025-12-02T13:32:22.932707Z"}},"outputs":[],"execution_count":null}]}