{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T08:48:46.908423Z","iopub.execute_input":"2025-12-03T08:48:46.908777Z","iopub.status.idle":"2025-12-03T08:48:46.932538Z","shell.execute_reply.started":"2025-12-03T08:48:46.908741Z","shell.execute_reply":"2025-12-03T08:48:46.931605Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# MULTI-LAYER PERCEPTRON (MLP) - FLOWER CLASSIFICATION","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers, models\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\nimport matplotlib.pyplot as plt\n\n# Set random seed untuk reproducibility\nnp.random.seed(42)\ntf.random.set_seed(42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T12:42:13.798068Z","iopub.execute_input":"2025-12-03T12:42:13.798865Z","iopub.status.idle":"2025-12-03T12:42:13.805475Z","shell.execute_reply.started":"2025-12-03T12:42:13.798835Z","shell.execute_reply":"2025-12-03T12:42:13.804363Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n[1] Loading Training Data from TFRecord...\")\n\n# Fungsi untuk decode TFRecord\ndef decode_image(image_data):\n    \"\"\"Decode image data dari TFRecord\"\"\"\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.image.resize(image, [224, 224])\n    image = image / 255.0  # Normalisasi ke range [0, 1]\n    return image\n\ndef read_tfrecord(example):\n    \"\"\"Parse TFRecord example - Features: id, class (label), image\"\"\"\n    feature_description = {\n        'id': tf.io.FixedLenFeature([], tf.string),\n        'class': tf.io.FixedLenFeature([], tf.int64),  # ✓ BENAR: 'class' bukan 'label'\n        'image': tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, feature_description)\n    image = decode_image(example['image'])\n    label = example['class']  # ✓ Extract 'class' sebagai label\n    return image, label\n\ndef read_tfrecord_test(example):\n    \"\"\"Parse TFRecord example untuk test data (tanpa label/class)\"\"\"\n    feature_description = {\n        'id': tf.io.FixedLenFeature([], tf.string),\n        'image': tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, feature_description)\n    image = decode_image(example['image'])\n    id_str = example['id']\n    return image, id_str\n\n# Load data dari folder Kaggle\nTRAIN_FILENAMES = tf.io.gfile.glob('/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224/train/*.tfrec')\nVAL_FILENAMES = tf.io.gfile.glob('/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224/val/*.tfrec')\nTEST_FILENAMES = tf.io.gfile.glob('/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224/test/*.tfrec')\n\n# Fallback jika path berbeda\nif len(TRAIN_FILENAMES) == 0:\n    TRAIN_FILENAMES = tf.io.gfile.glob('/kaggle/input/*/train/*.tfrec')\nif len(VAL_FILENAMES) == 0:\n    VAL_FILENAMES = tf.io.gfile.glob('/kaggle/input/*/val/*.tfrec')\nif len(TEST_FILENAMES) == 0:\n    TEST_FILENAMES = tf.io.gfile.glob('/kaggle/input/*/test/*.tfrec')\n\nprint(f\"Training files: {len(TRAIN_FILENAMES)}\")\nprint(f\"Validation files: {len(VAL_FILENAMES)}\")\nprint(f\"Test files: {len(TEST_FILENAMES)}\")\n\nif len(TRAIN_FILENAMES) == 0:\n    print(\"ERROR: Training files not found! Check Kaggle dataset path.\")\nelse:\n    print(f\"✓ Found training data: {TRAIN_FILENAMES[0][:60]}...\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T12:42:14.273013Z","iopub.execute_input":"2025-12-03T12:42:14.273426Z","iopub.status.idle":"2025-12-03T12:42:14.362869Z","shell.execute_reply.started":"2025-12-03T12:42:14.273398Z","shell.execute_reply":"2025-12-03T12:42:14.361893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n[2] Creating Dataset Pipeline...\")\n\nBATCH_SIZE = 32\nAUTOTUNE = tf.data.AUTOTUNE\n\n# Training dataset\ntrain_dataset = tf.data.TFRecordDataset(TRAIN_FILENAMES)\ntrain_dataset = train_dataset.map(read_tfrecord, num_parallel_calls=AUTOTUNE)\ntrain_dataset = train_dataset.shuffle(1000)\ntrain_dataset = train_dataset.batch(BATCH_SIZE)\ntrain_dataset = train_dataset.prefetch(AUTOTUNE)\n\n# Validation dataset\nval_dataset = tf.data.TFRecordDataset(VAL_FILENAMES)\nval_dataset = val_dataset.map(read_tfrecord, num_parallel_calls=AUTOTUNE)\nval_dataset = val_dataset.batch(BATCH_SIZE)\nval_dataset = val_dataset.prefetch(AUTOTUNE)\n\n# Test dataset\ntest_dataset = tf.data.TFRecordDataset(TEST_FILENAMES)\ntest_dataset = test_dataset.map(read_tfrecord_test, num_parallel_calls=AUTOTUNE)\ntest_dataset = test_dataset.batch(BATCH_SIZE)\n\nprint(\"Dataset pipeline created successfully!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T12:42:15.608829Z","iopub.execute_input":"2025-12-03T12:42:15.609148Z","iopub.status.idle":"2025-12-03T12:42:15.958989Z","shell.execute_reply.started":"2025-12-03T12:42:15.609125Z","shell.execute_reply":"2025-12-03T12:42:15.958087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n[3] Building Multi-Layer Perceptron Model...\")\n\ndef build_mlp_model(input_shape=(224, 224, 3), num_classes=104):\n    model = models.Sequential([\n        # Flatten input image menjadi 1D vector\n        layers.Flatten(input_shape=input_shape),\n        \n        # Hidden layer 1: 512 neuron, ReLU activation\n        layers.Dense(512, activation='relu', name='hidden_1'),\n        layers.BatchNormalization(),\n        layers.Dropout(0.3),\n        \n        # Hidden layer 2: 256 neuron, ReLU activation\n        layers.Dense(256, activation='relu', name='hidden_2'),\n        layers.BatchNormalization(),\n        layers.Dropout(0.3),\n        \n        # Hidden layer 3: 128 neuron, ReLU activation\n        layers.Dense(128, activation='relu', name='hidden_3'),\n        layers.BatchNormalization(),\n        layers.Dropout(0.2),\n        \n        # Hidden layer 4: 64 neuron, ReLU activation\n        layers.Dense(64, activation='relu', name='hidden_4'),\n        layers.Dropout(0.2),\n        \n        # Output layer: 104 neuron (jumlah class flower), Softmax activation\n        layers.Dense(num_classes, activation='softmax', name='output')\n    ])\n    \n    return model\n\n# Build model\nmodel = build_mlp_model()\nprint(\"\\nModel Architecture:\")\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T12:42:17.132621Z","iopub.execute_input":"2025-12-03T12:42:17.132949Z","iopub.status.idle":"2025-12-03T12:42:17.934546Z","shell.execute_reply.started":"2025-12-03T12:42:17.132925Z","shell.execute_reply":"2025-12-03T12:42:17.933598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n[4] Compiling Model...\")\n\n# Hyperparameter tuning\nLEARNING_RATE = 0.001\nOPTIMIZER = keras.optimizers.Adam(learning_rate=LEARNING_RATE)\n\nmodel.compile(\n    optimizer=OPTIMIZER,\n    loss='sparse_categorical_crossentropy',\n    metrics=['accuracy']\n)\n\nprint(f\"Optimizer: Adam (learning_rate={LEARNING_RATE})\")\nprint(\"Loss function: sparse_categorical_crossentropy\")\nprint(\"Metrics: accuracy\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T12:42:18.502936Z","iopub.execute_input":"2025-12-03T12:42:18.503596Z","iopub.status.idle":"2025-12-03T12:42:18.524811Z","shell.execute_reply.started":"2025-12-03T12:42:18.503566Z","shell.execute_reply":"2025-12-03T12:42:18.523884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n[5] Training Model...\")\n\nEPOCHS = 20\n\nhistory = model.fit(\n    train_dataset,\n    validation_data=val_dataset,\n    epochs=EPOCHS,\n    verbose=1\n)\n\nprint(\"\\nTraining completed!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T12:42:20.737832Z","iopub.execute_input":"2025-12-03T12:42:20.738127Z","execution_failed":"2025-12-03T16:03:22.085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n[6] Evaluating Model on Validation Set...\")\n\n# Get predictions pada validation data\nval_predictions = []\nval_true_labels = []\n\nfor images, labels in val_dataset:\n    predictions = model.predict(images, verbose=0)\n    pred_labels = np.argmax(predictions, axis=1)\n    val_predictions.extend(pred_labels)\n    val_true_labels.extend(labels.numpy())\n\nval_predictions = np.array(val_predictions)\nval_true_labels = np.array(val_true_labels)\n\n# Hitung evaluation metrics\naccuracy = accuracy_score(val_true_labels, val_predictions)\nprecision = precision_score(val_true_labels, val_predictions, average='macro', zero_division=0)\nrecall = recall_score(val_true_labels, val_predictions, average='macro', zero_division=0)\nf1 = f1_score(val_true_labels, val_predictions, average='macro', zero_division=0)\n\nprint(\"\\n\" + \"=\" * 60)\nprint(\"EVALUATION METRICS (Validation Set)\")\nprint(\"=\" * 60)\nprint(f\"Accuracy:  {accuracy:.4f}\")\nprint(f\"Precision: {precision:.4f}\")\nprint(f\"Recall:    {recall:.4f}\")\nprint(f\"F1-Score:  {f1:.4f}\")\nprint(\"=\" * 60)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-12-03T16:03:22.085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n[7] Visualizing Training History...\")\n\nfig, axes = plt.subplots(1, 2, figsize=(14, 5))\n\n# Plot Loss\naxes[0].plot(history.history['loss'], label='Training Loss', linewidth=2)\naxes[0].plot(history.history['val_loss'], label='Validation Loss', linewidth=2)\naxes[0].set_title('Loss vs Epoch', fontsize=12, fontweight='bold')\naxes[0].set_xlabel('Epoch')\naxes[0].set_ylabel('Loss')\naxes[0].legend()\naxes[0].grid(True, alpha=0.3)\n\n# Plot Accuracy\naxes[1].plot(history.history['accuracy'], label='Training Accuracy', linewidth=2)\naxes[1].plot(history.history['val_accuracy'], label='Validation Accuracy', linewidth=2)\naxes[1].set_title('Accuracy vs Epoch', fontsize=12, fontweight='bold')\naxes[1].set_xlabel('Epoch')\naxes[1].set_ylabel('Accuracy')\naxes[1].legend()\naxes[1].grid(True, alpha=0.3)\n\nplt.tight_layout()\nplt.savefig('training_history.png', dpi=100, bbox_inches='tight')\nprint(\"Saved: training_history.png\")\nplt.show()","metadata":{"trusted":true,"execution":{"execution_failed":"2025-12-03T16:03:22.085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n[8] Convergence Analysis...\")\n\nfinal_train_loss = history.history['loss'][-1]\nfinal_val_loss = history.history['val_loss'][-1]\nfinal_train_acc = history.history['accuracy'][-1]\nfinal_val_acc = history.history['val_accuracy'][-1]\n\nprint(f\"\\nFinal Training Loss: {final_train_loss:.4f}\")\nprint(f\"Final Validation Loss: {final_val_loss:.4f}\")\nprint(f\"Final Training Accuracy: {final_train_acc:.4f}\")\nprint(f\"Final Validation Accuracy: {final_val_acc:.4f}\")\n\nif final_val_loss < final_train_loss:\n    print(\"\\n✓ Model convergence: GOOD (val_loss < train_loss)\")\nelse:\n    print(\"\\n⚠ Potential OVERFITTING detected (train_loss < val_loss)\")\n    print(\"   Consider: more dropout, data augmentation, atau regularization\")","metadata":{"trusted":true,"execution":{"execution_failed":"2025-12-03T16:03:22.086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n[8] Convergence Analysis...\")\n\nfinal_train_loss = history.history['loss'][-1]\nfinal_val_loss = history.history['val_loss'][-1]\nfinal_train_acc = history.history['accuracy'][-1]\nfinal_val_acc = history.history['val_accuracy'][-1]\n\nprint(f\"\\nFinal Training Loss: {final_train_loss:.4f}\")\nprint(f\"Final Validation Loss: {final_val_loss:.4f}\")\nprint(f\"Final Training Accuracy: {final_train_acc:.4f}\")\nprint(f\"Final Validation Accuracy: {final_val_acc:.4f}\")\n\nif final_val_loss < final_train_loss:\n    print(\"\\n✓ Model convergence: GOOD (val_loss < train_loss)\")\nelse:\n    print(\"\\n⚠ Potential OVERFITTING detected (train_loss < val_loss)\")\n    print(\"   Consider: more dropout, data augmentation, atau regularization\")","metadata":{"trusted":true,"execution":{"execution_failed":"2025-12-03T16:03:22.086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n[9] Making Predictions on Test Set...\")\n\ntest_ids = []\ntest_predictions = []\n\nfor images, ids in test_dataset:\n    predictions = model.predict(images, verbose=0)\n    pred_labels = np.argmax(predictions, axis=1)\n    test_predictions.extend(pred_labels)\n    test_ids.extend(ids.numpy())\n\ntest_ids = np.array(test_ids)\ntest_predictions = np.array(test_predictions)\n\nprint(f\"✓ Total predictions: {len(test_ids)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}