{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31194,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T06:45:25.708874Z","iopub.execute_input":"2025-12-03T06:45:25.709624Z","iopub.status.idle":"2025-12-03T06:45:25.819645Z","shell.execute_reply.started":"2025-12-03T06:45:25.709602Z","shell.execute_reply":"2025-12-03T06:45:25.819067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# BAGIAN 1: SETUP & KONFIGURASI\n# Deskripsi: Menyiapkan library dan mendeteksi hardware (GPU).\n\nimport math, re, os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import classification_report\n\nprint(f\"TensorFlow version: {tf.__version__}\")\n\ntry:\n    \n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() \n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\nelse:\n    \n    strategy = tf.distribute.MirroredStrategy() \n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)\n\n# Konfigurasi Dataset\nIMAGE_SIZE = [128, 128] # Kita set 128 agar ringan untuk MLP\nBATCH_SIZE = 64 * strategy.num_replicas_in_sync\nEPOCHS = 25\nAUTO = tf.data.AUTOTUNE\nCLASSES = 104 \n\n# Path Data (Lokal)\nPATH_BASE = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-192x192'\nTRAINING_FILENAMES = tf.io.gfile.glob(PATH_BASE + '/train/*.tfrec')\nVALIDATION_FILENAMES = tf.io.gfile.glob(PATH_BASE + '/val/*.tfrec')\nTEST_FILENAMES = tf.io.gfile.glob(PATH_BASE + '/test/*.tfrec')\n\nprint(\"Setup Selesai.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T06:45:51.058266Z","iopub.execute_input":"2025-12-03T06:45:51.058813Z","iopub.status.idle":"2025-12-03T06:45:51.100032Z","shell.execute_reply.started":"2025-12-03T06:45:51.058788Z","shell.execute_reply":"2025-12-03T06:45:51.099433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# BAGIAN 2: DATA LOADING & PREPROCESSING\n\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    # Normalisasi (0-1)\n    image = tf.cast(image, tf.float32) / 255.0  \n    image = tf.image.resize(image, IMAGE_SIZE)\n    \n    # Flattening (Mengubah kotak jadi garis lurus untuk MLP)\n    image = tf.reshape(image, [-1]) \n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = example['class']\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, num_parallel_calls=AUTO)\n    return dataset\n\n# Membuat Dataset\nprint(\"Memuat dataset...\")\ntrain_dataset = load_dataset(TRAINING_FILENAMES, labeled=True).repeat().shuffle(2048).batch(BATCH_SIZE).prefetch(AUTO)\nvalid_dataset = load_dataset(VALIDATION_FILENAMES, labeled=True, ordered=True).batch(BATCH_SIZE).prefetch(AUTO)\ntest_dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=True).batch(BATCH_SIZE).prefetch(AUTO)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T06:46:04.703104Z","iopub.execute_input":"2025-12-03T06:46:04.703846Z","iopub.status.idle":"2025-12-03T06:46:04.872795Z","shell.execute_reply.started":"2025-12-03T06:46:04.703819Z","shell.execute_reply":"2025-12-03T06:46:04.872166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# BAGIAN 3: MEMBANGUN MODEL MLP & TRAINING\n\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization\n\n# Masuk ke scope strategy agar support GPU/TPU\nwith strategy.scope():\n    INPUT_DIM = IMAGE_SIZE[0] * IMAGE_SIZE[1] * 3\n    \n    # Arsitektur MLP Lengkap\n    model = Sequential([\n        tf.keras.layers.InputLayer(input_shape=(INPUT_DIM,)),\n        \n        # Hidden Layer 1 (ReLU)\n        Dense(512, activation='relu'),\n        BatchNormalization(),\n        Dropout(0.3),\n        \n        # Hidden Layer 2 (ReLU)\n        Dense(256, activation='relu'),\n        BatchNormalization(),\n        Dropout(0.3),\n        \n        # Hidden Layer 3\n        Dense(128, activation='relu'),\n        \n        # Output Layer (Softmax)\n        Dense(CLASSES, activation='softmax')\n    ])\n    \n    model.compile(\n        optimizer='adam',\n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\nmodel.summary()\n\n# Training\nprint(\"Mulai Training...\")\nNUM_TRAINING_IMAGES = 12753\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\n\n# Early Stopping\nearly_stop = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\n\nhistory = model.fit(\n    train_dataset,\n    steps_per_epoch=STEPS_PER_EPOCH,\n    epochs=EPOCHS,\n    validation_data=valid_dataset,\n    callbacks=[early_stop]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T06:46:36.632489Z","iopub.execute_input":"2025-12-03T06:46:36.632808Z","iopub.status.idle":"2025-12-03T06:48:10.157598Z","shell.execute_reply.started":"2025-12-03T06:46:36.632784Z","shell.execute_reply":"2025-12-03T06:48:10.156946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# BAGIAN 4: EVALUASI & SUBMISSION\n\n\n# 1. Grafik (Analisis Konvergensi)\nplt.figure(figsize=(12, 5))\nplt.subplot(1, 2, 1)\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Val Loss')\nplt.title('Loss')\nplt.legend()\nplt.subplot(1, 2, 2)\nplt.plot(history.history['sparse_categorical_accuracy'], label='Train Acc')\nplt.plot(history.history['val_sparse_categorical_accuracy'], label='Val Acc')\nplt.title('Accuracy')\nplt.legend()\nplt.show()\n\n# 2. Classification Report (Metrik Lengkap)\nprint(\"Menghitung Metrik Lengkap...\")\n\ny_true = []\nimages_val = []\nfor img, label in valid_dataset:\n    y_true.extend(label.numpy())\n    images_val.append(img)\n    \ny_true = np.array(y_true)\n# Prediksi data validasi\n# Catatan: Ambil sebagian saja untuk report agar memori aman\nimages_val_arr = np.concatenate(images_val[:len(y_true)//BATCH_SIZE+1]) # Trik hemat memori\nprobs_val = model.predict(images_val_arr[:len(y_true)], verbose=0)\ny_pred = np.argmax(probs_val, axis=-1)\n\nprint(classification_report(y_true, y_pred, zero_division=0))\n\n# 3. Submission \nprint(\"Membuat Submission...\")\ntest_ds = load_dataset(TEST_FILENAMES, labeled=False, ordered=True).batch(BATCH_SIZE).prefetch(AUTO)\n\ntest_images = []\ntest_ids = []\n\nfor image, idnum in test_ds:\n    test_images.append(image.numpy())\n    test_ids.append(idnum.numpy())\n\ntest_images_arr = np.concatenate(test_images)\ntest_ids_arr = np.concatenate(test_ids)\n\nprint(f\"Prediksi {len(test_ids_arr)} data test...\")\nprobs = model.predict(test_images_arr, verbose=1)\npreds = np.argmax(probs, axis=-1)\n\n# Bersihkan ID dari format b'...'\ntest_ids_str = [x.decode('utf-8') if isinstance(x, bytes) else str(x) for x in test_ids_arr]\n\nsubmission = pd.DataFrame({'id': test_ids_str, 'label': preds})\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"SUKSES! Silakan download submission.csv\")\nprint(submission.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T06:49:16.762989Z","iopub.execute_input":"2025-12-03T06:49:16.763534Z","iopub.status.idle":"2025-12-03T06:49:31.192672Z","shell.execute_reply.started":"2025-12-03T06:49:16.763508Z","shell.execute_reply":"2025-12-03T06:49:31.192084Z"}},"outputs":[],"execution_count":null}]}