{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow as tf\nfrom kaggle_datasets import KaggleDatasets\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\n# --- 1. SETUP RINGAN (Supaya Tidak Error Memori) ---\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\n    BATCH_SIZE = 16 * strategy.num_replicas_in_sync\nexcept ValueError:\n    strategy = tf.distribute.get_strategy()\n    BATCH_SIZE = 16 \n    print(\"Menggunakan GPU Strategy\")\n\n# PENTING: Pakai ukuran 192 agar muat di memori\nIMAGE_SIZE = [192, 192] \nEPOCHS = 10 \nGCS_DS_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started')\n\n# --- 2. DATASET PIPELINE ---\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0\n    image = tf.image.resize(image, IMAGE_SIZE) # Resize ke 192x192\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label \n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False \n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=tf.data.experimental.AUTOTUNE)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, \n                          num_parallel_calls=tf.data.experimental.AUTOTUNE)\n    return dataset\n\n# Ambil Data (Arahkan ke folder 192x192)\nfilenames_train = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/train/*.tfrec')\nfilenames_val = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/val/*.tfrec')\nfilenames_test = tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/test/*.tfrec')\n\ntraining_dataset = load_dataset(filenames_train, labeled=True).repeat().shuffle(2048).batch(BATCH_SIZE).prefetch(tf.data.experimental.AUTOTUNE)\nvalidation_dataset = load_dataset(filenames_val, labeled=True, ordered=False).batch(BATCH_SIZE).cache().prefetch(tf.data.experimental.AUTOTUNE)\n\n# --- 3. MODEL MLP VERSI HEMAT (512 Neuron) ---\nwith strategy.scope():    \n    model = tf.keras.Sequential([\n        tf.keras.layers.Flatten(input_shape=(*IMAGE_SIZE, 3)),\n        \n        # INI PERUBAHANNYA: Pakai 512 (Bukan 2048)\n        tf.keras.layers.Dense(512, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.3),\n        \n        # Layer 2\n        tf.keras.layers.Dense(256, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.3),\n        \n        # Output Layer\n        tf.keras.layers.Dense(104, activation='softmax')\n    ])\n    \n    model.compile(\n        optimizer='adam',\n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\n# --- 4. TRAINING & PREDIKSI ---\nprint(\"=== MULAI TRAINING (VERSI RINGAN) ===\")\nNUM_TRAINING_IMAGES = 12753\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\n\nhistory = model.fit(\n    training_dataset, \n    steps_per_epoch=STEPS_PER_EPOCH,\n    epochs=EPOCHS, \n    validation_data=validation_dataset\n)\n\nprint(\"=== MULAI PREDIKSI ===\")\ntest_ds = load_dataset(filenames_test, labeled=False, ordered=True)\ntest_ds = test_ds.batch(BATCH_SIZE)\ntest_images_ds = test_ds.map(lambda image, idnum: image)\n\nprobabilities = model.predict(test_images_ds)\npredictions = np.argmax(probabilities, axis=-1)\n\nprint(\"=== SIMPAN FILE ===\")\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(len(predictions)))).numpy().astype('U')\n\nsubmission = pd.DataFrame({'id': test_ids, 'label': predictions})\nsubmission.to_csv('submission.csv', index=False)\nprint('SUKSES! Tugas Selesai. Silakan Save Version.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T09:40:31.066895Z","iopub.execute_input":"2025-12-03T09:40:31.067538Z","iopub.status.idle":"2025-12-03T09:42:50.906282Z","shell.execute_reply.started":"2025-12-03T09:40:31.067511Z","shell.execute_reply":"2025-12-03T09:42:50.905612Z"}},"outputs":[],"execution_count":null}]}