{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Cell 1 — imports & strategy\nimport tensorflow as tf\nimport numpy as np\nimport pandas as pd\nimport math, os, re\nfrom pathlib import Path\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import f1_score, classification_report, confusion_matrix\n\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n\nprint(\"TF Version:\", tf.__version__)\n\ntry:\n    resolver = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(resolver)\n    tf.tpu.experimental.initialize_tpu_system(resolver)\n    strategy = tf.distribute.TPUStrategy(resolver)\n    DEVICE = \"TPU\"\nexcept:\n    strategy = tf.distribute.MirroredStrategy()\n    DEVICE = \"GPU\"\n\nprint(\"Using:\", DEVICE)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:33:37.00206Z","iopub.execute_input":"2025-12-03T13:33:37.002723Z","iopub.status.idle":"2025-12-03T13:33:43.595246Z","shell.execute_reply.started":"2025-12-03T13:33:37.002689Z","shell.execute_reply":"2025-12-03T13:33:43.594389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nos.listdir(\"/kaggle/input\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:34:11.280385Z","iopub.execute_input":"2025-12-03T13:34:11.281354Z","iopub.status.idle":"2025-12-03T13:34:11.288344Z","shell.execute_reply.started":"2025-12-03T13:34:11.281317Z","shell.execute_reply":"2025-12-03T13:34:11.287418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 2 — dataset paths\nBASE = Path(\"/kaggle/input/petals-to-the-metal-flower-classification\")\nassert BASE.exists(), \"Dataset tidak ditemukan!\"\n\nIMAGE_SIZE = 224\nBATCH_SIZE = 32\nSEED = 42\n\nTRAIN_TFRECORD_PATTERN = str(BASE / \"tfrecords-jpeg-224x224\" / \"train\" / \"*.tfrec\")\nTEST_TFRECORD_PATTERN  = str(BASE / \"tfrecords-jpeg-224x224\" / \"test\" / \"*.tfrec\")\n\nprint(\"Train pattern:\", TRAIN_TFRECORD_PATTERN)\nprint(\"Test pattern :\", TEST_TFRECORD_PATTERN)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:33:43.596628Z","iopub.execute_input":"2025-12-03T13:33:43.597068Z","iopub.status.idle":"2025-12-03T13:33:43.92308Z","shell.execute_reply.started":"2025-12-03T13:33:43.597046Z","shell.execute_reply":"2025-12-03T13:33:43.921358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 3 — decode TFRecord\ndef decode_example(example, labeled=True):\n    feature_description = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, feature_description)\n    img = tf.image.decode_jpeg(example[\"image\"], channels=3)\n    img = tf.image.resize(img, (IMAGE_SIZE, IMAGE_SIZE))\n    img = tf.cast(img, tf.float32) / 255.0\n\n    if labeled:\n        return img, example[\"class\"]\n    else:\n        return img, example[\"id\"]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:33:43.923722Z","iopub.status.idle":"2025-12-03T13:33:43.924074Z","shell.execute_reply.started":"2025-12-03T13:33:43.923935Z","shell.execute_reply":"2025-12-03T13:33:43.923946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 4 — dataset loader\nAUTOTUNE = tf.data.AUTOTUNE\n\ndef get_tfrecord_dataset(pattern, batch=BATCH_SIZE, labeled=True, shuffle=False):\n    files = tf.data.Dataset.list_files(pattern)\n    ds = files.interleave(\n        tf.data.TFRecordDataset,\n        num_parallel_calls=AUTOTUNE,\n        cycle_length=4\n    )\n    ds = ds.map(lambda x: decode_example(x, labeled=labeled), num_parallel_calls=AUTOTUNE)\n    if shuffle:\n        ds = ds.shuffle(2048)\n    ds = ds.batch(batch).prefetch(AUTOTUNE)\n    return ds\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:33:43.925328Z","iopub.status.idle":"2025-12-03T13:33:43.925623Z","shell.execute_reply.started":"2025-12-03T13:33:43.92546Z","shell.execute_reply":"2025-12-03T13:33:43.925474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 5 — Feature extractor\nfrom tensorflow.keras.applications import MobileNetV2\nfrom tensorflow.keras.applications.mobilenet_v2 import preprocess_input\n\nwith strategy.scope():\n    backbone = MobileNetV2(\n        include_top=False,\n        input_shape=(IMAGE_SIZE, IMAGE_SIZE, 3),\n        pooling=\"avg\",\n        weights=\"imagenet\"\n    )\n    backbone.trainable = False\n\n    inp = keras.Input(shape=(IMAGE_SIZE, IMAGE_SIZE, 3))\n    x = preprocess_input(inp)\n    feats = backbone(x)\n    feature_extractor = keras.Model(inp, feats)\n\nprint(\"Feature extractor ready!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:33:43.926951Z","iopub.status.idle":"2025-12-03T13:33:43.927406Z","shell.execute_reply.started":"2025-12-03T13:33:43.92719Z","shell.execute_reply":"2025-12-03T13:33:43.92721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 6 — extract train features\ntrain_ds = get_tfrecord_dataset(TRAIN_TFRECORD_PATTERN, labeled=True)\n\nX_parts = []\ny_parts = []\nids = []\n\nfor batch_imgs, batch_labels in train_ds:\n    feats = feature_extractor.predict(batch_imgs, verbose=0)\n    X_parts.append(feats)\n    y_parts.append(batch_labels.numpy())\n\nX = np.concatenate(X_parts)\ny = np.concatenate(y_parts)\n\nnp.save(\"/kaggle/working/X.npy\", X)\nnp.save(\"/kaggle/working/y.npy\", y)\n\nprint(\"Feature extraction done:\", X.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:33:43.928852Z","iopub.status.idle":"2025-12-03T13:33:43.929103Z","shell.execute_reply.started":"2025-12-03T13:33:43.928993Z","shell.execute_reply":"2025-12-03T13:33:43.929003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 7 — load features\nX = np.load(\"/kaggle/working/X.npy\")\ny = np.load(\"/kaggle/working/y.npy\")\n\nprint(\"Loaded:\", X.shape, y.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:33:43.929815Z","iopub.status.idle":"2025-12-03T13:33:43.930133Z","shell.execute_reply.started":"2025-12-03T13:33:43.929989Z","shell.execute_reply":"2025-12-03T13:33:43.930003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 8 — MLP model\ndef build_mlp(input_dim, n_classes, hidden=[512]):\n    model = keras.Sequential()\n    model.add(layers.Input(shape=(input_dim,)))\n    for h in hidden:\n        model.add(layers.Dense(h, activation=\"relu\"))\n        model.add(layers.Dropout(0.3))\n    model.add(layers.Dense(n_classes, activation=\"softmax\"))\n    model.compile(\n        optimizer=\"adam\",\n        loss=\"sparse_categorical_crossentropy\",\n        metrics=[\"accuracy\"]\n    )\n    return model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:33:43.931275Z","iopub.status.idle":"2025-12-03T13:33:43.931548Z","shell.execute_reply.started":"2025-12-03T13:33:43.931403Z","shell.execute_reply":"2025-12-03T13:33:43.931413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 9 — CV\nskf = StratifiedKFold(n_splits=5, shuffle=True, random_state=SEED)\nn_classes = len(np.unique(y))\n\nresults = []\n\nfor fold, (train_idx, val_idx) in enumerate(skf.split(X, y)):\n    Xtr, Xv = X[train_idx], X[val_idx]\n    ytr, yv = y[train_idx], y[val_idx]\n\n    model = build_mlp(X.shape[1], n_classes, hidden=[512])\n    hist = model.fit(Xtr, ytr, validation_data=(Xv, yv), epochs=15, batch_size=32, verbose=1)\n\n    pred = np.argmax(model.predict(Xv), axis=1)\n    f1 = f1_score(yv, pred, average=\"macro\")\n\n    results.append({\"f1\": f1})\n\nprint(results)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:33:43.932776Z","iopub.status.idle":"2025-12-03T13:33:43.933023Z","shell.execute_reply.started":"2025-12-03T13:33:43.932906Z","shell.execute_reply":"2025-12-03T13:33:43.932917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 10 — final training\nbest_hidden = [512]\n\nfinal_model = build_mlp(X.shape[1], n_classes, hidden=best_hidden)\nfinal_model.fit(X, y, epochs=15, batch_size=32, verbose=1)\n\nfinal_model.save(\"/kaggle/working/final_model.keras\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:33:43.93428Z","iopub.status.idle":"2025-12-03T13:33:43.934727Z","shell.execute_reply.started":"2025-12-03T13:33:43.934603Z","shell.execute_reply":"2025-12-03T13:33:43.934617Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 11 — test inference\ntest_ds = get_tfrecord_dataset(TEST_TFRECORD_PATTERN, labeled=False)\n\nX_test_parts = []\ntest_ids = []\n\nfor batch_imgs, batch_ids in test_ds:\n    feats = feature_extractor.predict(batch_imgs, verbose=0)\n    X_test_parts.append(feats)\n    test_ids.extend(batch_ids.numpy())\n\nX_test = np.concatenate(X_test_parts)\n\npreds = np.argmax(final_model.predict(X_test, verbose=1), axis=1)\n\ndf = pd.DataFrame({\"id\": test_ids, \"class\": preds})\ndf.to_csv(\"/kaggle/working/submission.csv\", index=False)\n\ndf.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:33:43.936319Z","iopub.status.idle":"2025-12-03T13:33:43.936677Z","shell.execute_reply.started":"2025-12-03T13:33:43.936489Z","shell.execute_reply":"2025-12-03T13:33:43.936503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 12 — output\nfrom IPython.display import FileLink\nFileLink(\"/kaggle/working/submission.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T13:33:43.937793Z","iopub.status.idle":"2025-12-03T13:33:43.938054Z","shell.execute_reply.started":"2025-12-03T13:33:43.93793Z","shell.execute_reply":"2025-12-03T13:33:43.937944Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}