{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":21154,"databundleVersionId":1243559,"isSourceIdPinned":false},{"sourceType":"kernelVersion","sourceId":37130068,"isSourceIdPinned":false}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n# IMPORT #\n\n","metadata":{}},{"cell_type":"code","source":"import math, re, os\nimport numpy as np\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\n\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.metrics import f1_score, precision_score, recall_score, confusion_matrix\n\nprint(\"Tensorflow version \" + tf.__version__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-26T11:17:07.064828Z","iopub.execute_input":"2026-02-26T11:17:07.06522Z","iopub.status.idle":"2026-02-26T11:17:23.451661Z","shell.execute_reply.started":"2026-02-26T11:17:07.065183Z","shell.execute_reply":"2026-02-26T11:17:23.450544Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# TPU Strategy #","metadata":{}},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS:\", strategy.num_replicas_in_sync)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-26T11:18:37.227075Z","iopub.execute_input":"2026-02-26T11:18:37.227537Z","iopub.status.idle":"2026-02-26T11:18:37.237615Z","shell.execute_reply.started":"2026-02-26T11:18:37.227438Z","shell.execute_reply":"2026-02-26T11:18:37.236422Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Dataset #","metadata":{}},{"cell_type":"code","source":"GCS_DS_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started')\n\nIMAGE_SIZE = [512, 512]\nAUTO = tf.data.AUTOTUNE\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\n\nGCS_PATH = GCS_DS_PATH + '/tfrecords-jpeg-512x512'\n\nTRAINING_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/train/*.tfrec')\nVALIDATION_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/val/*.tfrec')\nTEST_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/test/*.tfrec')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-26T11:18:55.754296Z","iopub.execute_input":"2026-02-26T11:18:55.755642Z","iopub.status.idle":"2026-02-26T11:18:56.320304Z","shell.execute_reply.started":"2026-02-26T11:18:55.755605Z","shell.execute_reply":"2026-02-26T11:18:56.319317Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Decode + Read TFRecord #","metadata":{}},{"cell_type":"code","source":"def decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0\n    image = tf.reshape(image, [*IMAGE_SIZE, 3])\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64)\n    }\n    example = tf.io.parse_single_example(example, LABELED)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string)\n    }\n    example = tf.io.parse_single_example(example, UNLABELED)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-26T11:19:10.478605Z","iopub.execute_input":"2026-02-26T11:19:10.47905Z","iopub.status.idle":"2026-02-26T11:19:10.487635Z","shell.execute_reply.started":"2026-02-26T11:19:10.479019Z","shell.execute_reply":"2026-02-26T11:19:10.486245Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Pipeline #","metadata":{}},{"cell_type":"code","source":"def load_dataset(filenames, labeled=True, ordered=False):\n    options = tf.data.Options()\n    if not ordered:\n        options.experimental_deterministic = False\n\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    dataset = dataset.with_options(options)\n\n    dataset = dataset.map(\n        read_labeled_tfrecord if labeled else read_unlabeled_tfrecord,\n        num_parallel_calls=AUTO\n    )\n    return dataset\n\ndef data_augment(image, label):\n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_flip_up_down(image)\n    image = tf.image.random_brightness(image, 0.2)\n    image = tf.image.random_contrast(image, 0.8, 1.2)\n    return image, label\n\ndef get_training_dataset():\n    dataset = load_dataset(TRAINING_FILENAMES, labeled=True)\n    dataset = dataset.map(data_augment, num_parallel_calls=AUTO)\n    dataset = dataset.repeat()\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\ndef get_validation_dataset():\n    dataset = load_dataset(VALIDATION_FILENAMES, labeled=True, ordered=True)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\ndef get_test_dataset():\n    dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=True)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-26T11:19:28.943473Z","iopub.execute_input":"2026-02-26T11:19:28.944438Z","iopub.status.idle":"2026-02-26T11:19:28.955074Z","shell.execute_reply.started":"2026-02-26T11:19:28.944399Z","shell.execute_reply":"2026-02-26T11:19:28.953508Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dataset Setup #","metadata":{}},{"cell_type":"code","source":"ds_train = get_training_dataset()\nds_valid = get_validation_dataset()\nds_test = get_test_dataset()\n\nprint(\"Datasets Ready\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-26T11:19:49.634643Z","iopub.execute_input":"2026-02-26T11:19:49.635709Z","iopub.status.idle":"2026-02-26T11:19:50.258258Z","shell.execute_reply.started":"2026-02-26T11:19:49.63567Z","shell.execute_reply":"2026-02-26T11:19:50.257141Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model (EfficientNet Upgrade) #","metadata":{}},{"cell_type":"code","source":"EPOCHS = 15\n\nwith strategy.scope():\n\n    base_model = tf.keras.applications.EfficientNetB4(\n        input_shape=[*IMAGE_SIZE, 3],\n        weights='imagenet',\n        include_top=False\n    )\n\n    base_model.trainable = False\n\n    x = base_model.output\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Dropout(0.3)(x)\n\n    output = tf.keras.layers.Dense(104, activation='softmax')(x)\n\n    model = tf.keras.Model(inputs=base_model.input, outputs=output)\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4),\n    loss='sparse_categorical_crossentropy',\n    metrics=['sparse_categorical_accuracy']\n)\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-26T11:20:06.04327Z","iopub.execute_input":"2026-02-26T11:20:06.044148Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training Callbacks #","metadata":{}},{"cell_type":"code","source":"early_stop = tf.keras.callbacks.EarlyStopping(\n    monitor='val_loss',\n    patience=3,\n    restore_best_weights=True\n)\n\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor='val_loss',\n    factor=0.3,\n    patience=2,\n    min_lr=1e-6\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training #","metadata":{}},{"cell_type":"code","source":"history = model.fit(\n    ds_train,\n    validation_data=ds_valid,\n    epochs=EPOCHS,\n    steps_per_epoch=200,\n    callbacks=[early_stop, reduce_lr]\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Fine Tuning (Score Booster) #","metadata":{}},{"cell_type":"code","source":"with strategy.scope():\n    base_model.trainable = True\n\n    for layer in base_model.layers[:-30]:\n        layer.trainable = False\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-5),\n    loss='sparse_categorical_crossentropy',\n    metrics=['sparse_categorical_accuracy']\n)\n\nmodel.fit(\n    ds_train,\n    validation_data=ds_valid,\n    epochs=5,\n    steps_per_epoch=200\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Prediction #","metadata":{}},{"cell_type":"code","source":"print(\"Predicting...\")\n\ntest_images_ds = ds_test.map(lambda image, idnum: image)\nprobabilities = model.predict(test_images_ds)\npredictions = np.argmax(probabilities, axis=-1)\n\ntest_ids_ds = ds_test.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(7382))).numpy().astype('U')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission File #","metadata":{}},{"cell_type":"code","source":"np.savetxt(\n    'submission.csv',\n    np.rec.fromarrays([test_ids, predictions]),\n    fmt=['%s', '%d'],\n    header='id,label',\n    comments=''\n)\n\nprint(\"Submission Ready\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}