{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"We'll generate a file `submission.csv`. This file is what you'll submit to get your score on the leaderboard.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\n\nIMAGE_SIZE = [224, 224]\nBATCH_SIZE = 32\n\n# Load directly from attached Kaggle input paths\nTRAIN_FILES = tf.io.gfile.glob('/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224/train/*.tfrec')\nVAL_FILES = tf.io.gfile.glob('/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224/val/*.tfrec')\nTEST_FILES = tf.io.gfile.glob('/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224/test/*.tfrec')\n\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0\n    return tf.reshape(image, [*IMAGE_SIZE, 3])\n\ndef read_labeled(example):\n    fmt = {\"image\": tf.io.FixedLenFeature([], tf.string), \"class\": tf.io.FixedLenFeature([], tf.int64)}\n    ex = tf.io.parse_single_example(example, fmt)\n    return decode_image(ex['image']), tf.cast(ex['class'], tf.int32)\n\ndef read_unlabeled(example):\n    fmt = {\"image\": tf.io.FixedLenFeature([], tf.string), \"id\": tf.io.FixedLenFeature([], tf.string)}\n    ex = tf.io.parse_single_example(example, fmt)\n    return decode_image(ex['image']), ex['id']\n\ndef load_ds(files, labeled=True, ordered=False):\n    opts = tf.data.Options()\n    if not ordered:\n        opts.experimental_deterministic = False\n    ds = tf.data.TFRecordDataset(files, num_parallel_reads=tf.data.AUTOTUNE).with_options(opts)\n    return ds.map(read_labeled if labeled else read_unlabeled, num_parallel_calls=tf.data.AUTOTUNE)\n\n# Build & Train Model on GPU\ntrain_ds = load_ds(TRAIN_FILES + VAL_FILES).repeat().shuffle(2048).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\nbase_model = tf.keras.applications.MobileNetV2(input_shape=[*IMAGE_SIZE, 3], include_top=False, weights='imagenet')\nbase_model.trainable = False\n\nmodel = tf.keras.Sequential([\n    base_model,\n    tf.keras.layers.GlobalAveragePooling2D(),\n    tf.keras.layers.Dense(104, activation='softmax')\n])\n\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\nmodel.fit(train_ds, steps_per_epoch=(12753 + 3712) // BATCH_SIZE, epochs=3)\n\n# Generate Predictions & Submission\ntest_ds = load_ds(TEST_FILES, labeled=False, ordered=True).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\nprobs = model.predict(test_ds.map(lambda img, idnum: img))\npreds = np.argmax(probs, axis=-1)\n\nids_ds = test_ds.map(lambda img, idnum: idnum).unbatch()\nids = next(iter(ids_ds.batch(7382))).numpy().astype('U')\n\npd.DataFrame({'id': ids, 'label': preds}).to_csv('submission.csv', index=False)\nprint(\"Done! submission.csv created.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-28T13:21:03.452029Z","iopub.execute_input":"2026-08-28T13:21:03.452393Z","iopub.status.idle":"2026-08-28T13:22:31.510736Z","shell.execute_reply.started":"2026-08-28T13:21:03.452364Z","shell.execute_reply":"2026-08-28T13:22:31.509532Z"}},"outputs":[],"execution_count":null}]}