{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print(\"Running on TPU:\", tpu.master())\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\nexcept ValueError:\n    print(\"TPU not found. Using default strategy.\")\n    strategy = tf.distribute.get_strategy()\n\nprint(\"Number of replicas:\", strategy.num_replicas_in_sync)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T11:08:17.405978Z","iopub.execute_input":"2026-06-30T11:08:17.406358Z","iopub.status.idle":"2026-06-30T11:08:40.658243Z","shell.execute_reply.started":"2026-06-30T11:08:17.406324Z","shell.execute_reply":"2026-06-30T11:08:40.657283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nfor dirname, _, filenames in os.walk(\"/kaggle/input\"):\n    print(dirname)\n    print(filenames[:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T11:08:40.660056Z","iopub.execute_input":"2026-06-30T11:08:40.660745Z","iopub.status.idle":"2026-06-30T11:08:40.833366Z","shell.execute_reply.started":"2026-06-30T11:08:40.660712Z","shell.execute_reply":"2026-06-30T11:08:40.832529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\nimport glob\n\nBASE_PATH = \"/kaggle/input/competitions/tpu-getting-started\"\nIMAGE_SIZE = [224, 224]\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\n\nTRAINING_FILENAMES = tf.io.gfile.glob(BASE_PATH + \"/tfrecords-jpeg-224x224/train/*.tfrec\")\nVALIDATION_FILENAMES = tf.io.gfile.glob(BASE_PATH + \"/tfrecords-jpeg-224x224/val/*.tfrec\")\nTEST_FILENAMES = tf.io.gfile.glob(BASE_PATH + \"/tfrecords-jpeg-224x224/test/*.tfrec\")\n\nprint(\"Train files:\", len(TRAINING_FILENAMES))\nprint(\"Validation files:\", len(VALIDATION_FILENAMES))\nprint(\"Test files:\", len(TEST_FILENAMES))\nprint(\"Batch size:\", BATCH_SIZE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T11:12:28.908927Z","iopub.execute_input":"2026-06-30T11:12:28.909441Z","iopub.status.idle":"2026-06-30T11:12:28.935192Z","shell.execute_reply.started":"2026-06-30T11:12:28.909404Z","shell.execute_reply":"2026-06-30T11:12:28.934277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"AUTO = tf.data.AUTOTUNE\nNUM_CLASSES = 104\n\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.reshape(image, [*IMAGE_SIZE, 3])\n    image = tf.cast(image, tf.float32)/255.0\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),}\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example[\"image\"])\n    label = tf.cast(example[\"class\"], tf.int32)\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),}\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example[\"image\"])\n    image_id = example[\"id\"]\n    return image, image_id","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T11:24:25.402342Z","iopub.execute_input":"2026-06-30T11:24:25.402745Z","iopub.status.idle":"2026-06-30T11:24:25.411013Z","shell.execute_reply.started":"2026-06-30T11:24:25.402711Z","shell.execute_reply":"2026-06-30T11:24:25.410063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    \n    if not ordered:\n        ignore_order.experimental_deterministic = False\n    \n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    dataset = dataset.with_options(ignore_order)\n    \n    if labeled:\n        dataset = dataset.map(read_labeled_tfrecord, num_parallel_calls=AUTO)\n    else:\n        dataset = dataset.map(read_unlabeled_tfrecord, num_parallel_calls=AUTO)\n        \n    return dataset\n\ndef get_training_dataset():\n    dataset = load_dataset(TRAINING_FILENAMES, labeled=True)\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\ndef get_validation_dataset():\n    dataset = load_dataset(VALIDATION_FILENAMES, labeled=True, ordered=True)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\ndef get_test_dataset():\n    dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=True)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T11:24:53.328864Z","iopub.execute_input":"2026-06-30T11:24:53.330225Z","iopub.status.idle":"2026-06-30T11:24:53.337833Z","shell.execute_reply.started":"2026-06-30T11:24:53.33008Z","shell.execute_reply":"2026-06-30T11:24:53.336864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset = get_training_dataset()\n\nimages, labels = next(iter(train_dataset))\n\nplt.figure(figsize=(12, 8))\n\nfor i in range(12):\n    plt.subplot(3, 4, i + 1)\n    plt.imshow(images[i])\n    plt.title(f\"Class: {labels[i].numpy()}\")\n    plt.axis(\"off\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T11:24:56.486393Z","iopub.execute_input":"2026-06-30T11:24:56.487342Z","iopub.status.idle":"2026-06-30T11:24:58.400272Z","shell.execute_reply.started":"2026-06-30T11:24:56.487302Z","shell.execute_reply":"2026-06-30T11:24:58.399122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Build the Model\n\nwith strategy.scope():\n    base_model = tf.keras.applications.EfficientNetB0(\n        include_top=False,\n        weights=\"imagenet\",\n        input_shape=[224, 224, 3])\n\n    base_model.trainable = False\n\n    model = tf.keras.Sequential([\n        layers.Input(shape=[224, 224, 3]),\n        layers.Rescaling(255.0),\n        base_model,\n        layers.GlobalAveragePooling2D(),\n        layers.Dropout(0.3),\n        layers.Dense(NUM_CLASSES, activation=\"softmax\")])\n\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n        loss=\"sparse_categorical_crossentropy\",\n        metrics=[\"sparse_categorical_accuracy\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T11:25:27.704508Z","iopub.execute_input":"2026-06-30T11:25:27.704952Z","iopub.status.idle":"2026-06-30T11:25:28.835323Z","shell.execute_reply.started":"2026-06-30T11:25:27.70492Z","shell.execute_reply":"2026-06-30T11:25:28.834214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(train_dataset,validation_data=valid_dataset,epochs=5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T11:26:02.675383Z","iopub.execute_input":"2026-06-30T11:26:02.675744Z","iopub.status.idle":"2026-06-30T12:15:12.027224Z","shell.execute_reply.started":"2026-06-30T11:26:02.675694Z","shell.execute_reply":"2026-06-30T12:15:12.025735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history.history","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T12:17:06.690762Z","iopub.execute_input":"2026-06-30T12:17:06.691142Z","iopub.status.idle":"2026-06-30T12:17:06.699271Z","shell.execute_reply.started":"2026-06-30T12:17:06.691088Z","shell.execute_reply":"2026-06-30T12:17:06.698335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(7,5))\nplt.plot(history.history[\"sparse_categorical_accuracy\"], label=\"Train Accuracy\")\nplt.plot(history.history[\"val_sparse_categorical_accuracy\"], label=\"Validation Accuracy\")\nplt.title(\"Training and Validation Accuracy\")\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Accuracy\")\nplt.legend()\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T12:19:05.52977Z","iopub.execute_input":"2026-06-30T12:19:05.530135Z","iopub.status.idle":"2026-06-30T12:19:05.783574Z","shell.execute_reply.started":"2026-06-30T12:19:05.530087Z","shell.execute_reply":"2026-06-30T12:19:05.776905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(7,5))\nplt.plot(history.history[\"loss\"], label=\"Train Loss\")\nplt.plot(history.history[\"val_loss\"], label=\"Validation Loss\")\nplt.title(\"Training and Validation Loss\")\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Loss\")\nplt.legend()\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T12:19:18.500755Z","iopub.execute_input":"2026-06-30T12:19:18.501305Z","iopub.status.idle":"2026-06-30T12:19:18.686812Z","shell.execute_reply.started":"2026-06-30T12:19:18.501271Z","shell.execute_reply":"2026-06-30T12:19:18.68573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_dataset = get_test_dataset()\n\ntest_images_ds = test_dataset.map(lambda image, image_id: image)\ntest_ids_ds = test_dataset.map(lambda image, image_id: image_id)\n\nprobabilities = model.predict(test_images_ds)\npredictions = np.argmax(probabilities, axis=-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T12:19:31.463527Z","iopub.execute_input":"2026-06-30T12:19:31.464368Z","iopub.status.idle":"2026-06-30T12:23:55.905574Z","shell.execute_reply.started":"2026-06-30T12:19:31.464318Z","shell.execute_reply":"2026-06-30T12:23:55.904685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_ids = []\n\nfor batch in test_ids_ds:\n    test_ids.extend(batch.numpy())\n\ntest_ids = [x.decode(\"utf-8\") for x in test_ids]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T12:24:04.548183Z","iopub.execute_input":"2026-06-30T12:24:04.548532Z","iopub.status.idle":"2026-06-30T12:24:07.446802Z","shell.execute_reply.started":"2026-06-30T12:24:04.5485Z","shell.execute_reply":"2026-06-30T12:24:07.445887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame({\n    \"id\": test_ids,\n    \"label\": predictions})\n\nsubmission.to_csv(\"submission.csv\", index=False)\n\nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T12:24:12.13152Z","iopub.execute_input":"2026-06-30T12:24:12.131995Z","iopub.status.idle":"2026-06-30T12:24:12.200954Z","shell.execute_reply.started":"2026-06-30T12:24:12.131964Z","shell.execute_reply":"2026-06-30T12:24:12.200159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T12:24:16.132549Z","iopub.execute_input":"2026-06-30T12:24:16.132882Z","iopub.status.idle":"2026-06-30T12:24:16.139274Z","shell.execute_reply.started":"2026-06-30T12:24:16.132853Z","shell.execute_reply":"2026-06-30T12:24:16.138136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)\n\nimport os\nprint(os.listdir(\"/kaggle/working\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-30T12:36:41.693982Z","iopub.execute_input":"2026-06-30T12:36:41.695317Z","iopub.status.idle":"2026-06-30T12:36:41.726381Z","shell.execute_reply.started":"2026-06-30T12:36:41.695263Z","shell.execute_reply":"2026-06-30T12:36:41.724858Z"}},"outputs":[],"execution_count":null}]}