{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-25T02:29:59.620022Z","iopub.execute_input":"2025-11-25T02:29:59.62021Z","iopub.status.idle":"2025-11-25T02:30:01.519901Z","shell.execute_reply.started":"2025-11-25T02:29:59.620191Z","shell.execute_reply":"2025-11-25T02:30:01.518839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#import necessary libraries\nimport tensorflow as tf\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras import layers, models\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T02:33:33.755857Z","iopub.execute_input":"2025-11-25T02:33:33.756177Z","iopub.status.idle":"2025-11-25T02:33:33.760843Z","shell.execute_reply.started":"2025-11-25T02:33:33.756153Z","shell.execute_reply":"2025-11-25T02:33:33.759931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# path to the dataset\ntpu_getting_started_path=\"/kaggle/input/tpu-getting-started\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T02:38:14.737515Z","iopub.execute_input":"2025-11-25T02:38:14.737911Z","iopub.status.idle":"2025-11-25T02:38:14.742402Z","shell.execute_reply.started":"2025-11-25T02:38:14.737887Z","shell.execute_reply":"2025-11-25T02:38:14.741579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#path to 192x192 images \ntpu_192x192_path = tpu_getting_started_path + \"/tfrecords-jpeg-192x192\"\n#path to 224x224 images\ntpu_224x224_path = tpu_getting_started_path + \"/tfrecords-jpeg-224x224\"\n#path  to 331x331 images\ntpu_331x331_path = tpu_getting_started_path + \"/tfrecords-jpeg-331x331\"\n#path to 512x512 images\ntpu_512x512_path = tpu_getting_started_path + \"/tfrecords-jpeg-512x512\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T02:38:26.219247Z","iopub.execute_input":"2025-11-25T02:38:26.219504Z","iopub.status.idle":"2025-11-25T02:38:26.224213Z","shell.execute_reply.started":"2025-11-25T02:38:26.219487Z","shell.execute_reply":"2025-11-25T02:38:26.223016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#load sample 192x192 image .tfrecord file from train\nfilenames = tf.io.gfile.glob(tpu_192x192_path + \"/train/*.tfrec\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T02:38:36.619231Z","iopub.execute_input":"2025-11-25T02:38:36.62007Z","iopub.status.idle":"2025-11-25T02:38:36.629981Z","shell.execute_reply.started":"2025-11-25T02:38:36.620042Z","shell.execute_reply":"2025-11-25T02:38:36.629204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#function to read tfrecord files\nIMAGE_SIZE = [192, 192] \ndef read_tfrecord(example):\n    tfrec_format = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, tfrec_format)\n\n    # Decode JPEG/PNG\n    image = tf.image.decode_jpeg(example[\"image\"], channels=3)\n    image = tf.image.resize(image, IMAGE_SIZE)\n    image = tf.cast(image, tf.float32) / 255.0  # normalize\n\n    label = example[\"class\"]\n    return image, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T02:38:47.362829Z","iopub.execute_input":"2025-11-25T02:38:47.363147Z","iopub.status.idle":"2025-11-25T02:38:47.368286Z","shell.execute_reply.started":"2025-11-25T02:38:47.363127Z","shell.execute_reply":"2025-11-25T02:38:47.367515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#create dataset from tfrecord files\ndataset = tf.data.TFRecordDataset(filenames)\ndataset = dataset.map(read_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T02:38:59.015628Z","iopub.execute_input":"2025-11-25T02:38:59.015955Z","iopub.status.idle":"2025-11-25T02:38:59.045098Z","shell.execute_reply.started":"2025-11-25T02:38:59.015932Z","shell.execute_reply":"2025-11-25T02:38:59.044405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#display a single image from the dataset\nimage, label = next(iter(dataset))\n\nplt.imshow(image.numpy())\nplt.title(f\"Label: {label.numpy()}\")\nplt.axis(\"off\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T02:39:07.668192Z","iopub.execute_input":"2025-11-25T02:39:07.669187Z","iopub.status.idle":"2025-11-25T02:39:07.974314Z","shell.execute_reply.started":"2025-11-25T02:39:07.669161Z","shell.execute_reply":"2025-11-25T02:39:07.973438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#display all images\nplt.figure(figsize=(10, 10))\nfor i, (image, label) in enumerate(dataset.take(15)):\n    plt.subplot(3, 5, i + 1)\n    plt.imshow(image.numpy())\n    plt.title(f\"Label: {label.numpy()}\")\n    plt.axis(\"off\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T02:39:22.057616Z","iopub.execute_input":"2025-11-25T02:39:22.058491Z","iopub.status.idle":"2025-11-25T02:39:22.947794Z","shell.execute_reply.started":"2025-11-25T02:39:22.058465Z","shell.execute_reply":"2025-11-25T02:39:22.946774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#check which images have same label\nlabel_to_images = {}\nfor image, label in dataset:\n    label = label.numpy()\n    if label not in label_to_images:\n        label_to_images[label] = []\n    label_to_images[label].append(image.numpy())\nfor label, images in label_to_images.items():\n    if len(images) > 1:\n        print(f\"Label {label} has {len(images)} images.\")\n        plt.figure(figsize=(10, 5))\n        for i, img in enumerate(images):\n            plt.subplot(1, len(images), i + 1)\n            plt.imshow(img)\n            plt.title(f\"Label: {label}\")\n            plt.axis(\"off\")\n        plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T02:39:53.513282Z","iopub.execute_input":"2025-11-25T02:39:53.513619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#create a table for label distribution\nlabel_counts = {}\nfor image, label in dataset:\n    label = label.numpy()\n    if label not in label_counts:\n        label_counts[label] = 0\n    label_counts[label] += 1\nprint(\"Label Distribution:\")\nfor label, count in label_counts.items():\n    print(f\"Label {label}: {count} samples\")\n#visualize label distribution\nlabels = list(label_counts.keys())\ncounts = list(label_counts.values())\nplt.bar(labels, counts)\nplt.xlabel(\"Labels\")\nplt.ylabel(\"Number of Samples\")\nplt.title(\"Label Distribution\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T02:42:42.643014Z","iopub.execute_input":"2025-11-25T02:42:42.643287Z","iopub.status.idle":"2025-11-25T02:42:48.019621Z","shell.execute_reply.started":"2025-11-25T02:42:42.643268Z","shell.execute_reply":"2025-11-25T02:42:48.018898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#label distribution for 224x224 images\nfilenames_224 = tf.io.gfile.glob(tpu_224x224_path + \"/train/*.tfrec\")\ndataset_224 = tf.data.TFRecordDataset(filenames_224)\ndataset_224 = dataset_224.map(read_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)\nlabel_counts_224 = {}\nfor image, label in dataset_224:\n    label = label.numpy()\n    if label not in label_counts_224:\n        label_counts_224[label] = 0\n    label_counts_224[label] += 1\nprint(\"Label Distribution for 224x224 images:\")\nfor label, count in label_counts_224.items():\n    print(f\"Label {label}: {count} samples\")\n#visualize label distribution for 224x224 images\nlabels_224 = list(label_counts_224.keys())\ncounts_224 = list(label_counts_224.values())\nplt.bar(labels_224, counts_224)\nplt.xlabel(\"Labels\")\nplt.ylabel(\"Number of Samples\")\nplt.title(\"Label Distribution for 224x224 images\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T02:43:07.003986Z","iopub.execute_input":"2025-11-25T02:43:07.004277Z","iopub.status.idle":"2025-11-25T02:43:16.32914Z","shell.execute_reply.started":"2025-11-25T02:43:07.004256Z","shell.execute_reply":"2025-11-25T02:43:16.328226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#lable distribution for 331x331 images\nfilenames_331 = tf.io.gfile.glob(tpu_331x331_path + \"/train/*.tfrec\")\ndataset_331 = tf.data.TFRecordDataset(filenames_331)\ndataset_331 = dataset_331.map(read_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)\nlabel_counts_331 = {}\nfor image, label in dataset_331:\n    label = label.numpy()\n    if label not in label_counts_331:\n        label_counts_331[label] = 0\n    label_counts_331[label] += 1\nprint(\"Label Distribution for 331x331 images:\")\nfor label, count in label_counts_331.items():\n    print(f\"Label {label}: {count} samples\")\n#visualize label distribution for 331x331 images\nlabels_331 = list(label_counts_331.keys())\ncounts_331 = list(label_counts_331.values())\nplt.bar(labels_331, counts_331)\nplt.xlabel(\"Labels\")\nplt.ylabel(\"Number of Samples\")\nplt.title(\"Label Distribution for 331x331 images\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T02:43:34.353309Z","iopub.execute_input":"2025-11-25T02:43:34.353675Z","iopub.status.idle":"2025-11-25T02:43:47.330088Z","shell.execute_reply.started":"2025-11-25T02:43:34.353652Z","shell.execute_reply":"2025-11-25T02:43:47.32927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#label distribution for 512x512 images\nfilenames_512 = tf.io.gfile.glob(tpu_512x512_path + \"/train/*.tfrec\")\ndataset_512 = tf.data.TFRecordDataset(filenames_512)\ndataset_512 = dataset_512.map(read_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)\nlabel_counts_512 = {}\nfor image, label in dataset_512:\n    label = label.numpy()\n    if label not in label_counts_512:\n        label_counts_512[label] = 0\n    label_counts_512[label] += 1\nprint(\"Label Distribution for 512x512 images:\")\nfor label, count in label_counts_512.items():\n    print(f\"Label {label}: {count} samples\")\n#visualize label distribution for 512x512 images\nlabels_512 = list(label_counts_512.keys())\ncounts_512 = list(label_counts_512.values())\nplt.bar(labels_512, counts_512)\nplt.xlabel(\"Labels\")\nplt.ylabel(\"Number of Samples\")\nplt.title(\"Label Distribution for 512x512 images\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T02:44:13.348374Z","iopub.execute_input":"2025-11-25T02:44:13.348744Z","iopub.status.idle":"2025-11-25T02:44:32.775521Z","shell.execute_reply.started":"2025-11-25T02:44:13.34869Z","shell.execute_reply":"2025-11-25T02:44:32.774775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#compare label distributions across different image sizes\nimport numpy as np\nlabels = sorted(set(label_counts.keys()) | set(label_counts_224.keys()) | set(label_counts_331.keys()) | set(label_counts_512.keys()))\ncounts_192 = [label_counts.get(label, 0) for label in labels]\ncounts_224 = [label_counts_224.get(label, 0) for label in labels]\ncounts_331 = [label_counts_331.get(label, 0) for label in labels]\ncounts_512 = [label_counts_512.get(label, 0) for label in labels]\nx = np.arange(len(labels))\nwidth = 0.2\nplt.bar(x - 1.5*width, counts_192, width, label='192x192')\nplt.bar(x - 0.5*width, counts_224, width, label='224x224')\nplt.bar(x + 0.5*width, counts_331, width, label='331x331')\nplt.bar(x + 1.5*width, counts_512, width, label='512x512')\nplt.xlabel(\"Labels\")\nplt.ylabel(\"Number of Samples\")\nplt.title(\"Label Distribution Across Different Image Sizes\")\nplt.xticks(x, labels)\nplt.legend()\nplt.show()\n#identify any missing labels in different image sizes\nfor label in labels:\n    missing_sizes = []\n    if label not in label_counts:\n        missing_sizes.append(\"192x192\")\n    if label not in label_counts_224:\n        missing_sizes.append(\"224x224\")\n    if label not in label_counts_331:\n        missing_sizes.append(\"331x331\")\n    if label not in label_counts_512:\n        missing_sizes.append(\"512x512\")\n    if missing_sizes:\n        print(f\"Label {label} is missing in sizes: {', '.join(missing_sizes)}\")\n#analyze image quality across different sizes\ndef analyze_image_quality(dataset):\n    brightness_values = []\n    contrast_values = []\n    for image, label in dataset:\n        image_np = image.numpy()\n        brightness = np.mean(image_np)\n        contrast = np.std(image_np)\n        brightness_values.append(brightness)\n        contrast_values.append(contrast)\n    return np.mean(brightness_values), np.mean(contrast_values)\nbrightness_192, contrast_192 = analyze_image_quality(dataset)\nbrightness_224, contrast_224 = analyze_image_quality(dataset_224)\nbrightness_331, contrast_331 = analyze_image_quality(dataset_331)\nbrightness_512, contrast_512 = analyze_image_quality(dataset_512)\n#visualize brightness and contrast across different sizes\nsizes = ['192x192', '224x224', '331x331', '512x512']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T02:45:19.199141Z","iopub.execute_input":"2025-11-25T02:45:19.199409Z","iopub.status.idle":"2025-11-25T02:46:12.575651Z","shell.execute_reply.started":"2025-11-25T02:45:19.199391Z","shell.execute_reply":"2025-11-25T02:46:12.574673Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"All Datasets seem to have identical data, only differene is the size, but we will train on all sizes, since the validation submission requires it","metadata":{}},{"cell_type":"code","source":"label_counts.values()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T02:49:07.35994Z","iopub.execute_input":"2025-11-25T02:49:07.360755Z","iopub.status.idle":"2025-11-25T02:49:07.366089Z","shell.execute_reply.started":"2025-11-25T02:49:07.360694Z","shell.execute_reply":"2025-11-25T02:49:07.365456Z"}},"outputs":[],"execution_count":null}]}