{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-23T18:57:17.573714Z","iopub.execute_input":"2025-11-23T18:57:17.57406Z","iopub.status.idle":"2025-11-23T18:57:19.657351Z","shell.execute_reply.started":"2025-11-23T18:57:17.574032Z","shell.execute_reply":"2025-11-23T18:57:19.656157Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#import necessary libraries\nimport tensorflow as tf\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras import layers, models\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-23T18:58:44.741417Z","iopub.execute_input":"2025-11-23T18:58:44.742121Z","iopub.status.idle":"2025-11-23T18:58:44.747436Z","shell.execute_reply.started":"2025-11-23T18:58:44.742092Z","shell.execute_reply":"2025-11-23T18:58:44.746054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tpu_getting_started_path=\"/kaggle/input/tpu-getting-started\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-23T18:58:46.764746Z","iopub.execute_input":"2025-11-23T18:58:46.765072Z","iopub.status.idle":"2025-11-23T18:58:46.770079Z","shell.execute_reply.started":"2025-11-23T18:58:46.765051Z","shell.execute_reply":"2025-11-23T18:58:46.76914Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#path to 192x192 images \ntpu_192x192_path = tpu_getting_started_path + \"/tfrecords-jpeg-192x192\"\n#path to 224x224 images\ntpu_224x224_path = tpu_getting_started_path + \"/tfrecords-jpeg-224x224\"\n#path  to 331x331 images\ntpu_331x331_path = tpu_getting_started_path + \"/tfrecords-jpeg-331x331\"\n#path to 512x512 images\ntpu_512x512_path = tpu_getting_started_path + \"/tfrecords-jpeg-512x512\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-23T18:58:46.929842Z","iopub.execute_input":"2025-11-23T18:58:46.930164Z","iopub.status.idle":"2025-11-23T18:58:46.935831Z","shell.execute_reply.started":"2025-11-23T18:58:46.930144Z","shell.execute_reply":"2025-11-23T18:58:46.934593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#load sample 192x192 image .tfrecord file from train\nfilenames = tf.io.gfile.glob(tpu_192x192_path + \"/train/*.tfrec\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-23T18:58:48.508435Z","iopub.execute_input":"2025-11-23T18:58:48.509479Z","iopub.status.idle":"2025-11-23T18:58:48.52431Z","shell.execute_reply.started":"2025-11-23T18:58:48.509448Z","shell.execute_reply":"2025-11-23T18:58:48.5231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#function to read tfrecord files\nIMAGE_SIZE = [192, 192] \ndef read_tfrecord(example):\n    tfrec_format = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, tfrec_format)\n\n    # Decode JPEG/PNG\n    image = tf.image.decode_jpeg(example[\"image\"], channels=3)\n    image = tf.image.resize(image, IMAGE_SIZE)\n    image = tf.cast(image, tf.float32) / 255.0  # normalize\n\n    label = example[\"class\"]\n    return image, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-23T18:58:53.843985Z","iopub.execute_input":"2025-11-23T18:58:53.844312Z","iopub.status.idle":"2025-11-23T18:58:53.850387Z","shell.execute_reply.started":"2025-11-23T18:58:53.84429Z","shell.execute_reply":"2025-11-23T18:58:53.849224Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#create dataset from tfrecord files\ndataset = tf.data.TFRecordDataset(filenames)\ndataset = dataset.map(read_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-23T18:59:01.798914Z","iopub.execute_input":"2025-11-23T18:59:01.799304Z","iopub.status.idle":"2025-11-23T18:59:01.833291Z","shell.execute_reply.started":"2025-11-23T18:59:01.799258Z","shell.execute_reply":"2025-11-23T18:59:01.83212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#analyze a random label for 192x192 images: analyyzing label 102\ntarget_label = 102\nimages_with_target_label = []\nfor image, label in dataset:\n    if label.numpy() == target_label:\n        images_with_target_label.append(image.numpy())\nprint(f\"Number of images with label {target_label}: {len(images_with_target_label)}\")\nplt.figure(figsize=(10, 5))\nfor i, img in enumerate(images_with_target_label):\n    plt.subplot(1, len(images_with_target_label), i + 1)\n    plt.imshow(img)\n    plt.title(f\"Label: {target_label}\")\n    plt.axis(\"off\")\nplt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-23T18:59:06.463974Z","iopub.execute_input":"2025-11-23T18:59:06.464261Z","iopub.status.idle":"2025-11-23T18:59:32.779099Z","shell.execute_reply.started":"2025-11-23T18:59:06.464241Z","shell.execute_reply":"2025-11-23T18:59:32.778135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#label counts got from work from teammate\nlabel_counts=[63, 136, 134, 390, 131, 104, 48, 139, 55, 87, 119, 26, 306, 201, 563, 110, 263, 89, 83, 703, 23, 460, 743, 26, 167, 139, 90, 460, 120, 137, 227, 272, 422, 261, 782, 172, 118, 115, 96, 127, 105, 55, 34, 86, 125, 111, 29, 25, 260, 64, 112, 24, 92, 101, 105, 84, 46, 106, 100, 73, 21, 119, 96, 29, 125, 153, 31, 146, 58, 31, 43, 96, 24, 94, 109, 105, 85, 87, 21, 50, 21, 18, 20, 41, 63, 24, 58, 57, 21, 34, 19, 27, 20, 18, 37, 28, 36, 93, 19, 33, 36, 18, 26, 19]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-23T18:59:32.780639Z","iopub.execute_input":"2025-11-23T18:59:32.781056Z","iopub.status.idle":"2025-11-23T18:59:32.787253Z","shell.execute_reply.started":"2025-11-23T18:59:32.78102Z","shell.execute_reply":"2025-11-23T18:59:32.786192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#undersample labels to balance the dataset\ndef undersample_dataset(dataset, target_count):\n    label_to_images = {}\n    for image, label in dataset:\n        label = label.numpy()\n        if label not in label_to_images:\n            label_to_images[label] = []\n        label_to_images[label].append(image.numpy())\n    balanced_images = []\n    balanced_labels = []\n    for label, images in label_to_images.items():\n        if len(images) > target_count:\n            images = images[:target_count]\n        for img in images:\n            balanced_images.append(img)\n            balanced_labels.append(label)\n    balanced_dataset = tf.data.Dataset.from_tensor_slices((balanced_images, balanced_labels))\n    return balanced_dataset\nmin_count = min(label_counts)\nbalanced_dataset = undersample_dataset(dataset, min_count)\nbalanced_label_counts = {}\nfor image, label in balanced_dataset:\n    label = label.numpy()\n    if label not in balanced_label_counts:\n        balanced_label_counts[label] = 0\n    balanced_label_counts[label] += 1\nprint(\"Balanced Label Distribution:\")\nfor label, count in balanced_label_counts.items():\n    print(f\"Label {label}: {count} samples\")\n#visualize balanced label distribution\nlabels_balanced = list(balanced_label_counts.keys())\ncounts_balanced = list(balanced_label_counts.values())\nplt.bar(labels_balanced, counts_balanced)\nplt.xlabel(\"Labels\")\nplt.ylabel(\"Number of Samples\")\nplt.title(\"Balanced Label Distribution\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-23T18:59:32.788174Z","iopub.execute_input":"2025-11-23T18:59:32.788507Z","iopub.status.idle":"2025-11-23T19:01:12.275295Z","shell.execute_reply.started":"2025-11-23T18:59:32.788478Z","shell.execute_reply":"2025-11-23T19:01:12.274346Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now each label has 18 samples, too less to train. we will limit the undersampling to 100","metadata":{}},{"cell_type":"code","source":"#limiting undersambling to 100\ndef undersample_dataset_limited(dataset, max_count):\n    label_to_images = {}\n    for image, label in dataset:\n        label = label.numpy()\n        if label not in label_to_images:\n            label_to_images[label] = []\n        label_to_images[label].append(image.numpy())\n    balanced_images = []\n    balanced_labels = []\n    for label, images in label_to_images.items():\n        if len(images) > max_count:\n            images = images[:max_count]\n        for img in images:\n            balanced_images.append(img)\n            balanced_labels.append(label)\n    balanced_dataset = tf.data.Dataset.from_tensor_slices((balanced_images, balanced_labels))\n    return balanced_dataset\nbalanced_dataset_limited = undersample_dataset_limited(dataset, 100)\nbalanced_label_counts_limited = {}\nfor image, label in balanced_dataset_limited:\n    label = label.numpy()\n    if label not in balanced_label_counts_limited:\n        balanced_label_counts_limited[label] = 0\n    balanced_label_counts_limited[label] += 1\nprint(\"Balanced Label Distribution with Limit 100:\")\nfor label, count in balanced_label_counts_limited.items():\n    print(f\"Label {label}: {count} samples\")\n#visualize balanced label distribution with limit 100\nlabels_balanced_limited = list(balanced_label_counts_limited.keys())\ncounts_balanced_limited = list(balanced_label_counts_limited.values())\nplt.bar(labels_balanced_limited, counts_balanced_limited)\nplt.xlabel(\"Labels\")\nplt.ylabel(\"Number of Samples\")\nplt.title(\"Balanced Label Distribution with Limit 100\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-23T19:01:12.276993Z","iopub.execute_input":"2025-11-23T19:01:12.27732Z","iopub.status.idle":"2025-11-23T19:06:56.384214Z","shell.execute_reply.started":"2025-11-23T19:01:12.277272Z","shell.execute_reply":"2025-11-23T19:06:56.383189Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We'll try to oversample the labels which have samples less than 100 to balance it out","metadata":{}},{"cell_type":"code","source":"#oversampling minority classes to 100 samples each to balance the dataset\ndef oversample_dataset(dataset, target_count):\n    label_to_images = {}\n    for image, label in dataset:\n        label = label.numpy()\n        if label not in label_to_images:\n            label_to_images[label] = []\n        label_to_images[label].append(image.numpy())\n    balanced_images = []\n    balanced_labels = []\n    for label, images in label_to_images.items():\n        current_count = len(images)\n        if current_count < target_count:\n            # Oversample by repeating images\n            images_to_add = images * (target_count // current_count) + images[:(target_count % current_count)]\n        else:\n            images_to_add = images[:target_count]\n        for img in images_to_add:\n            balanced_images.append(img)\n            balanced_labels.append(label)\n    balanced_dataset = tf.data.Dataset.from_tensor_slices((balanced_images, balanced_labels))\n    return balanced_dataset\nbalanced_dataset_oversampled = oversample_dataset(dataset, 100)\nbalanced_label_counts_oversampled = {}\nfor image, label in balanced_dataset_oversampled:\n    label = label.numpy()\n    if label not in balanced_label_counts_oversampled:\n        balanced_label_counts_oversampled[label] = 0\n    balanced_label_counts_oversampled[label] += 1\nprint(\"Oversampled Balanced Label Distribution with 100 samples each:\")\nfor label, count in balanced_label_counts_oversampled.items():\n    print(f\"Label {label}: {count} samples\")\n#visualize oversampled balanced label distribution with 100 samples each\nlabels_balanced_oversampled = list(balanced_label_counts_oversampled.keys())\ncounts_balanced_oversampled = list(balanced_label_counts_oversampled.values())\nplt.bar(labels_balanced_oversampled, counts_balanced_oversampled)\nplt.xlabel(\"Labels\")\nplt.ylabel(\"Number of Samples\")\nplt.title(\"Oversampled Balanced Label Distribution with 100 samples each\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-23T19:06:56.385119Z","iopub.execute_input":"2025-11-23T19:06:56.38539Z","iopub.status.idle":"2025-11-23T19:14:59.920152Z","shell.execute_reply.started":"2025-11-23T19:06:56.385366Z","shell.execute_reply":"2025-11-23T19:14:59.919299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}