{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-24T00:22:44.103223Z","iopub.execute_input":"2025-11-24T00:22:44.104944Z","iopub.status.idle":"2025-11-24T00:22:44.141426Z","shell.execute_reply.started":"2025-11-24T00:22:44.1049Z","shell.execute_reply":"2025-11-24T00:22:44.140224Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#import necessary libraries\nimport tensorflow as tf\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras import layers, models\nimport matplotlib.pyplot as plt\nprint(\"test\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-24T00:22:44.143661Z","iopub.execute_input":"2025-11-24T00:22:44.143961Z","iopub.status.idle":"2025-11-24T00:22:44.150552Z","shell.execute_reply.started":"2025-11-24T00:22:44.143938Z","shell.execute_reply":"2025-11-24T00:22:44.149378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# path to the dataset\ntpu_getting_started_path=\"/kaggle/input/tpu-getting-started\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-24T00:22:44.151737Z","iopub.execute_input":"2025-11-24T00:22:44.152493Z","iopub.status.idle":"2025-11-24T00:22:44.17558Z","shell.execute_reply.started":"2025-11-24T00:22:44.152464Z","shell.execute_reply":"2025-11-24T00:22:44.174426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#path to 192x192 images \ntpu_192x192_path = tpu_getting_started_path + \"/tfrecords-jpeg-192x192\"\n#path to 224x224 images\ntpu_224x224_path = tpu_getting_started_path + \"/tfrecords-jpeg-224x224\"\n#path  to 331x331 images\ntpu_331x331_path = tpu_getting_started_path + \"/tfrecords-jpeg-331x331\"\n#path to 512x512 images\ntpu_512x512_path = tpu_getting_started_path + \"/tfrecords-jpeg-512x512\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-24T00:22:44.176628Z","iopub.execute_input":"2025-11-24T00:22:44.176959Z","iopub.status.idle":"2025-11-24T00:22:44.195709Z","shell.execute_reply.started":"2025-11-24T00:22:44.17693Z","shell.execute_reply":"2025-11-24T00:22:44.194518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#load sample 192x192 image .tfrecord file from train\nfilenames = tf.io.gfile.glob(tpu_192x192_path + \"/train/*.tfrec\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-24T00:22:44.198288Z","iopub.execute_input":"2025-11-24T00:22:44.198566Z","iopub.status.idle":"2025-11-24T00:22:44.224583Z","shell.execute_reply.started":"2025-11-24T00:22:44.198545Z","shell.execute_reply":"2025-11-24T00:22:44.223554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#function to read tfrecord files\nIMAGE_SIZE = [192, 192] \ndef read_tfrecord(example):\n    tfrec_format = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, tfrec_format)\n\n    # Decode JPEG/PNG\n    image = tf.image.decode_jpeg(example[\"image\"], channels=3)\n    image = tf.image.resize(image, IMAGE_SIZE)\n    image = tf.cast(image, tf.float32) / 255.0  # normalize\n\n    label = example[\"class\"]\n    return image, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-24T00:22:44.22574Z","iopub.execute_input":"2025-11-24T00:22:44.225999Z","iopub.status.idle":"2025-11-24T00:22:44.238546Z","shell.execute_reply.started":"2025-11-24T00:22:44.22598Z","shell.execute_reply":"2025-11-24T00:22:44.237106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#create dataset from tfrecord files\ndataset = tf.data.TFRecordDataset(filenames)\ndataset = dataset.map(read_tfrecord, num_parallel_calls=tf.data.AUTOTUNE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-24T00:22:44.239945Z","iopub.execute_input":"2025-11-24T00:22:44.240367Z","iopub.status.idle":"2025-11-24T00:22:44.352183Z","shell.execute_reply.started":"2025-11-24T00:22:44.240334Z","shell.execute_reply":"2025-11-24T00:22:44.351018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#display all images\nplt.figure(figsize=(10, 10))\nfor i, (image, label) in enumerate(dataset.take(15)):\n    plt.subplot(3, 5, i + 1)\n    plt.imshow(image.numpy())\n    plt.title(f\"Label: {label.numpy()}\")\n    plt.axis(\"off\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-24T00:22:44.353451Z","iopub.execute_input":"2025-11-24T00:22:44.353745Z","iopub.status.idle":"2025-11-24T00:22:45.384093Z","shell.execute_reply.started":"2025-11-24T00:22:44.353723Z","shell.execute_reply":"2025-11-24T00:22:45.383012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#check if some images in dataset are blurry\nimport cv2\ndef is_blurry(image, threshold=100.0):\n    gray = cv2.cvtColor((image.numpy() * 255).astype('uint8'), cv2.COLOR_RGB2GRAY)\n    laplacian_var = cv2.Laplacian(gray, cv2.CV_64F).var()\n    return laplacian_var < threshold\nblurry_images = []\nfor image, label in dataset:\n    if is_blurry(image):\n        blurry_images.append((image, label))\nprint(f\"Number of blurry images: {len(blurry_images)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-24T00:22:45.385574Z","iopub.execute_input":"2025-11-24T00:22:45.386008Z","iopub.status.idle":"2025-11-24T00:22:58.820333Z","shell.execute_reply.started":"2025-11-24T00:22:45.385978Z","shell.execute_reply":"2025-11-24T00:22:58.819205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#display blurry images first 30 blurry images\nif blurry_images:\n    plt.figure(figsize=(10, 10))\n    for i, (image, label) in enumerate(blurry_images[:30]):\n        plt.subplot(5, 6, i + 1)\n        plt.imshow(image.numpy())\n        plt.title(f\"Label: {label.numpy()}\")\n        plt.axis(\"off\")\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-24T00:22:58.821321Z","iopub.execute_input":"2025-11-24T00:22:58.821665Z","iopub.status.idle":"2025-11-24T00:23:00.925984Z","shell.execute_reply.started":"2025-11-24T00:22:58.821644Z","shell.execute_reply":"2025-11-24T00:23:00.924579Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Some images are blurry, we might remove them from dataset to get better accuracy","metadata":{}},{"cell_type":"code","source":"#check for grainy images in dataset\ndef is_grainy(image, threshold=50.0):\n    gray = cv2.cvtColor((image.numpy() * 255).astype('uint8'), cv2.COLOR_RGB2GRAY)\n    mean, stddev = cv2.meanStdDev(gray)\n    return stddev[0][0] > threshold\ngrainy_images = []\nfor image, label in dataset:\n    if is_grainy(image):\n        grainy_images.append((image, label))\nprint(f\"Number of grainy images: {len(grainy_images)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-24T00:23:00.927413Z","iopub.execute_input":"2025-11-24T00:23:00.927882Z","iopub.status.idle":"2025-11-24T00:23:10.455229Z","shell.execute_reply.started":"2025-11-24T00:23:00.927851Z","shell.execute_reply":"2025-11-24T00:23:10.454238Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Too many grainy images found, increasing the threshold to filter out most grainy images","metadata":{}},{"cell_type":"code","source":"#increasing threshold to 90.0 and checking for grainy images again\ngrainy_images = []\nfor image, label in dataset:\n    if is_grainy(image, threshold=90.0):\n        grainy_images.append((image, label))\nprint(f\"Number of grainy images with threshold 90.0: {len(grainy_images)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-24T00:23:10.456486Z","iopub.execute_input":"2025-11-24T00:23:10.456886Z","iopub.status.idle":"2025-11-24T00:23:19.774438Z","shell.execute_reply.started":"2025-11-24T00:23:10.45686Z","shell.execute_reply":"2025-11-24T00:23:19.773083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#display first 30 grainy images\nif grainy_images:\n    plt.figure(figsize=(10, 10))\n    for i, (image, label) in enumerate(grainy_images[:30]):\n        plt.subplot(5, 6, i + 1)\n        plt.imshow(image.numpy())\n        plt.title(f\"Label: {label.numpy()}\")\n        plt.axis(\"off\")\n    plt.show()\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-24T00:23:19.775635Z","iopub.execute_input":"2025-11-24T00:23:19.776006Z","iopub.status.idle":"2025-11-24T00:23:21.463692Z","shell.execute_reply.started":"2025-11-24T00:23:19.77597Z","shell.execute_reply":"2025-11-24T00:23:21.462164Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Images look alright, might not mark them as grainy","metadata":{}},{"cell_type":"code","source":"#checking if dataset consists of black and white images\ndef is_black_and_white(image, threshold=5.0):\n    gray = cv2.cvtColor((image.numpy() * 255).astype('uint8'), cv2.COLOR_RGB2GRAY)\n    b, g, r = cv2.split((image.numpy() * 255).astype('uint8'))\n    diff_rg = cv2.absdiff(r, g)\n    diff_rb = cv2.absdiff(r, b)\n    diff_gb = cv2.absdiff(g, b)\n    mean_diff = (cv2.mean(diff_rg)[0] + cv2.mean(diff_rb)[0] + cv2.mean(diff_gb)[0]) / 3\n    return mean_diff < threshold\nblack_and_white_images = []\nfor image, label in dataset:\n    if is_black_and_white(image):\n        black_and_white_images.append((image, label))\nprint(f\"Number of black and white images: {len(black_and_white_images)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-24T00:23:21.468192Z","iopub.execute_input":"2025-11-24T00:23:21.468548Z","iopub.status.idle":"2025-11-24T00:23:33.409812Z","shell.execute_reply.started":"2025-11-24T00:23:21.46852Z","shell.execute_reply":"2025-11-24T00:23:33.408776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#display first 30 black and white images\nif black_and_white_images:\n    plt.figure(figsize=(10, 10))\n    for i, (image, label) in enumerate(black_and_white_images[:30]):\n        plt.subplot(5, 6, i + 1)\n        plt.imshow(image.numpy())\n        plt.title(f\"Label: {label.numpy()}\")\n        plt.axis(\"off\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-24T00:23:33.410643Z","iopub.execute_input":"2025-11-24T00:23:33.411043Z","iopub.status.idle":"2025-11-24T00:23:35.421708Z","shell.execute_reply.started":"2025-11-24T00:23:33.411019Z","shell.execute_reply":"2025-11-24T00:23:35.420324Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Will Remove black and white images from dataset. Black and white images will be created later in data augmentation","metadata":{}}]}