{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91498,"databundleVersionId":11655853,"sourceType":"competition"},{"sourceId":7884485,"sourceType":"datasetVersion","datasetId":4628051}],"dockerImageVersionId":31040,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:28.226230Z","iopub.execute_input":"2025-05-24T21:44:28.226530Z","iopub.status.idle":"2025-05-24T21:44:33.045654Z","shell.execute_reply.started":"2025-05-24T21:44:28.226502Z","shell.execute_reply":"2025-05-24T21:44:33.044446Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import libraries\nimport tensorflow as tf\nfrom tensorflow import keras\n\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense\nfrom tensorflow.keras.models import Sequential\n\nimport os\nimport pandas as pd\n\nfrom matplotlib import pyplot as plt\nimport time","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:33.047702Z","iopub.execute_input":"2025-05-24T21:44:33.048284Z","iopub.status.idle":"2025-05-24T21:44:52.361437Z","shell.execute_reply.started":"2025-05-24T21:44:33.048256Z","shell.execute_reply":"2025-05-24T21:44:52.360279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Defining Constants\nImage_size = (128, 128)\n\nImg_width = 128\nImg_height = 128\n\nbatch_size = 32\nepochs = 15\nchannels = 3\n\n# Define image dimensions\nnum_classes = 7","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:52.362460Z","iopub.execute_input":"2025-05-24T21:44:52.363129Z","iopub.status.idle":"2025-05-24T21:44:52.368489Z","shell.execute_reply.started":"2025-05-24T21:44:52.363096Z","shell.execute_reply":"2025-05-24T21:44:52.367425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the datasets\n# 1. sample_submission.csv\nsample_submission = pd.read_csv(\"/kaggle/input/image-matching-challenge-2025/sample_submission.csv\")\nsample_submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:52.369442Z","iopub.execute_input":"2025-05-24T21:44:52.369749Z","iopub.status.idle":"2025-05-24T21:44:52.468636Z","shell.execute_reply.started":"2025-05-24T21:44:52.369721Z","shell.execute_reply":"2025-05-24T21:44:52.467629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. train_labels.csv\ntrain_labels = pd.read_csv(\"/kaggle/input/image-matching-challenge-2025/train_labels.csv\")\ntrain_labels.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:52.470642Z","iopub.execute_input":"2025-05-24T21:44:52.470920Z","iopub.status.idle":"2025-05-24T21:44:52.499926Z","shell.execute_reply.started":"2025-05-24T21:44:52.470899Z","shell.execute_reply":"2025-05-24T21:44:52.498859Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3. categories.csv\ncategories = pd.read_csv(\"/kaggle/input/image-matching-challenge-2025/train_thresholds.csv\")\ncategories.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:52.500918Z","iopub.execute_input":"2025-05-24T21:44:52.501189Z","iopub.status.idle":"2025-05-24T21:44:52.515479Z","shell.execute_reply.started":"2025-05-24T21:44:52.501166Z","shell.execute_reply":"2025-05-24T21:44:52.514502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plotting\n# 1. sample_submission.csv\nsample_submission['dataset'].value_counts().plot(kind='bar')\nplt.xlabel('Category')\nplt.ylabel('Vals')\nplt.title('Sample Submission')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:52.516505Z","iopub.execute_input":"2025-05-24T21:44:52.516943Z","iopub.status.idle":"2025-05-24T21:44:52.898237Z","shell.execute_reply.started":"2025-05-24T21:44:52.516911Z","shell.execute_reply":"2025-05-24T21:44:52.897212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for missing values in the three datasets\nprint(\"Missing values in Sample-Submission:\", sample_submission.isnull().sum())\n\nprint(\"Missing values in Train-labels:\", train_labels.isnull().sum())\n\nprint(\"Missing values in Categories:\", categories.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:52.899210Z","iopub.execute_input":"2025-05-24T21:44:52.899529Z","iopub.status.idle":"2025-05-24T21:44:52.910250Z","shell.execute_reply.started":"2025-05-24T21:44:52.899505Z","shell.execute_reply":"2025-05-24T21:44:52.908977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for duplicated values in the three datasets\nprint(\"Duplicates in Sample-Submission:\", sample_submission.duplicated().any())\n\nprint(\"Duplicates in Train-labels:\", train_labels.duplicated().any())\n\nprint(\"Duplicates in Categories:\", categories.duplicated().any())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:52.911028Z","iopub.execute_input":"2025-05-24T21:44:52.911284Z","iopub.status.idle":"2025-05-24T21:44:52.943974Z","shell.execute_reply.started":"2025-05-24T21:44:52.911263Z","shell.execute_reply":"2025-05-24T21:44:52.943021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Path to the train and test directories\ntrain_dir = '/kaggle/input/image-matching-challenge-2025/train'\ntest_dir = '/kaggle/input/image-matching-challenge-2025/test'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:52.945197Z","iopub.execute_input":"2025-05-24T21:44:52.945509Z","iopub.status.idle":"2025-05-24T21:44:52.960099Z","shell.execute_reply.started":"2025-05-24T21:44:52.945485Z","shell.execute_reply":"2025-05-24T21:44:52.958932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Path for the training images and testing\ntrain_images = tf.data.Dataset.list_files(train_dir + '/*')\ntest_images = tf.data.Dataset.list_files(test_dir + '/*')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:52.961512Z","iopub.execute_input":"2025-05-24T21:44:52.961810Z","iopub.status.idle":"2025-05-24T21:44:53.059152Z","shell.execute_reply.started":"2025-05-24T21:44:52.961786Z","shell.execute_reply":"2025-05-24T21:44:53.058064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filtering the training and testing images to exclude .csv\ntrain_images = train_images.filter(lambda x: not tf.strings.regex_full_match(x, '.*\\.csv'))\ntest_images = test_images.filter(lambda x: not tf.strings.regex_full_match(x, '.*\\.csv'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:53.060231Z","iopub.execute_input":"2025-05-24T21:44:53.060587Z","iopub.status.idle":"2025-05-24T21:44:53.118827Z","shell.execute_reply.started":"2025-05-24T21:44:53.060549Z","shell.execute_reply":"2025-05-24T21:44:53.117751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for file_path in train_images:\n    print(file_path.numpy())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:53.120223Z","iopub.execute_input":"2025-05-24T21:44:53.120963Z","iopub.status.idle":"2025-05-24T21:44:53.189446Z","shell.execute_reply.started":"2025-05-24T21:44:53.120931Z","shell.execute_reply":"2025-05-24T21:44:53.188534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for file_path in test_images:\n    print(file_path.numpy())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:53.193459Z","iopub.execute_input":"2025-05-24T21:44:53.193760Z","iopub.status.idle":"2025-05-24T21:44:53.212925Z","shell.execute_reply.started":"2025-05-24T21:44:53.193736Z","shell.execute_reply":"2025-05-24T21:44:53.211879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Shuffing\ntrain_images = train_images.shuffle(500)\ntest_images = test_images.shuffle(500)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:53.214510Z","iopub.execute_input":"2025-05-24T21:44:53.214951Z","iopub.status.idle":"2025-05-24T21:44:53.226407Z","shell.execute_reply.started":"2025-05-24T21:44:53.214918Z","shell.execute_reply":"2025-05-24T21:44:53.225181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for file_path in train_images:\n    print(file_path.numpy())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:53.227625Z","iopub.execute_input":"2025-05-24T21:44:53.228023Z","iopub.status.idle":"2025-05-24T21:44:53.260891Z","shell.execute_reply.started":"2025-05-24T21:44:53.227994Z","shell.execute_reply":"2025-05-24T21:44:53.259947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Classes in our dataset both of training and testing\ntrain_classes = [\"church\", \"dioscuri\", \"lizard\", \"multi-temporal-temple-baalshar\", \"pond\", \"transp_obj_glass_cup\", \"transp_obj_glass_cylinder\"]\ntest_classes = [\"church\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:53.262159Z","iopub.execute_input":"2025-05-24T21:44:53.263054Z","iopub.status.idle":"2025-05-24T21:44:53.267249Z","shell.execute_reply.started":"2025-05-24T21:44:53.263028Z","shell.execute_reply":"2025-05-24T21:44:53.266114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Finding the count of classes for training\ntrain_images_count = 0\nfor _ in train_images:\n    train_images_count += 1\n\nprint(train_images_count)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:53.268431Z","iopub.execute_input":"2025-05-24T21:44:53.268745Z","iopub.status.idle":"2025-05-24T21:44:53.302765Z","shell.execute_reply.started":"2025-05-24T21:44:53.268715Z","shell.execute_reply":"2025-05-24T21:44:53.301045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Finding the count of classes for testing\ntest_images_count = 0\nfor _ in test_images:\n    test_images_count += 1\n\nprint(test_images_count)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:53.304005Z","iopub.execute_input":"2025-05-24T21:44:53.304351Z","iopub.status.idle":"2025-05-24T21:44:53.328464Z","shell.execute_reply.started":"2025-05-24T21:44:53.304320Z","shell.execute_reply":"2025-05-24T21:44:53.326936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_size = int(train_images_count + test_images_count * 0.8)\n\ntrain_ds = train_images.take(train_size)\ntest_ds = train_images.skip(train_size)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:53.329690Z","iopub.execute_input":"2025-05-24T21:44:53.330036Z","iopub.status.idle":"2025-05-24T21:44:53.346571Z","shell.execute_reply.started":"2025-05-24T21:44:53.330006Z","shell.execute_reply":"2025-05-24T21:44:53.345315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_label(file_path):\n    return tf.strings.split(file_path, os.path.sep)[-2]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:53.347734Z","iopub.execute_input":"2025-05-24T21:44:53.348557Z","iopub.status.idle":"2025-05-24T21:44:53.364505Z","shell.execute_reply.started":"2025-05-24T21:44:53.348506Z","shell.execute_reply":"2025-05-24T21:44:53.363438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_image(file_path):\n    label = get_label(file_path)\n\n    img = tf.io.read_file(file_path)\n    img = tf.image.decode_jpeg(img)\n    img = tf.image.resize(img, [128, 128])\n\n    return img, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:53.365567Z","iopub.execute_input":"2025-05-24T21:44:53.365915Z","iopub.status.idle":"2025-05-24T21:44:53.382672Z","shell.execute_reply.started":"2025-05-24T21:44:53.365883Z","shell.execute_reply":"2025-05-24T21:44:53.381604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for t in train_ds.take(4):\n    print (t.numpy())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:53.383648Z","iopub.execute_input":"2025-05-24T21:44:53.383998Z","iopub.status.idle":"2025-05-24T21:44:53.421159Z","shell.execute_reply.started":"2025-05-24T21:44:53.383970Z","shell.execute_reply":"2025-05-24T21:44:53.420261Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.image as mpimg","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:53.422277Z","iopub.execute_input":"2025-05-24T21:44:53.422732Z","iopub.status.idle":"2025-05-24T21:44:53.427218Z","shell.execute_reply.started":"2025-05-24T21:44:53.422707Z","shell.execute_reply":"2025-05-24T21:44:53.425983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_image(file_path, label):\n\n    img = tf.io.read_file(file_path)\n\n    img = tf.image.decode_jpeg(img, channels=3)  # Adjust channels if needed\n\n    img = tf.image.resize(img, [img_height, img_width])\n\n    img = img / 255.0\n    return img, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:53.428066Z","iopub.execute_input":"2025-05-24T21:44:53.428322Z","iopub.status.idle":"2025-05-24T21:44:53.446070Z","shell.execute_reply.started":"2025-05-24T21:44:53.428296Z","shell.execute_reply":"2025-05-24T21:44:53.445147Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Showing Random Images from Directory (Train and Test)","metadata":{}},{"cell_type":"code","source":"image_path = \"/kaggle/input/image-matching-challenge-2025/train/pt_stpeters_stpauls/st_peters_square_35727766_2927321004.png\"\n\n# Load the image\nimage = mpimg.imread(image_path)\n\n# Display the image\nplt.imshow(image)\nplt.axis('off')  # Turn off axis\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:53.447087Z","iopub.execute_input":"2025-05-24T21:44:53.447412Z","iopub.status.idle":"2025-05-24T21:44:53.837834Z","shell.execute_reply.started":"2025-05-24T21:44:53.447383Z","shell.execute_reply":"2025-05-24T21:44:53.836937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_path = \"/kaggle/input/image-matching-challenge-2025/train/pt_stpeters_stpauls/st_peters_square_35727766_2927321004.png\"\n\n# Load the image\nimage = mpimg.imread(image_path)\n\n# Display the image\nplt.imshow(image)\nplt.axis('off')  # Turn off axis\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:53.838988Z","iopub.execute_input":"2025-05-24T21:44:53.839298Z","iopub.status.idle":"2025-05-24T21:44:54.173990Z","shell.execute_reply.started":"2025-05-24T21:44:53.839274Z","shell.execute_reply":"2025-05-24T21:44:54.173062Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Loading Images from Test directory\n","metadata":{}},{"cell_type":"code","source":"image_path = \"/kaggle/input/image-matching-challenge-2025/train/amy_gardens/peach_0026.png\"\n\n# Load the image\nimage = mpimg.imread(image_path)\n\n# Display the image\nplt.imshow(image)\nplt.axis('off')  # Turn off axis\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:54.175100Z","iopub.execute_input":"2025-05-24T21:44:54.175717Z","iopub.status.idle":"2025-05-24T21:44:54.400201Z","shell.execute_reply.started":"2025-05-24T21:44:54.175675Z","shell.execute_reply":"2025-05-24T21:44:54.399225Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Applying Prefetch and Cache for performance enhancement\n","metadata":{}},{"cell_type":"markdown","source":"# 1. Prefetch\n","metadata":{}},{"cell_type":"code","source":"def process_image(file_path, label):\n    img = tf.io.read_file(file_path)\n    img = tf.image.decode_jpeg(img, channels=3)\n    img = tf.image.resize(img, [128, 128])\n    img = img / 255.0\n    return img, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:54.401098Z","iopub.execute_input":"2025-05-24T21:44:54.401346Z","iopub.status.idle":"2025-05-24T21:44:54.406642Z","shell.execute_reply.started":"2025-05-24T21:44:54.401327Z","shell.execute_reply":"2025-05-24T21:44:54.405675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ds = train_images.take(train_size).map(lambda x: process_image(x, get_label(x)), num_parallel_calls=tf.data.experimental.AUTOTUNE)\ntest_ds = train_images.skip(train_size).map(lambda x: process_image(x, get_label(x)), num_parallel_calls=tf.data.experimental.AUTOTUNE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:54.407539Z","iopub.execute_input":"2025-05-24T21:44:54.407787Z","iopub.status.idle":"2025-05-24T21:44:54.686149Z","shell.execute_reply.started":"2025-05-24T21:44:54.407770Z","shell.execute_reply":"2025-05-24T21:44:54.685094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Before prefetching\nstart_time = time.time()\n\n# Define and preprocess the dataset without prefetching\ntrain_ds = train_images.take(train_size).map(lambda x: process_image(x, get_label(x)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:54.687526Z","iopub.execute_input":"2025-05-24T21:44:54.687888Z","iopub.status.idle":"2025-05-24T21:44:54.793007Z","shell.execute_reply.started":"2025-05-24T21:44:54.687844Z","shell.execute_reply":"2025-05-24T21:44:54.792121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Apply prefetching\ntrain_ds = train_ds.prefetch(buffer_size=1)\n\n# Calculate time taken\ntime_after_prefetch1 = time.time() - start_time","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:54.794084Z","iopub.execute_input":"2025-05-24T21:44:54.794380Z","iopub.status.idle":"2025-05-24T21:44:54.800786Z","shell.execute_reply.started":"2025-05-24T21:44:54.794359Z","shell.execute_reply":"2025-05-24T21:44:54.799673Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Time taken after prefetching buffer_size 1:\", time_after_prefetch1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:54.802088Z","iopub.execute_input":"2025-05-24T21:44:54.802394Z","iopub.status.idle":"2025-05-24T21:44:54.822092Z","shell.execute_reply.started":"2025-05-24T21:44:54.802363Z","shell.execute_reply":"2025-05-24T21:44:54.821031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Apply prefetching\ntrain_ds = train_ds.prefetch(buffer_size=2)\n\n# Calculate time taken\ntime_after_prefetch2 = time.time() - start_time","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:54.823007Z","iopub.execute_input":"2025-05-24T21:44:54.823264Z","iopub.status.idle":"2025-05-24T21:44:54.849509Z","shell.execute_reply.started":"2025-05-24T21:44:54.823244Z","shell.execute_reply":"2025-05-24T21:44:54.848520Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Time taken after prefetching buffer size 2:\", time_after_prefetch2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:54.850457Z","iopub.execute_input":"2025-05-24T21:44:54.850797Z","iopub.status.idle":"2025-05-24T21:44:54.862522Z","shell.execute_reply.started":"2025-05-24T21:44:54.850767Z","shell.execute_reply":"2025-05-24T21:44:54.861286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Apply prefetching\ntrain_ds = train_ds.prefetch(buffer_size=tf.data.experimental.AUTOTUNE)\n\n# Calculate time taken\ntime_after_prefetch_AUTOTUNE = time.time() - start_time","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:54.863830Z","iopub.execute_input":"2025-05-24T21:44:54.864134Z","iopub.status.idle":"2025-05-24T21:44:54.887574Z","shell.execute_reply.started":"2025-05-24T21:44:54.864112Z","shell.execute_reply":"2025-05-24T21:44:54.886544Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Time taken after prefetching buffer size AUTOTUNE:\", time_after_prefetch_AUTOTUNE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:54.888576Z","iopub.execute_input":"2025-05-24T21:44:54.888929Z","iopub.status.idle":"2025-05-24T21:44:54.906949Z","shell.execute_reply.started":"2025-05-24T21:44:54.888905Z","shell.execute_reply":"2025-05-24T21:44:54.905944Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Cache\n","metadata":{}},{"cell_type":"code","source":"# Before caching\nstart_time = time.time()\n\n# Define and preprocess the dataset without prefetching and caching\ntrain_ds_no_cache = train_images.take(train_size).map(lambda x: process_image(x, get_label(x)), num_parallel_calls=tf.data.experimental.AUTOTUNE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:54.908073Z","iopub.execute_input":"2025-05-24T21:44:54.908466Z","iopub.status.idle":"2025-05-24T21:44:55.031939Z","shell.execute_reply.started":"2025-05-24T21:44:54.908438Z","shell.execute_reply":"2025-05-24T21:44:55.030682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate time taken\ntime_before_cache = time.time() - start_time\n\n# Apply caching\ntrain_ds_cached = train_ds_no_cache.cache()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:55.032938Z","iopub.execute_input":"2025-05-24T21:44:55.033182Z","iopub.status.idle":"2025-05-24T21:44:55.044036Z","shell.execute_reply.started":"2025-05-24T21:44:55.033163Z","shell.execute_reply":"2025-05-24T21:44:55.042715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate time taken\ntime_after_cache = time.time() - start_time","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:55.045209Z","iopub.execute_input":"2025-05-24T21:44:55.045647Z","iopub.status.idle":"2025-05-24T21:44:55.059392Z","shell.execute_reply.started":"2025-05-24T21:44:55.045609Z","shell.execute_reply":"2025-05-24T21:44:55.058468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Time taken before caching:\", time_before_cache)\nprint(\"Time taken after caching:\", time_after_cache)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:44:55.064515Z","iopub.execute_input":"2025-05-24T21:44:55.064954Z","iopub.status.idle":"2025-05-24T21:44:55.078185Z","shell.execute_reply.started":"2025-05-24T21:44:55.064930Z","shell.execute_reply":"2025-05-24T21:44:55.077321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport io\n\ndata = \"\"\"dataset,scene,image,rotation_matrix,translation_vector\ndataset1,cluster1,image1.png,0.1;0.2;0.3;0.4;0.5;0.6;0.7;0.8;0.9,0.1;0.2;0.3\ndataset1,cluster1,image2.png,0.1;0.2;0.3;0.4;0.5;0.6;0.7;0.8;0.9,0.1;0.2;0.3\ndataset1,cluster2,image3.png,0.1;0.2;0.3;0.4;0.5;0.6;0.7;0.8;0.9,0.1;0.2;0.3\ndataset1,cluster2,image4.png,0.1;0.2;0.3;0.4;0.5;0.6;0.7;0.8;0.9,0.1;0.2;0.3\ndataset1,outliers,image5.png,nan;nan;nan;nan;nan;nan;nan;nan;nan,nan;nan;nan\ndataset2,cluster1,image1.png,0.1;0.2;0.3;0.4;0.5;0.6;0.7;0.8;0.9,0.1;0.2;0.3\n\"\"\"\n\ndf = pd.read_csv(io.StringIO(data))\n\ndef parse_matrix(matrix_str):\n    try:\n        return np.array([float(x) for x in matrix_str.split(';')]).reshape((3, 3))\n    except:\n        return np.full((3, 3), np.nan)\n\ndef parse_vector(vector_str):\n    try:\n        return np.array([float(x) for x in vector_str.split(';')]).reshape((3,))\n    except:\n        return np.full((3,), np.nan)\n\ndf['rotation_matrix'] = df['rotation_matrix'].apply(parse_matrix)\ndf['translation_vector'] = df['translation_vector'].apply(parse_vector)\n\nprint(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:46:41.382569Z","iopub.execute_input":"2025-05-24T21:46:41.382915Z","iopub.status.idle":"2025-05-24T21:46:41.401812Z","shell.execute_reply.started":"2025-05-24T21:46:41.382891Z","shell.execute_reply":"2025-05-24T21:46:41.400762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport io\n\ndata = \"\"\"dataset,scene,image,rotation_matrix,translation_vector\ndataset1,cluster1,image1.png,0.1;0.2;0.3;0.4;0.5;0.6;0.7;0.8;0.9,0.1;0.2;0.3\ndataset1,cluster1,image2.png,0.1;0.2;0.3;0.4;0.5;0.6;0.7;0.8;0.9,0.1;0.2;0.3\ndataset1,cluster2,image3.png,0.1;0.2;0.3;0.4;0.5;0.6;0.7;0.8;0.9,0.1;0.2;0.3\ndataset1,cluster2,image4.png,0.1;0.2;0.3;0.4;0.5;0.6;0.7;0.8;0.9,0.1;0.2;0.3\ndataset1,outliers,image5.png,nan;nan;nan;nan;nan;nan;nan;nan;nan,nan;nan;nan\ndataset2,cluster1,image1.png,0.1;0.2;0.3;0.4;0.5;0.6;0.7;0.8;0.9,0.1;0.2;0.3\ndataset2,cluster1,image2.png,0.1;0.2;0.3;0.4;0.5;0.6;0.7;0.8;0.9,0.1;0.2;0.3\ndataset2,outliers,image3.png,nan;nan;nan;nan;nan;nan;nan;nan;nan,nan;nan;nan\ndataset3,unclustered,image1.png,nan;nan;nan;nan;nan;nan;nan;nan;nan,nan;nan;nan\ndataset3,unclustered,image2.png,nan;nan;nan;nan;nan;nan;nan;nan;nan,nan;nan;nan\n\"\"\"\n\ndf = pd.read_csv(io.StringIO(data))\n\ndef parse_matrix(matrix_str):\n    values = matrix_str.split(';')\n    try:\n        return np.array([float(x) for x in values]).reshape((3, 3))\n    except ValueError:\n        return np.full((3, 3), np.nan)\n\ndef parse_vector(vector_str):\n    values = vector_str.split(';')\n    try:\n        return np.array([float(x) for x in values]).reshape((3,))\n    except ValueError:\n        return np.full((3,), np.nan)\n\ndf['rotation_matrix'] = df['rotation_matrix'].apply(parse_matrix)\ndf['translation_vector'] = df['translation_vector'].apply(parse_vector)\n\nprint(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T21:48:55.108177Z","iopub.execute_input":"2025-05-24T21:48:55.108502Z","iopub.status.idle":"2025-05-24T21:48:55.129242Z","shell.execute_reply.started":"2025-05-24T21:48:55.108479Z","shell.execute_reply":"2025-05-24T21:48:55.128135Z"}},"outputs":[],"execution_count":null}]}