{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":6799,"databundleVersionId":4225553,"sourceType":"competition"}],"dockerImageVersionId":30699,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-25T21:49:13.707119Z","iopub.execute_input":"2024-04-25T21:49:13.70815Z","iopub.status.idle":"2024-04-25T21:49:13.714431Z","shell.execute_reply.started":"2024-04-25T21:49:13.708114Z","shell.execute_reply":"2024-04-25T21:49:13.71308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install torch torchvision","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:49:13.716491Z","iopub.execute_input":"2024-04-25T21:49:13.717335Z","iopub.status.idle":"2024-04-25T21:49:28.159716Z","shell.execute_reply.started":"2024-04-25T21:49:13.717295Z","shell.execute_reply":"2024-04-25T21:49:28.158349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# importing libraries\nimport torch\nimport torchvision\nfrom torchvision.models import resnet50, ResNet50_Weights\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:49:28.16146Z","iopub.execute_input":"2024-04-25T21:49:28.161831Z","iopub.status.idle":"2024-04-25T21:49:28.168684Z","shell.execute_reply.started":"2024-04-25T21:49:28.161796Z","shell.execute_reply":"2024-04-25T21:49:28.167638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_torch = resnet50(weights=ResNet50_Weights.IMAGENET1K_V1)","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:49:28.17127Z","iopub.execute_input":"2024-04-25T21:49:28.171626Z","iopub.status.idle":"2024-04-25T21:49:28.799953Z","shell.execute_reply.started":"2024-04-25T21:49:28.171597Z","shell.execute_reply":"2024-04-25T21:49:28.798799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install ftfy regex tqdm\n!pip install git+https://github.com/openai/CLIP.git","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:49:28.801549Z","iopub.execute_input":"2024-04-25T21:49:28.801989Z","iopub.status.idle":"2024-04-25T21:49:59.756945Z","shell.execute_reply.started":"2024-04-25T21:49:28.801959Z","shell.execute_reply":"2024-04-25T21:49:59.755829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import clip\nclip.available_models()","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:49:59.758827Z","iopub.execute_input":"2024-04-25T21:49:59.759188Z","iopub.status.idle":"2024-04-25T21:49:59.767541Z","shell.execute_reply.started":"2024-04-25T21:49:59.759153Z","shell.execute_reply":"2024-04-25T21:49:59.76595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_clip, preprocess_clip = clip.load('RN50', device)\nprint(model_clip)","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:49:59.769874Z","iopub.execute_input":"2024-04-25T21:49:59.770388Z","iopub.status.idle":"2024-04-25T21:50:03.938556Z","shell.execute_reply.started":"2024-04-25T21:49:59.770347Z","shell.execute_reply":"2024-04-25T21:50:03.937425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.2","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# 2.3","metadata":{}},{"cell_type":"code","source":"#importing other libraries\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:50:03.940054Z","iopub.execute_input":"2024-04-25T21:50:03.940391Z","iopub.status.idle":"2024-04-25T21:50:03.945394Z","shell.execute_reply.started":"2024-04-25T21:50:03.940363Z","shell.execute_reply":"2024-04-25T21:50:03.944265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to use CLIP model to evaluate an image and give back its class\n\n# Load class labels\nwith open(\"/kaggle/input/imagenet-object-localization-challenge/LOC_synset_mapping.txt\") as f:\n    image_classes = [line.strip() for line in f.readlines()]\n\ntext_input = clip.tokenize(image_classes).to(device)\n    \ndef predict_CLIP(image, top_k =1):\n    image_input = preprocess_clip(image).unsqueeze(0).to(device)\n    \n    model_clip.cuda().eval()\n    with torch.no_grad():\n        image_features = model_clip.encode_image(image_input)\n        text_features = model_clip.encode_text(text_input)\n        \n        \n    image_features /= image_features.norm(dim=-1, keepdim=True)\n    text_features /= text_features.norm(dim=-1, keepdim=True)\n    similarity = (100.0 * image_features @ text_features.T).softmax(dim=-1)\n    values, indices = similarity[0].topk(top_k)\n#     print(values, indices)\n    \n    if top_k ==1:\n        temp = image_classes[indices]\n        class_parts = temp.split(', ')\n        class_names = []\n        for part in class_parts:\n            words = part.split(' ')\n            if len(words) > 1:\n                class_names.append(' '.join(words[1:]))  # Join the words after the first one with space\n            else:\n                class_names.append(words[0])  # If there's only one word, add it as it is\n\n        predicted_class = str(class_names)\n    else:\n        predicted_class = []\n        for i in indices:\n            temp = image_classes[i]\n            class_parts = temp.split(',')\n            class_names = []\n            for part in class_parts:\n                words = part.split(' ')\n                if len(words) > 1:\n                    class_names.append(' '.join(words[1:]))  # Join the words after the first one with space\n                else:\n                    class_names.append(words[0])  # If there's only one word, add it as it is\n\n                    class_name = temp.split(',')[0].split(' ', 1)[1].strip()\n            predicted_class.append(class_names)\n        \n    \n    return predicted_class, values\n\n\n ","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:50:03.946955Z","iopub.execute_input":"2024-04-25T21:50:03.947286Z","iopub.status.idle":"2024-04-25T21:50:04.139682Z","shell.execute_reply.started":"2024-04-25T21:50:03.947258Z","shell.execute_reply":"2024-04-25T21:50:04.138817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_images_list = ['/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/n01818515/n01818515_10062.JPEG' ,'/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/n02071294/n02071294_10318.JPEG','/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/n01664065/n01664065_10099.JPEG' ,'/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/n02091244/n02091244_100.JPEG', '/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/n01443537/n01443537_10034.JPEG']\nnum_images = len(test_images_list)\n\nnum_cols = min(1, num_images)  # Maximum 3 columns\nnum_rows = (num_images + num_cols - 1) // num_cols\n\nfig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 5*num_rows))\naxes = axes.flatten()\n\nfor i in range(num_images):\n    test_image = Image.open(test_images_list[i])\n    predicted_class, probability = predict_CLIP(test_image, 1)\n    axes[i].imshow(test_image)\n    axes[i].axis(\"off\")\n    title = \"Predicted class: \" + str(predicted_class) + \" with probability: \" + str(probability.item())\n    axes[i].set_title(title)\n    \nfor j in range(i+1, len(axes)):\n    fig.delaxes(axes[j])\n\nplt.tight_layout()\nplt.show()\n   \n    ","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:50:04.143482Z","iopub.execute_input":"2024-04-25T21:50:04.143836Z","iopub.status.idle":"2024-04-25T21:50:09.65121Z","shell.execute_reply.started":"2024-04-25T21:50:04.143807Z","shell.execute_reply":"2024-04-25T21:50:09.649971Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to predict using image net\nimport torch.nn.functional as F   \n    \ndef predict_ImageNet(image, top_k = 1):\n    weights = ResNet50_Weights.IMAGENET1K_V1\n    preprocess_imagenet = weights.transforms().to(device)\n    image = image.convert('RGB')\n    image_input = preprocess_imagenet(image).unsqueeze(0).to(device)\n\n    \n    model_torch.cuda().eval()\n    with torch.no_grad():\n        logits = model_torch(image_input)\n    probs = F.softmax(logits, dim=1) \n#     print(len(logits))\n#     print(len(logits[0]))\n#     print(len(probs))\n#     print(len(probs[0]))\n\n    top_probs, top_indices = torch.topk(probs, top_k, dim=1)\n    \n    if top_k ==1:\n        values = top_probs\n        predicted_class = weights.meta['categories'][top_indices]\n    else:\n        values = top_probs\n        predicted_class = []\n        for i in top_indices:\n            for j in i:\n                predicted_class.append(weights.meta['categories'][j])\n        \n    return predicted_class, values\n \n","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:50:09.652585Z","iopub.execute_input":"2024-04-25T21:50:09.652946Z","iopub.status.idle":"2024-04-25T21:50:09.663006Z","shell.execute_reply.started":"2024-04-25T21:50:09.652914Z","shell.execute_reply":"2024-04-25T21:50:09.661577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# testing the above function on 1 image\ntest_images_list = ['/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/n01818515/n01818515_10062.JPEG' ,'/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/n02071294/n02071294_10318.JPEG','/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/n01664065/n01664065_10099.JPEG' ,'/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/n02091244/n02091244_100.JPEG', '/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/n01443537/n01443537_10034.JPEG']\nnum_images = len(test_images_list)\n\nnum_cols = min(1, num_images)  # Maximum 3 columns\nnum_rows = (num_images + num_cols - 1) // num_cols\n\nfig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 5*num_rows))\naxes = axes.flatten()\n\nfor i in range(num_images):\n    test_image = Image.open(test_images_list[i])\n    predicted_class, probability = predict_ImageNet(test_image, 1)\n\n    axes[i].imshow(test_image)\n    axes[i].axis(\"off\")\n    title = \"Predicted class: \" + str(predicted_class) + \" with probability: \" + str(probability.item())\n    axes[i].set_title(title)\n    \nfor j in range(i+1, len(axes)):\n    fig.delaxes(axes[j])\n\nplt.tight_layout()\nplt.show()\n   ","metadata":{"jupyter":{"source_hidden":true,"outputs_hidden":true},"execution":{"iopub.status.busy":"2024-04-25T21:50:09.664798Z","iopub.execute_input":"2024-04-25T21:50:09.665197Z","iopub.status.idle":"2024-04-25T21:50:11.686579Z","shell.execute_reply.started":"2024-04-25T21:50:09.665146Z","shell.execute_reply":"2024-04-25T21:50:11.685089Z"},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.4","metadata":{}},{"cell_type":"markdown","source":"**to do 2.4, we will look at various folders, iterate through the images in them and use the models to predict them.\nif the top 5 are not predicted by some model but correctly by the other one, then we will append it to a list**","metadata":{}},{"cell_type":"code","source":"# modified clip and imagenet\nimport os\n    \ndef predict_CLIP_m(images, correct_index, top_k =5):\n    \n    with open(\"/kaggle/input/imagenet-object-localization-challenge/LOC_synset_mapping.txt\") as f:\n        image_classes = [line.strip() for line in f.readlines()]\n    text_input = clip.tokenize(image_classes).to(device)\n    \n    image_inputs = []\n    for image in images:\n        image_input = preprocess_clip(image).unsqueeze(0).to(device)\n        image_inputs.append(image_input)\n    image_inputs =torch.cat(image_inputs, dim=0).to(device)\n\n    model_clip.cuda().eval()\n    with torch.no_grad():\n        image_features = model_clip.encode_image(image_inputs)\n        text_features = model_clip.encode_text(text_input)\n        \n        \n    image_features /= image_features.norm(dim=-1, keepdim=True)\n    text_features /= text_features.norm(dim=-1, keepdim=True)\n    similarity = (100.0 * image_features @ text_features.T).softmax(dim=-1)\n    values, indices = similarity.topk(top_k)\n    inc_out = []\n    for i in range(len(values)):\n        if correct_index not in indices[i]:\n            inc_out.append(i)\n          \n    return inc_out\n\n\n# function to predict using image net\nimport torch.nn.functional as F   \ndef predict_ImageNet_m(images, correct_index ,top_k = 5):\n    image_inputs= []\n    weights = ResNet50_Weights.IMAGENET1K_V1\n    preprocess_imagenet = weights.transforms().to(device)\n    for image in images:\n        image = image.convert('RGB')\n        image_input = preprocess_imagenet(image).unsqueeze(0).to(device)\n        image_inputs.append(image_input)\n    image_inputs =torch.cat(image_inputs, dim=0).to(device)\n    \n    \n    model_torch.cuda().eval()\n    with torch.no_grad():\n        probs = model_torch(image_inputs).squeeze(0).softmax(-1)\n    \n    inc_out = []\n    \n    for i in range(len(probs)):\n        sorted_idx = np.argsort(probs[i].cpu().numpy())\n        top_k_indices = sorted_idx[-top_k:][::-1]\n        if correct_index not in top_k_indices:\n            inc_out.append(i)     \n#         values, indices = probs[i].topk(top_k)\n#         if correct_index not in indices:\n#             inc_out.append(i)\n            \n    return inc_out\n \n","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:50:11.688312Z","iopub.execute_input":"2024-04-25T21:50:11.688675Z","iopub.status.idle":"2024-04-25T21:50:11.705199Z","shell.execute_reply.started":"2024-04-25T21:50:11.688644Z","shell.execute_reply":"2024-04-25T21:50:11.703936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\ndirectory_path = '/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/'\n\ndef find_index(folder_name):\n    \n    with open(\"/kaggle/input/imagenet-object-localization-challenge/LOC_synset_mapping.txt\") as f:\n        gt = [line.strip() for line in f.readlines()]\n\n    l =[]\n    for line in gt:\n        l1 = line.split()[0]\n        l.append(l1)\n    \n    for i in range(len(l)):\n        if l[i] == folder_name:\n            return i\n\ndef find_incorrect(folder_name):\n    \n    folder_path = directory_path + str(folder_name) + '/'\n    files = os.listdir(folder_path)\n#     files = sorted(files, key=lambda x: int(x.split(\"_\")[1].split(\".\")[0]))\n    num_samples = len(files)\n    files = files[0:int(0.5*num_samples)]\n    \n    count =0 # used for loop breaking in debugging\n\n    images= []\n    for file in files:\n        count = count +1 \n        image_path = folder_path + str(file)\n        image = Image.open(image_path)\n        images.append(image)\n\n    folder_index = find_index(folder_name)\n    result_image_net = predict_ImageNet_m(images, folder_index)\n    result_clip = predict_CLIP_m(images, folder_index)\n    \n    torch.cuda.empty_cache()\n    \n    return  result_image_net, result_clip\n\n\ndef find_different(folder_name):\n    \n    inc_imagenet , inc_clip = find_incorrect(folder_name)\n    inc_clip_cor_inet = np.setdiff1d(inc_clip,inc_imagenet).tolist()\n    inc_inet_cor_clip = np.setdiff1d(inc_imagenet,inc_clip).tolist()\n    rand_idx_inc_inet_cor_clip = random.sample(inc_inet_cor_clip, 2)\n    rand_idx_inc_clip_cor_inet = random.sample(inc_clip_cor_inet,1)\n    \n    final_idx = [rand_idx_inc_inet_cor_clip[0],rand_idx_inc_inet_cor_clip[1],rand_idx_inc_clip_cor_inet[0]]\n    return final_idx","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:50:11.706647Z","iopub.execute_input":"2024-04-25T21:50:11.707117Z","iopub.status.idle":"2024-04-25T21:50:11.722935Z","shell.execute_reply.started":"2024-04-25T21:50:11.707088Z","shell.execute_reply":"2024-04-25T21:50:11.721614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for establishing ground truth \nwith open(\"/kaggle/input/imagenet-object-localization-challenge/LOC_synset_mapping.txt\") as f:\n    gt = [line.strip() for line in f.readlines()]\n\ngt_dict = {}\n\nfor line in gt:\n    parts = line.split()\n    key = parts[0]\n    t_names = ' '.join(parts[1:]).split(', ')\n    gt_dict[str(key)] = t_names\n","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:50:11.724255Z","iopub.execute_input":"2024-04-25T21:50:11.724651Z","iopub.status.idle":"2024-04-25T21:50:11.735458Z","shell.execute_reply.started":"2024-04-25T21:50:11.724623Z","shell.execute_reply":"2024-04-25T21:50:11.734381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folders_list = ['n07880968','n07760859','n09428293','n02823428', 'n03721384' ,'n03937543', 'n02793495', 'n02708093', 'n04548280','n01843065']\nnum_images = len(folders_list) *3\nnum_cols = 1\nnum_rows = num_images\n\nfig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 5*num_rows))\nrow_counter = 0\n\nfor i in range(len(folders_list)):\n    folder_name = folders_list[i]\n    final_idx = find_different(folder_name)\n    \n    files = os.listdir(directory_path + folder_name + '/')\n    \n    for j in range(3):\n        image_path = directory_path + folder_name + '/' + files[final_idx[j]]\n        image = Image.open(image_path)\n        axes[row_counter].imshow(image)  \n        clip1, clip2 = predict_CLIP(image, 5)\n        i1, i2 = predict_ImageNet(image,5)\n        title = \"Ground truth: \"+ str(gt_dict[folder_name]) +\"CLIP predicted: \" + str(clip1) + \" ImageNet predicted: \" + str(i1) \n        axes[row_counter].set_title(title)\n        row_counter = row_counter +1\nplt.tight_layout()  \nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:50:11.737216Z","iopub.execute_input":"2024-04-25T21:50:11.737541Z","iopub.status.idle":"2024-04-25T21:52:46.834449Z","shell.execute_reply.started":"2024-04-25T21:50:11.737514Z","shell.execute_reply":"2024-04-25T21:52:46.832702Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.5","metadata":{}},{"cell_type":"code","source":"# the default model is 32 fp only\n# to get model with 16 fp, we half the model\n\nmodel_16fp, preprocess_16fp = clip.load('RN50', device)\n\nparams_og = []\nprint(\"Before Halving: \")\nfor params in model_16fp.parameters():\n    params_og.append(str(params.dtype))\nparams ,count = np.unique(params_og, return_counts =True)\nfor i in range(len(params)):\n    print(params[i], count[i])\n    \n# after halving   \nmodel_16fp.half()\nparams_half = []\nprint(\"After Halving: \")\nfor params in model_16fp.parameters():\n    params_half.append(str(params.dtype))\nparams ,count = np.unique(params_half, return_counts =True)\nfor i in range(len(params)):\n    print(params[i], count[i])\n\nmodel_16fp.cuda().eval()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:52:46.836812Z","iopub.execute_input":"2024-04-25T21:52:46.838052Z","iopub.status.idle":"2024-04-25T21:52:50.970065Z","shell.execute_reply.started":"2024-04-25T21:52:46.837977Z","shell.execute_reply":"2024-04-25T21:52:50.969053Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_32fp, preprocess_32fp = clip.load('RN50', device)\nmodel_32fp.cuda().eval()","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:52:50.971583Z","iopub.execute_input":"2024-04-25T21:52:50.972248Z","iopub.status.idle":"2024-04-25T21:52:55.002265Z","shell.execute_reply.started":"2024-04-25T21:52:50.97221Z","shell.execute_reply":"2024-04-25T21:52:55.001213Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\ntest_images_num = 100\n\ntime_taken_16 = np.zeros(test_images_num)\ntime_taken_32 = np.zeros(test_images_num)\n\n# choosing a test_folder\ntest_folder = '/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/n01601694'\nfiles = os.listdir(test_folder)\nrand_files = random.sample(files, test_images_num)\n\n# trying out the model fp 32\nfor i in range(test_images_num):\n    file = rand_files[i]\n    image_file = test_folder+ '/' + str(file)\n    image = Image.open(image_file)\n    start_t = time.time()\n    p_image = preprocess_32fp(image).unsqueeze(0).to(device)\n    with torch.no_grad():\n        output = model_32fp.encode_image(p_image)\n    end_t = time.time()\n    time_dif = end_t - start_t\n    time_taken_32[i] = time_dif\n\n# trying out the model fp 16\nfor i in range(test_images_num):\n    file = rand_files[i]\n    image_file = test_folder+ '/' + str(file)\n    image = Image.open(image_file)\n    start_t = time.time()\n    p_image = preprocess_16fp(image).unsqueeze(0).to(device).half()\n    with torch.no_grad():\n        output = model_16fp.encode_image(p_image)\n    end_t = time.time()\n    time_dif = end_t - start_t\n    time_taken_16[i] = time_dif\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:52:55.003848Z","iopub.execute_input":"2024-04-25T21:52:55.004559Z","iopub.status.idle":"2024-04-25T21:52:59.435839Z","shell.execute_reply.started":"2024-04-25T21:52:55.004521Z","shell.execute_reply":"2024-04-25T21:52:59.434831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(time_taken_16 , label='16fp')\nplt.plot(time_taken_32, label = '32fp')\nplt.legend()\nplt.title('Comparison of sample wise time taken')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-25T22:45:41.538191Z","iopub.execute_input":"2024-04-25T22:45:41.538612Z","iopub.status.idle":"2024-04-25T22:45:41.898189Z","shell.execute_reply.started":"2024-04-25T22:45:41.53858Z","shell.execute_reply":"2024-04-25T22:45:41.896829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ndf_16 = pd.DataFrame(time_taken_16)\ndf_32 = pd.DataFrame(time_taken_32)\n\nprint(\"Metrics for 16fp and 32fp model are as follows:\")\ndf_16.describe(), df_32.describe()","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:53:42.216726Z","iopub.execute_input":"2024-04-25T21:53:42.217127Z","iopub.status.idle":"2024-04-25T21:53:42.236007Z","shell.execute_reply.started":"2024-04-25T21:53:42.217098Z","shell.execute_reply":"2024-04-25T21:53:42.2349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predicting with half model\n\nwith open(\"/kaggle/input/imagenet-object-localization-challenge/LOC_synset_mapping.txt\") as f:\n    image_classes = [line.strip() for line in f.readlines()]\ntext_input = clip.tokenize(image_classes).to(device)\n    \ndef predict_16(image, top_k =1):\n    image_input = preprocess_16fp(image).unsqueeze(0).to(device).half()\n    \n    model_16fp.cuda().eval()\n    with torch.no_grad():\n        image_features = model_16fp.encode_image(image_input)\n        text_features = model_32fp.encode_text(text_input)# is same in both\n        \n        \n    image_features /= image_features.norm(dim=-1, keepdim=True)\n    text_features /= text_features.norm(dim=-1, keepdim=True)\n    similarity = (100.0 * image_features @ text_features.T).softmax(dim=-1)\n    values, indices = similarity[0].topk(top_k)\n#     print(values, indices)\n    \n    if top_k ==1:\n        temp = image_classes[indices]\n        class_parts = temp.split(', ')\n        class_names = []\n        for part in class_parts:\n            words = part.split(' ')\n            if len(words) > 1:\n                class_names.append(' '.join(words[1:]))  # Join the words after the first one with space\n            else:\n                class_names.append(words[0])  # If there's only one word, add it as it is\n\n        predicted_class = str(class_names)\n    else:\n        predicted_class = []\n        for i in indices:\n            temp = image_classes[i]\n            class_parts = temp.split(',')\n            class_names = []\n            for part in class_parts:\n                words = part.split(' ')\n                if len(words) > 1:\n                    class_names.append(' '.join(words[1:]))  # Join the words after the first one with space\n                else:\n                    class_names.append(words[0])  # If there's only one word, add it as it is\n\n                    class_name = temp.split(',')[0].split(' ', 1)[1].strip()\n            predicted_class.append(class_names)\n        \n    \n    return predicted_class, values\n\n\n \n","metadata":{"execution":{"iopub.status.busy":"2024-04-25T22:33:43.802861Z","iopub.execute_input":"2024-04-25T22:33:43.803565Z","iopub.status.idle":"2024-04-25T22:33:43.989019Z","shell.execute_reply.started":"2024-04-25T22:33:43.803529Z","shell.execute_reply":"2024-04-25T22:33:43.98808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predicting with full model\n\nwith open(\"/kaggle/input/imagenet-object-localization-challenge/LOC_synset_mapping.txt\") as f:\n    image_classes = [line.strip() for line in f.readlines()]\ntext_input = clip.tokenize(image_classes).to(device)\n    \ndef predict_32(image, top_k =1):\n    image_input = preprocess_32fp(image).unsqueeze(0).to(device)\n    \n    model_32fp.cuda().eval()\n    with torch.no_grad():\n        image_features = model_32fp.encode_image(image_input)\n        text_features = model_32fp.encode_text(text_input)\n        \n        \n    image_features /= image_features.norm(dim=-1, keepdim=True)\n    text_features /= text_features.norm(dim=-1, keepdim=True)\n    similarity = (100.0 * image_features @ text_features.T).softmax(dim=-1)\n    values, indices = similarity[0].topk(top_k)\n#     print(values, indices)\n    \n    if top_k ==1:\n        temp = image_classes[indices]\n        class_parts = temp.split(', ')\n        class_names = []\n        for part in class_parts:\n            words = part.split(' ')\n            if len(words) > 1:\n                class_names.append(' '.join(words[1:]))  # Join the words after the first one with space\n            else:\n                class_names.append(words[0])  # If there's only one word, add it as it is\n\n        predicted_class = str(class_names)\n    else:\n        predicted_class = []\n        for i in indices:\n            temp = image_classes[i]\n            class_parts = temp.split(',')\n            class_names = []\n            for part in class_parts:\n                words = part.split(' ')\n                if len(words) > 1:\n                    class_names.append(' '.join(words[1:]))  # Join the words after the first one with space\n                else:\n                    class_names.append(words[0])  # If there's only one word, add it as it is\n\n                    class_name = temp.split(',')[0].split(' ', 1)[1].strip()\n            predicted_class.append(class_names)\n        \n    \n    return predicted_class, values\n\n\n \n","metadata":{"execution":{"iopub.status.busy":"2024-04-25T22:33:44.962502Z","iopub.execute_input":"2024-04-25T22:33:44.963462Z","iopub.status.idle":"2024-04-25T22:33:45.154493Z","shell.execute_reply.started":"2024-04-25T22:33:44.963422Z","shell.execute_reply":"2024-04-25T22:33:45.153437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classes = ['/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/n01580077/n01580077_1000.JPEG','/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/n01484850/n01484850_10016.JPEG','/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/n01514859/n01514859_1.JPEG','/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/n01622779/n01622779_10.JPEG','/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train/n01601694/n01601694_10034.JPEG']\n\nnum_rows = len(classes)\nnum_cols = 1\n\nfig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 5*num_rows))\n\nfor i in range(len(classes)):\n    image_path = classes[i]\n    image = Image.open(image_path)\n    image_name = image_path.split('/')[-1]\n    image_name = image_name.split('_')[0]\n    gt_class = gt_dict[image_name]\n    c16, p16 = predict_16(image)\n    c32, p32 = predict_32(image)\n    axes[i].imshow(image)\n    axes[i].set_title(\"Ground Truth is: \" + str(gt_class) + \"FP16 model top predicted is: \" + str(c16) + \" with probablity: \" + str(p16.item()) + \" FP32 model top predicted is: \" + str(c32) + \" with probablity: \" + str(p32.item()))\n    \nplt.tight_layout()\nplt.show()\n    ","metadata":{"execution":{"iopub.status.busy":"2024-04-25T22:39:19.349078Z","iopub.execute_input":"2024-04-25T22:39:19.349574Z","iopub.status.idle":"2024-04-25T22:39:28.078063Z","shell.execute_reply.started":"2024-04-25T22:39:19.34954Z","shell.execute_reply":"2024-04-25T22:39:28.076677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# nvidia smi\nimport subprocess\n\ndef monitor_gpu_memory():\n    result = subprocess.run(['nvidia-smi'], stdout=subprocess.PIPE)\n    output = result.stdout.decode('utf-8')\n    return output\n\nimage = Image.open(classes[0])\n\nprint(\"FP16 model\")\ntorch.cuda.empty_cache()\n# process2_5(trainPath, newClasses[-1], half=True)\npredict_16(image)\nprint(\"Memory usage after training FP16 model:\")\nprint(monitor_gpu_memory())\n\nprint(\"FP32 model\")\ntorch.cuda.empty_cache()\npredict_32(image)\nprint(\"Memory usage after training FP32 model:\")\nprint(monitor_gpu_memory())","metadata":{"execution":{"iopub.status.busy":"2024-04-25T22:48:19.48824Z","iopub.execute_input":"2024-04-25T22:48:19.489226Z","iopub.status.idle":"2024-04-25T22:48:20.992587Z","shell.execute_reply.started":"2024-04-25T22:48:19.489188Z","shell.execute_reply":"2024-04-25T22:48:20.99157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom torch.profiler import profile, record_function, ProfilerActivity\n\nimage = Image.open(classes[0])\ndef my_model16():\n#     process2_5(trainPath, newClasses[-1], half=True)\n    predict_16(image)\n\ndef my_model32():\n#     process2_5(trainPath, newClasses[-1], half=False)\n    predict_32(image)\n    \n# Enable the memory profiler and run the code for FP16 model\ntorch.cuda.empty_cache()\nwith profile(activities=[ProfilerActivity.CUDA], record_shapes=True) as prof:\n    with record_function(\"model_inference\"):\n        my_model16()\n\nprint(\"For FP16 Model:\")\nprint(prof.key_averages().table(sort_by=\"self_cuda_memory_usage\", row_limit=10))\n\n# Enable the memory profiler and run the code for FP32 model\ntorch.cuda.empty_cache()\nwith profile(activities=[ProfilerActivity.CUDA], record_shapes=True) as prof:\n    with record_function(\"model_inference\"):\n        my_model32()\n\nprint(\"For FP32 Model:\")\nprint(prof.key_averages().table(sort_by=\"self_cuda_memory_usage\", row_limit=10))\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-25T22:58:39.822653Z","iopub.execute_input":"2024-04-25T22:58:39.82313Z","iopub.status.idle":"2024-04-25T22:58:41.440138Z","shell.execute_reply.started":"2024-04-25T22:58:39.823094Z","shell.execute_reply":"2024-04-25T22:58:41.439022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}