{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# importing libraries used in this notebook\n\nfrom keras.preprocessing import image\nfrom keras.applications.vgg16 import VGG16\nfrom keras.applications.vgg16 import preprocess_input\nimport numpy as np\nfrom sklearn.cluster import KMeans\nimport os, shutil, glob, os.path\nfrom PIL import Image\nimage.LOAD_TRUNCATED_IMAGES = True\nfrom tqdm.notebook import tqdm \nimport glob\nimport math\nimport pickle\n%pylab inline\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# command to download the dataset given in the problem statement\n!wget --header=\"Host: doc-0g-2k-docs.googleusercontent.com\" --header=\"User-Agent: Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/84.0.4147.105 Safari/537.36\" --header=\"Accept: text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.9\" --header=\"Accept-Language: en-US,en;q=0.9\" --header=\"Referer: https://drive.google.com/\" --header=\"Cookie: AUTH_t749vl4skj7cn551ncp80lljbg5da3jf_nonce=49nkonhl8kcc2\" --header=\"Connection: keep-alive\" \"https://doc-0g-2k-docs.googleusercontent.com/docs/securesc/1s3jo9hqchijmcbr6j69dtugek0datcv/9e7mrnehkvsggebq8hvgcd27sm2mu26p/1596822900000/07496480791912752493/12633858806009087463/1VT-8w1rTT2GCE5IE5zFJPMzv7bqca-Ri?e=download&authuser=0&nonce=49nkonhl8kcc2&user=12633858806009087463&hash=m498o18dip5brnhau0oqhdkbcc1idmp7\" -c -O 'dataset.zip'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# unziping the dataset folder\n!unzip dataset.zip ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# location where the dataset is downloaded\nout_dir = '/kaggle/working/dataset'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# storing the path of all the jpj file in filelist from our dataset\nfilelist = glob.glob(os.path.join(out_dir,'*.jpg'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# sorting the path of our images\nfilelist.sort()\nfeaturelist = []","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#printing the path of first five images\nfilelist[0:5]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# using weights of vgg16 model trained on imagenet data for featurization\nmodel = VGG16(weights='imagenet', include_top=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i, imagepath in tqdm(enumerate(filelist)):\n    img = image.load_img(imagepath)\n    img_data = image.img_to_array(img)\n    img_data = np.expand_dims(img_data, axis=0)\n    img_data = preprocess_input(img_data)\n    features = np.array(model.predict(img_data))\n    featurelist.append(features.flatten())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# dumping the features of our images in a pickle file so that it can be used for further use\nwith open('featurelist', 'wb') as f: \n    pickle.dump(featurelist, f)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#loading the featurelist that consist of bottleneck features of all the images\nwith open('/kaggle/working/featurelist', 'rb') as f: \n    featurelist = pickle.load(f) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# printing lenght of filelist\nlen(filelist)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# printing lenght of featurelist\nlen(featurelist)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#sorting the path of jpg images\nfilelist.sort()\nfilelist[0:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# loading the weights of vgg16 model that is train on imagenet data that consists of 1000 classes\nmodel = VGG16(weights='imagenet', include_top=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# function that return the weight of a vector\n\ndef square_rooted(x):\n    return math.sqrt(sum([a*a for a in x]))\n\n\n# function to return the cosine similarity between two vectors\n\ndef cosine_similarity(x,y):\n    numerator = sum(a*b for a,b in zip(x,y))\n    denominator = square_rooted(x)*square_rooted(y)\n    return numerator/float(denominator)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# function to find cosine similarity of a given image with all the images in the dataset.\n\ndef find_similar(input_image_path,input_features,filelist):\n    for i in tqdm(range(len(filelist))):\n        if filelist[i]== input_image_path:\n            continue\n        dic_store[filelist[i]] = cosine_similarity(input_features,featurelist[i])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# function to return the bottleneck feature of a given input image\n\ndef input_features(input_image_path):\n    img = image.load_img(input_image_path)\n    img_data = image.img_to_array(img)\n    img_data = np.expand_dims(img_data, axis=0)\n    img_data = preprocess_input(img_data)\n    features = np.array(model.predict(img_data))  \n    return features.flatten()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# function to represent the images as described in the problem statement\n\ndef print_images(N,input_image_path,sorted_x):\n    columns = 3\n    rows = ceil(N/3) +1\n    fig=plt.figure(figsize=(8, 8))\n    input_image = mpimg.imread(input_image_path)\n    fig.add_subplot(rows, columns, 1)\n    plt.imshow(input_image)\n\n    i=3\n    for key in sorted_x:\n        i+=1\n        if(i>N+3):\n            break\n        fig.add_subplot(rows, columns, i)\n        img=mpimg.imread(key[0])\n        imgplot = plt.imshow(img)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# finding the cosine similarity between our image 3 and image 5\n\ncosine_similarity(featurelist[2],featurelist[5]) # for demonstration","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# dictionary to store the cosine similarity of input image with other images\ndic_store = {}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# image for which you have to find similar images \n\ninput_image_path = '/kaggle/working/dataset/1000.jpg' # write image path","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# diaplaying input image\nim = Image.open(input_image_path)\nim","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# will return input features of the input image\nfeatures = input_features(input_image_path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# will find the cosine similarity of input image with other images\n\nfind_similar(input_image_path,features,filelist)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# sorting dictionary in descending order on values\nsorted_x = sorted(dic_store.items(), key=lambda kv: kv[1],reverse = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# printing top ten cosine similaity values and the images path\nsorted_x[0:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# define the number of simailar images you want\nN = 10 #you can take any number","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# will return the N images similar to input image\nprint_images(N,input_image_path,sorted_x)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}