{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":31254,"databundleVersionId":3103714},{"sourceType":"datasetVersion","sourceId":7993454,"datasetId":4695582,"databundleVersionId":8104645},{"sourceType":"datasetVersion","sourceId":7978648,"datasetId":4695680,"databundleVersionId":8089159}],"dockerImageVersionId":30674,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from keras.applications import vgg19\nfrom keras.utils import load_img,img_to_array\nfrom keras.models import Model\nfrom keras.applications.imagenet_utils import preprocess_input\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nfrom PIL import Image\nimport os\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom sklearn.metrics.pairwise import cosine_similarity\nimport pandas as pd\nimport time\n\nimport cv2","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-01T00:20:29.005056Z","iopub.execute_input":"2024-04-01T00:20:29.005493Z","iopub.status.idle":"2024-04-01T00:20:29.011709Z","shell.execute_reply.started":"2024-04-01T00:20:29.005461Z","shell.execute_reply":"2024-04-01T00:20:29.010702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs_path = \"/kaggle/input/images/\"\nimgs_model_width, imgs_model_height = 224, 224","metadata":{"execution":{"iopub.status.busy":"2024-04-01T00:20:29.013259Z","iopub.execute_input":"2024-04-01T00:20:29.013571Z","iopub.status.idle":"2024-04-01T00:20:29.035088Z","shell.execute_reply.started":"2024-04-01T00:20:29.013548Z","shell.execute_reply":"2024-04-01T00:20:29.034138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load the model\nvgg_model = vgg19.VGG19(weights='imagenet')\n\n# remove the last layers in order to get features instead of predictions\nfeat_extractor = Model(inputs=vgg_model.input, outputs=vgg_model.get_layer(\"fc2\").output)\n\n# print the layers of the CNN\nfeat_extractor.summary()","metadata":{"execution":{"iopub.status.busy":"2024-04-01T00:20:29.036203Z","iopub.execute_input":"2024-04-01T00:20:29.036494Z","iopub.status.idle":"2024-04-01T00:20:31.322310Z","shell.execute_reply.started":"2024-04-01T00:20:29.036472Z","shell.execute_reply":"2024-04-01T00:20:31.321413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files = [imgs_path + x for x in os.listdir(imgs_path) if \"jpg\" in x]\n\nprint(\"number of images:\",len(files))\n\nimport random\nrandom.shuffle(files)","metadata":{"execution":{"iopub.status.busy":"2024-04-01T00:20:31.324652Z","iopub.execute_input":"2024-04-01T00:20:31.325389Z","iopub.status.idle":"2024-04-01T00:20:31.334684Z","shell.execute_reply.started":"2024-04-01T00:20:31.325362Z","shell.execute_reply":"2024-04-01T00:20:31.333858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load all the images and prepare them for feeding into the CNN\n\nimportedImages = []\n\nfor f in files:\n    filename = f\n    original = load_img(filename, target_size=(224, 224))\n    numpy_image = img_to_array(original)\n    image_batch = np.expand_dims(numpy_image, axis=0)\n    \n    importedImages.append(image_batch)\n    \nimages = np.vstack(importedImages)\n\nprocessed_imgs = preprocess_input(images.copy())\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-01T00:20:31.335725Z","iopub.execute_input":"2024-04-01T00:20:31.335979Z","iopub.status.idle":"2024-04-01T00:20:31.433196Z","shell.execute_reply.started":"2024-04-01T00:20:31.335957Z","shell.execute_reply":"2024-04-01T00:20:31.432262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# extract the images features\n\nimgs_features = feat_extractor.predict(processed_imgs)\n\n\n\nprint(\"features successfully extracted!\")\nimgs_features.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-01T00:20:31.435777Z","iopub.execute_input":"2024-04-01T00:20:31.436077Z","iopub.status.idle":"2024-04-01T00:20:32.137225Z","shell.execute_reply.started":"2024-04-01T00:20:31.436054Z","shell.execute_reply":"2024-04-01T00:20:32.136295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"time_ = []","metadata":{"execution":{"iopub.status.busy":"2024-04-01T00:20:32.138696Z","iopub.execute_input":"2024-04-01T00:20:32.138996Z","iopub.status.idle":"2024-04-01T00:20:32.142756Z","shell.execute_reply.started":"2024-04-01T00:20:32.138972Z","shell.execute_reply":"2024-04-01T00:20:32.141851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to retrieve the most similar products for a given one\n\ndef retrieve_most_similar_products(given_img):\n    start = time.time()\n    print(\"-----------------------------------------------------------------------\")\n    print(\"original product:\")\n\n    original = load_img(given_img, target_size=(imgs_model_width, imgs_model_height))\n    plt.imshow(original)\n    plt.show()\n    \n    importedImage = []\n    test = load_img(given_img, target_size=(224, 224))\n    numpy_image = img_to_array(test)\n    image_batch = np.expand_dims(numpy_image, axis=0)\n    importedImage.append(image_batch)\n    image = np.vstack(importedImage)\n    processed_img = preprocess_input(image.copy())\n    img_features = feat_extractor.predict(processed_img)\n    test = np.vstack([imgs_features, img_features])\n    \n    files_copy = files.copy()\n    files_copy.append(given_img)\n    cosSimilarities = cosine_similarity(test)\n    cos_similarities_df = pd.DataFrame(cosSimilarities, columns=files_copy, index=files_copy)\n\n    print(\"-----------------------------------------------------------------------\")\n    print(\"most similar products:\")\n\n    closest_imgs = cos_similarities_df[given_img]\n    # Assuming closest_imgs is a DataFrame with two columns (image paths and scores)\n    \n    # Reset the column names to make them unique\n    closest_imgs.columns = ['image_path', 'score']\n    \n    # Sort the DataFrame by the scores in descending order\n    sorted_imgs = closest_imgs.sort_values(by='score', ascending=False)\n    \n    sorted_imgs = sorted_imgs[sorted_imgs['image_path'] != given_img]\n    # Get the top five rows\n    top_five = sorted_imgs.head(5)\n    \n    # Display the result\n    print(top_five)\n\n    \n    time_.append(time.time() - start)","metadata":{"execution":{"iopub.status.busy":"2024-04-01T00:20:32.143919Z","iopub.execute_input":"2024-04-01T00:20:32.144175Z","iopub.status.idle":"2024-04-01T00:20:32.156469Z","shell.execute_reply.started":"2024-04-01T00:20:32.144140Z","shell.execute_reply":"2024-04-01T00:20:32.155790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"retrieve_most_similar_products(\"/kaggle/input/images/cropped_image_1.jpg\")","metadata":{"execution":{"iopub.status.busy":"2024-04-01T00:21:46.315872Z","iopub.execute_input":"2024-04-01T00:21:46.316784Z","iopub.status.idle":"2024-04-01T00:21:46.729782Z","shell.execute_reply.started":"2024-04-01T00:21:46.316753Z","shell.execute_reply":"2024-04-01T00:21:46.728915Z"},"trusted":true},"execution_count":null,"outputs":[]}]}