{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Product Similarity # Results\n\nIn this notebook we are going to use the embeddings extracted from [[H&M] Product Similarity #1 Embeddings&KNN](https://www.kaggle.com/joelqv/h-m-product-similarity-1-embeddings-knn) to retrieve similar products.","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport joblib\n\ndf_articles = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')\nknn = joblib.load('../input/h-m-product-similarity-1-embeddings-knn/knn.joblib')\nimage_embeddings = np.load('../input/h-m-product-similarity-1-embeddings-knn/hm_embeddings_effb0.npy')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-16T15:20:30.978836Z","iopub.execute_input":"2022-02-16T15:20:30.979408Z","iopub.status.idle":"2022-02-16T15:20:56.270605Z","shell.execute_reply.started":"2022-02-16T15:20:30.979285Z","shell.execute_reply":"2022-02-16T15:20:56.269069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\ndef get_article_images_df(path='../input/h-and-m-personalized-fashion-recommendations/images'):\n    article_ids = []\n    image_paths = []\n    for dirname, _, filenames in os.walk(path):\n        for filename in filenames:\n            fullpath = os.path.join(dirname, filename)\n            image_path = fullpath\n            article_id = fullpath.split('/')[-1].replace('.jpg', '')\n            article_ids.append(article_id)\n            image_paths.append(fullpath)\n    return pd.DataFrame({'article_id': article_ids, 'image': image_paths})","metadata":{"execution":{"iopub.status.busy":"2022-02-16T15:20:56.277681Z","iopub.execute_input":"2022-02-16T15:20:56.278606Z","iopub.status.idle":"2022-02-16T15:20:56.286424Z","shell.execute_reply.started":"2022-02-16T15:20:56.278568Z","shell.execute_reply":"2022-02-16T15:20:56.285584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = get_article_images_df()","metadata":{"execution":{"iopub.status.busy":"2022-02-16T15:20:56.287575Z","iopub.execute_input":"2022-02-16T15:20:56.288402Z","iopub.status.idle":"2022-02-16T15:22:03.368066Z","shell.execute_reply.started":"2022-02-16T15:20:56.288358Z","shell.execute_reply":"2022-02-16T15:22:03.367344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\n\ndef compute_distances(df, idx, model, knn):\n    \"\"\"\n    Returns distances indices of most similar products based on embeddings extracted from model\n    \"\"\"\n    X = np.zeros((1, 256, 256, 3), dtype='float32')\n    img = cv2.imread(df.iloc[idx].image)    # TODO: cv2.readfrombinary\n    img = cv2.resize(img, (256, 256))\n    X[0,] = img\n    model = EfficientNetB0(weights='imagenet', include_top=False, pooling='avg', input_shape=None)\n    inf_embeddings = model.predict(X, verbose=1)\n    distances, indices = knn.kneighbors(inf_embeddings)\n    return distances, indices\n\ndef plot_results(df, indices, distances, col_size=3, row_size=4):\n    fig, axs = plt.subplots(col_size, row_size, figsize=(20, 15))\n    axs = axs.flatten()\n    i = 0\n    for ax, idx in zip(axs, indices[0]):\n        if i == 0: \n            ax.set_title('Query image')\n        else: \n            ax.set_title(f'{i} most similar')\n        img = cv2.imread(df.iloc[idx].image)\n        #img = cv2.resize(img, (256, 256))\n        im_bgr = cv2.cvtColor(img, cv2.COLOR_RGB2BGR)\n        ax.axis('off')\n        ax.imshow(im_bgr, aspect='auto')\n        i+=1\n    fig.suptitle('Similar products', fontsize=36)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T15:37:31.296756Z","iopub.execute_input":"2022-02-16T15:37:31.297042Z","iopub.status.idle":"2022-02-16T15:37:31.306518Z","shell.execute_reply.started":"2022-02-16T15:37:31.297012Z","shell.execute_reply":"2022-02-16T15:37:31.305691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetB0\n\nmodel = EfficientNetB0(weights='imagenet', include_top=False, pooling='avg', input_shape=None)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T15:22:03.814689Z","iopub.execute_input":"2022-02-16T15:22:03.814990Z","iopub.status.idle":"2022-02-16T15:22:12.527426Z","shell.execute_reply.started":"2022-02-16T15:22:03.814945Z","shell.execute_reply":"2022-02-16T15:22:12.526346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"distances, indices = compute_distances(df, 0, model, knn)\nplot_results(df, indices, distances)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T15:39:13.950182Z","iopub.execute_input":"2022-02-16T15:39:13.950726Z","iopub.status.idle":"2022-02-16T15:39:24.899674Z","shell.execute_reply.started":"2022-02-16T15:39:13.950677Z","shell.execute_reply":"2022-02-16T15:39:24.898871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"distances, indices = compute_distances(df, 100, model, knn)\nplot_results(df, indices, distances)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T15:39:24.900883Z","iopub.execute_input":"2022-02-16T15:39:24.901103Z","iopub.status.idle":"2022-02-16T15:39:34.151794Z","shell.execute_reply.started":"2022-02-16T15:39:24.901077Z","shell.execute_reply":"2022-02-16T15:39:34.150908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"distances, indices = compute_distances(df, 1000, model, knn)\nplot_results(df, indices, distances)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T15:39:43.064242Z","iopub.execute_input":"2022-02-16T15:39:43.064522Z","iopub.status.idle":"2022-02-16T15:39:51.891868Z","shell.execute_reply.started":"2022-02-16T15:39:43.064491Z","shell.execute_reply":"2022-02-16T15:39:51.891052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"distances, indices = compute_distances(df, 5000, model, knn)\nplot_results(df, indices, distances)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T21:53:30.326336Z","iopub.execute_input":"2022-02-15T21:53:30.32721Z","iopub.status.idle":"2022-02-15T21:53:35.775258Z","shell.execute_reply.started":"2022-02-15T21:53:30.327174Z","shell.execute_reply":"2022-02-15T21:53:35.774354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"distances, indices = compute_distances(df, 10000, model, knn)\nplot_results(df, indices, distances)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T21:53:45.214259Z","iopub.execute_input":"2022-02-15T21:53:45.214576Z","iopub.status.idle":"2022-02-15T21:53:49.744394Z","shell.execute_reply.started":"2022-02-15T21:53:45.214545Z","shell.execute_reply":"2022-02-15T21:53:49.742963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"distances, indices = compute_distances(df, 20000, model, knn)\nplot_results(df, indices, distances)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T21:53:57.844326Z","iopub.execute_input":"2022-02-15T21:53:57.844577Z","iopub.status.idle":"2022-02-15T21:54:02.90453Z","shell.execute_reply.started":"2022-02-15T21:53:57.844552Z","shell.execute_reply":"2022-02-15T21:54:02.903095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"distances, indices = compute_distances(df, 30000, model, knn)\nplot_results(df, indices, distances)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T21:54:10.740921Z","iopub.execute_input":"2022-02-15T21:54:10.741198Z","iopub.status.idle":"2022-02-15T21:54:16.277887Z","shell.execute_reply.started":"2022-02-15T21:54:10.741167Z","shell.execute_reply":"2022-02-15T21:54:16.276712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"distances, indices = compute_distances(df, 40000, model, knn)\nplot_results(df, indices, distances)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T21:54:22.936953Z","iopub.execute_input":"2022-02-15T21:54:22.93761Z","iopub.status.idle":"2022-02-15T21:54:28.717922Z","shell.execute_reply.started":"2022-02-15T21:54:22.937539Z","shell.execute_reply":"2022-02-15T21:54:28.716598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"distances, indices = compute_distances(df, 50000, model, knn)\nplot_results(df, indices, distances)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T21:58:32.038767Z","iopub.execute_input":"2022-02-15T21:58:32.039014Z","iopub.status.idle":"2022-02-15T21:58:37.233795Z","shell.execute_reply.started":"2022-02-15T21:58:32.03899Z","shell.execute_reply":"2022-02-15T21:58:37.233325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"distances, indices = compute_distances(df, 60000, model, knn)\nplot_results(df, indices, distances)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T21:54:35.737378Z","iopub.execute_input":"2022-02-15T21:54:35.737657Z","iopub.status.idle":"2022-02-15T21:54:42.538048Z","shell.execute_reply.started":"2022-02-15T21:54:35.737624Z","shell.execute_reply":"2022-02-15T21:54:42.537461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"distances, indices = compute_distances(df, 70000, model, knn)\nplot_results(df, indices, distances)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T21:54:58.518659Z","iopub.execute_input":"2022-02-15T21:54:58.518953Z","iopub.status.idle":"2022-02-15T21:55:04.189523Z","shell.execute_reply.started":"2022-02-15T21:54:58.51892Z","shell.execute_reply":"2022-02-15T21:55:04.188353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"distances, indices = compute_distances(df, 80000, model, knn)\nplot_results(df, indices, distances)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T21:55:10.446058Z","iopub.execute_input":"2022-02-15T21:55:10.446363Z","iopub.status.idle":"2022-02-15T21:55:17.024998Z","shell.execute_reply.started":"2022-02-15T21:55:10.446334Z","shell.execute_reply":"2022-02-15T21:55:17.023824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"distances, indices = compute_distances(df, 90000, model, knn)\nplot_results(df, indices, distances)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T21:55:26.912553Z","iopub.execute_input":"2022-02-15T21:55:26.91303Z","iopub.status.idle":"2022-02-15T21:55:32.538649Z","shell.execute_reply.started":"2022-02-15T21:55:26.912995Z","shell.execute_reply":"2022-02-15T21:55:32.537448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"distances, indices = compute_distances(df, 100000, model, knn)\nplot_results(df, indices, distances)","metadata":{"execution":{"iopub.status.busy":"2022-02-15T21:55:43.309185Z","iopub.execute_input":"2022-02-15T21:55:43.309744Z","iopub.status.idle":"2022-02-15T21:55:47.800705Z","shell.execute_reply.started":"2022-02-15T21:55:43.309709Z","shell.execute_reply":"2022-02-15T21:55:47.799689Z"},"trusted":true},"execution_count":null,"outputs":[]}]}