{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Overview\n\n- Goal: Create clustering based on image, and compare with the category provided from articles table. How good is the category to group similar items together? Is there a new category of similar item that is not tagged by the categorization?\n- The first part of this notebook was taken from this notebook https://www.kaggle.com/hamditarek/similar-image-cnn-cosine-similarity, and then extended to look into clustering. \n- Side output of this notebook is to export the feature vector csv, so that others can immediately load from there without having to do the transformation","metadata":{}},{"cell_type":"code","source":"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Loading and Feature Generation\n- This step is taken from https://www.kaggle.com/hamditarek/similar-image-cnn-cosine-similarity","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom sklearn.utils import shuffle\nfrom sklearn.preprocessing import LabelBinarizer\nfrom keras.applications.xception import Xception,preprocess_input\nimport tensorflow as tf\nfrom keras.preprocessing import image\nfrom keras.layers import Input\nfrom keras.backend import reshape\nfrom sklearn.neighbors import NearestNeighbors\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-03-03T12:04:34.072332Z","iopub.execute_input":"2022-03-03T12:04:34.073205Z","iopub.status.idle":"2022-03-03T12:04:40.494307Z","shell.execute_reply.started":"2022-03-03T12:04:34.073071Z","shell.execute_reply":"2022-03-03T12:04:40.493569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_IMAGE = 10000\n\nimages_dir = '../input/h-and-m-personalized-fashion-recommendations/images'\n\ndef getImagePaths(path):\n    \"\"\"\n    Function to Combine Directory Path with individual Image Paths\n    \n    parameters: path(string) - Path of directory\n    returns: image_names(string) - Full Image Path\n    \"\"\"\n    image_names = []\n    for dirname, _, filenames in os.walk(path):\n        for filename in filenames:\n            fullpath = os.path.join(dirname, filename)\n            image_names.append(fullpath)\n    return image_names\n\ndef preprocess_img(img_path):\n    dsize = (225,225)\n    new_image=cv2.imread(img_path)\n    new_image=cv2.resize(new_image,dsize,interpolation=cv2.INTER_NEAREST)  \n    new_image=np.expand_dims(new_image,axis=0)\n    new_image=preprocess_input(new_image)\n    return new_image\n\ndef load_data():\n    output=[]\n    output=getImagePaths(images_dir)[:N_IMAGE]\n    return output\n\ndef model():\n    model=Xception(weights='imagenet',include_top=False)\n    for layer in model.layers:\n        layer.trainable=False\n        #model.summary()\n    return model\n\ndef feature_extraction(image_data,model):\n    features=model.predict(image_data)\n    features=np.array(features)\n    features=features.flatten()\n    return features","metadata":{"execution":{"iopub.status.busy":"2022-03-03T12:04:40.495967Z","iopub.execute_input":"2022-03-03T12:04:40.496405Z","iopub.status.idle":"2022-03-03T12:04:40.505185Z","shell.execute_reply.started":"2022-03-03T12:04:40.496371Z","shell.execute_reply":"2022-03-03T12:04:40.504407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features=[]\noutput=load_data()\nmain_model=model()\n#Limiting the data for training\nfor i in output[:N_IMAGE-1]:\n    new_img=preprocess_img(i)\n    features.append(feature_extraction(new_img,main_model))\nfeature_vec = np.array(features)","metadata":{"execution":{"iopub.status.busy":"2022-03-03T12:04:40.506188Z","iopub.execute_input":"2022-03-03T12:04:40.506408Z","iopub.status.idle":"2022-03-03T12:37:19.973359Z","shell.execute_reply.started":"2022-03-03T12:04:40.506383Z","shell.execute_reply":"2022-03-03T12:37:19.972328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfFeatures = pd.DataFrame(feature_vec)","metadata":{"execution":{"iopub.status.busy":"2022-03-03T12:37:19.975423Z","iopub.execute_input":"2022-03-03T12:37:19.976172Z","iopub.status.idle":"2022-03-03T12:37:19.981543Z","shell.execute_reply.started":"2022-03-03T12:37:19.976128Z","shell.execute_reply":"2022-03-03T12:37:19.980927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfFeatures.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T12:37:19.982603Z","iopub.execute_input":"2022-03-03T12:37:19.982846Z","iopub.status.idle":"2022-03-03T12:37:20.032596Z","shell.execute_reply.started":"2022-03-03T12:37:19.982816Z","shell.execute_reply":"2022-03-03T12:37:20.031652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get article_id and merge back to articles table","metadata":{}},{"cell_type":"code","source":"def getImagePaths_articleID(path):\n    \"\"\"\n    Function to Combine Directory Path with individual Image Paths\n    \n    parameters: path(string) - Path of directory\n    returns: image_names(string) - Full Image Path\n    \"\"\"\n    list_article_id = []\n    for dirname, _, filenames in os.walk(path):\n        for filename in filenames:\n            list_article_id.append(filename)\n    return list_article_id","metadata":{"execution":{"iopub.status.busy":"2022-03-03T12:37:20.033971Z","iopub.execute_input":"2022-03-03T12:37:20.034420Z","iopub.status.idle":"2022-03-03T12:37:20.039294Z","shell.execute_reply.started":"2022-03-03T12:37:20.034389Z","shell.execute_reply":"2022-03-03T12:37:20.038380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_article_id = getImagePaths_articleID(images_dir)","metadata":{"execution":{"iopub.status.busy":"2022-03-03T12:37:20.040782Z","iopub.execute_input":"2022-03-03T12:37:20.041637Z","iopub.status.idle":"2022-03-03T12:37:33.788760Z","shell.execute_reply.started":"2022-03-03T12:37:20.041588Z","shell.execute_reply":"2022-03-03T12:37:33.787962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfFeatures['article_id'] = [x[1:-4] for x in list_article_id[:N_IMAGE-1]]\ndfFeatures.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T12:37:33.790473Z","iopub.execute_input":"2022-03-03T12:37:33.791059Z","iopub.status.idle":"2022-03-03T12:37:33.859781Z","shell.execute_reply.started":"2022-03-03T12:37:33.791011Z","shell.execute_reply":"2022-03-03T12:37:33.859170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfFeatures.to_csv('dfFeatures.csv')","metadata":{"execution":{"iopub.status.busy":"2022-03-03T12:37:33.861014Z","iopub.execute_input":"2022-03-03T12:37:33.861494Z","iopub.status.idle":"2022-03-03T13:13:05.244301Z","shell.execute_reply.started":"2022-03-03T12:37:33.861461Z","shell.execute_reply":"2022-03-03T13:13:05.242867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Clustering Visulization and EDA\n","metadata":{}}]}