{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib as plt\nimport seaborn as sns\nfrom collections import Counter\nimport os.path\nfrom sklearn.metrics import pairwise_distances\nfrom PIL import Image\nimport requests\nfrom io import BytesIO\nimport matplotlib.pyplot as plt\nimport pickle\nimport warnings\nfrom bs4 import BeautifulSoup\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\nimport nltk\nimport math\nimport time\nimport re\nimport os\nimport seaborn as sns\nfrom collections import Counter\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.metrics.pairwise import cosine_similarity  \nfrom sklearn.metrics import pairwise_distances\nfrom matplotlib import gridspec\nfrom scipy.sparse import hstack\nimport plotly\nimport plotly.figure_factory as ff\nfrom plotly.graph_objs import Scatter, Layout\nplotly.offline.init_notebook_mode(connected=True)\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:46:02.408558Z","iopub.execute_input":"2022-03-16T11:46:02.408815Z","iopub.status.idle":"2022-03-16T11:46:02.429039Z","shell.execute_reply.started":"2022-03-16T11:46:02.408786Z","shell.execute_reply":"2022-03-16T11:46:02.428401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HM_data = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\nHM_data=HM_data.head(10000)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:32:28.114355Z","iopub.execute_input":"2022-03-16T11:32:28.115086Z","iopub.status.idle":"2022-03-16T11:32:28.996618Z","shell.execute_reply.started":"2022-03-16T11:32:28.115049Z","shell.execute_reply":"2022-03-16T11:32:28.995886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of data-points in the data:\" , HM_data.shape[0])\nprint(\"Number of features in the data :\" , HM_data.shape[1])","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:32:30.416993Z","iopub.execute_input":"2022-03-16T11:32:30.417259Z","iopub.status.idle":"2022-03-16T11:32:30.424834Z","shell.execute_reply.started":"2022-03-16T11:32:30.417231Z","shell.execute_reply":"2022-03-16T11:32:30.424008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(HM_data.columns)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:32:32.129091Z","iopub.execute_input":"2022-03-16T11:32:32.131491Z","iopub.status.idle":"2022-03-16T11:32:32.135885Z","shell.execute_reply.started":"2022-03-16T11:32:32.131452Z","shell.execute_reply":"2022-03-16T11:32:32.135109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_path=[]\ni=0\nfor p in HM_data['article_id'].tolist():\n    path= '../input/h-and-m-personalized-fashion-recommendations/images/' + '0' + str(p)[:2] + '/'+ '0' + str(p) +'.jpg'\n\n    if os.path.exists(path):\n        i+=1\n        images_path.append(path)\n    else: images_path.append(None)\nprint(f'There is {i} article with corresponding image')\n    ","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:32:33.553507Z","iopub.execute_input":"2022-03-16T11:32:33.554027Z","iopub.status.idle":"2022-03-16T11:32:50.554749Z","shell.execute_reply.started":"2022-03-16T11:32:33.553988Z","shell.execute_reply":"2022-03-16T11:32:50.553966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HM_data['image_path']=images_path\nHM_data = HM_data[['article_id', 'prod_name', 'colour_group_name', 'image_path',\n             'product_type_name', 'product_group_name', 'detail_desc']]\nprint(\"Number of data-points in the data:\" , HM_data.shape[0])\nprint(\"Number of features in the data :\" , HM_data.shape[1])","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:33:07.0265Z","iopub.execute_input":"2022-03-16T11:33:07.026755Z","iopub.status.idle":"2022-03-16T11:33:07.050687Z","shell.execute_reply.started":"2022-03-16T11:33:07.026726Z","shell.execute_reply":"2022-03-16T11:33:07.049961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HM_data.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:33:08.062969Z","iopub.execute_input":"2022-03-16T11:33:08.063224Z","iopub.status.idle":"2022-03-16T11:33:08.081494Z","shell.execute_reply.started":"2022-03-16T11:33:08.063195Z","shell.execute_reply":"2022-03-16T11:33:08.080857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#finding the 10 most frequent product_type_names.\nprint(HM_data['product_type_name'].describe())","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:33:09.154909Z","iopub.execute_input":"2022-03-16T11:33:09.155159Z","iopub.status.idle":"2022-03-16T11:33:09.167705Z","shell.execute_reply.started":"2022-03-16T11:33:09.155131Z","shell.execute_reply":"2022-03-16T11:33:09.166677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"product_type_count = Counter(list(HM_data['product_type_name']))\nproduct_type_count.most_common(10)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:33:10.189019Z","iopub.execute_input":"2022-03-16T11:33:10.189426Z","iopub.status.idle":"2022-03-16T11:33:10.199037Z","shell.execute_reply.started":"2022-03-16T11:33:10.18939Z","shell.execute_reply":"2022-03-16T11:33:10.198226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(HM_data['colour_group_name'].describe())","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:33:11.162597Z","iopub.execute_input":"2022-03-16T11:33:11.163076Z","iopub.status.idle":"2022-03-16T11:33:11.172024Z","shell.execute_reply.started":"2022-03-16T11:33:11.163041Z","shell.execute_reply":"2022-03-16T11:33:11.171354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#find the 10 most frequent colors.\ncolor_count = Counter(list(HM_data['colour_group_name']))\ncolor_count.most_common(10)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:33:12.177876Z","iopub.execute_input":"2022-03-16T11:33:12.178593Z","iopub.status.idle":"2022-03-16T11:33:12.18893Z","shell.execute_reply.started":"2022-03-16T11:33:12.178558Z","shell.execute_reply":"2022-03-16T11:33:12.18822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(HM_data['detail_desc'].describe())","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:33:13.318861Z","iopub.execute_input":"2022-03-16T11:33:13.319353Z","iopub.status.idle":"2022-03-16T11:33:13.331008Z","shell.execute_reply.started":"2022-03-16T11:33:13.319287Z","shell.execute_reply":"2022-03-16T11:33:13.330143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of duplicates present in articles: {}\".format(sum(HM_data.duplicated(\"detail_desc\"))))","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:33:14.337179Z","iopub.execute_input":"2022-03-16T11:33:14.337756Z","iopub.status.idle":"2022-03-16T11:33:14.346683Z","shell.execute_reply.started":"2022-03-16T11:33:14.337716Z","shell.execute_reply":"2022-03-16T11:33:14.345889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HM_data","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:33:15.347595Z","iopub.execute_input":"2022-03-16T11:33:15.348102Z","iopub.status.idle":"2022-03-16T11:33:15.365815Z","shell.execute_reply.started":"2022-03-16T11:33:15.348064Z","shell.execute_reply":"2022-03-16T11:33:15.365121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#removing data-points with short-titles\nHM_data_sorted = HM_data[HM_data['detail_desc'].apply(lambda x: len(str(x).split())>4)]\nprint(\"After removal of products with short description:\", HM_data_sorted.shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:33:19.634201Z","iopub.execute_input":"2022-03-16T11:33:19.634894Z","iopub.status.idle":"2022-03-16T11:33:19.65676Z","shell.execute_reply.started":"2022-03-16T11:33:19.634857Z","shell.execute_reply":"2022-03-16T11:33:19.655939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HM_data_sorted.sort_values('detail_desc',inplace=True, ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:33:20.780031Z","iopub.execute_input":"2022-03-16T11:33:20.783347Z","iopub.status.idle":"2022-03-16T11:33:20.849955Z","shell.execute_reply.started":"2022-03-16T11:33:20.783214Z","shell.execute_reply":"2022-03-16T11:33:20.848104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"indices = []\nfor i,row in HM_data_sorted.iterrows():\n    indices.append(i)\n    \nimport itertools\nstage1_dedupe_asins = []\ni = 0\nj = 0\nnum_data_points = HM_data_sorted.shape[0]\nwhile i < num_data_points and j < num_data_points:\n       \n    previous_i = i\n    a = HM_data['detail_desc'].loc[indices[i]].split()\n\n    # search for the similar products sequentially \n    j = i+1\n    while j < num_data_points:\n        b = HM_data['detail_desc'].loc[indices[j]].split()        \n        length = max(len(a), len(b))    \n        count  = 0\n\n        for k in itertools.zip_longest(a,b): \n            if (k[0] == k[1]):\n                count += 1\n\n        # if the number of words in which both strings differ are > 2 , we are considering it as those two apperals are different\n        # if the number of words in which both strings differ are < 2 , we are considering it as those two apperals are same, hence we are ignoring them\n        if (length - count) > 2: # number of words in which both sensences differ\n            # if both strings are differ by more than 2 words we include the 1st string index\n            stage1_dedupe_asins.append(HM_data_sorted['article_id'].loc[indices[i]])\n\n            # if the comaprision between is between num_data_points, num_data_points-1 strings and they differ in more than 2 words we include both\n            if j == num_data_points-1: stage1_dedupe_asins.append(HM_data_sorted['article_id'].loc[indices[j]])\n\n            # start searching for similar apperals corresponds 2nd string\n            i = j\n            break\n        else:\n            j += 1\n    if previous_i == i:\n        break    ","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:33:21.982408Z","iopub.execute_input":"2022-03-16T11:33:21.982887Z","iopub.status.idle":"2022-03-16T11:33:22.698641Z","shell.execute_reply.started":"2022-03-16T11:33:21.982851Z","shell.execute_reply":"2022-03-16T11:33:22.69792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#removing duplicated\nHM_data = HM_data.loc[HM_data['article_id'].isin(stage1_dedupe_asins)]","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:33:25.102047Z","iopub.execute_input":"2022-03-16T11:33:25.102358Z","iopub.status.idle":"2022-03-16T11:33:25.10951Z","shell.execute_reply.started":"2022-03-16T11:33:25.102312Z","shell.execute_reply":"2022-03-16T11:33:25.108454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#getting the number of data-points after removing the duplicates\nprint('Number of data points : ', HM_data.shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:33:26.48808Z","iopub.execute_input":"2022-03-16T11:33:26.488356Z","iopub.status.idle":"2022-03-16T11:33:26.494201Z","shell.execute_reply.started":"2022-03-16T11:33:26.488322Z","shell.execute_reply":"2022-03-16T11:33:26.49346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"indices = []\nfor i,row in HM_data.iterrows():\n    indices.append(i)\n\nstage2_dedupe_asins = []\nwhile len(indices)!=0:\n    i = indices.pop()\n    stage2_dedupe_asins.append(HM_data['article_id'].loc[i])\n    # consider the first apperal's title\n    a = HM_data['detail_desc'].loc[i].split()\n \n    for j in indices:\n        \n        b = HM_data['detail_desc'].loc[j].split()\n        length = max(len(a),len(b))\n        \n        # count is used to store the number of words that are matched in both strings\n        count  = 0\n\n        for k in itertools.zip_longest(a,b): \n            if (k[0]==k[1]):\n                count += 1\n\n        #if the number of words in which both strings differ are < 3 ,\n        #we are considering it as those two apperals are same, hence we are ignoring them\n        if (length - count) < 3:\n            indices.remove(j)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:33:28.993826Z","iopub.execute_input":"2022-03-16T11:33:28.994305Z","iopub.status.idle":"2022-03-16T11:35:20.342674Z","shell.execute_reply.started":"2022-03-16T11:33:28.994257Z","shell.execute_reply":"2022-03-16T11:35:20.341941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#getting the number of data-points after removing all the duplicates\nprint('Number of data points after stage two of dedupe: ',HM_data.shape[0])\n","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:43:07.370092Z","iopub.execute_input":"2022-03-16T11:43:07.370984Z","iopub.status.idle":"2022-03-16T11:43:07.377758Z","shell.execute_reply.started":"2022-03-16T11:43:07.370946Z","shell.execute_reply":"2022-03-16T11:43:07.376991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HM_data","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:43:08.255057Z","iopub.execute_input":"2022-03-16T11:43:08.255732Z","iopub.status.idle":"2022-03-16T11:43:08.271981Z","shell.execute_reply.started":"2022-03-16T11:43:08.255684Z","shell.execute_reply":"2022-03-16T11:43:08.271338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# removing stop-words\nimport nltk\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\n\nstop_words = set(stopwords.words('english'))\nprint ('list of stop words:', stop_words)\n\ndef nlp_preprocessing(total_text, index, column):\n    if type(total_text) is not int:\n        string = \"\"\n        for words in total_text.split():    \n            word = (\"\".join(e for e in words if e.isalnum()))      \n            word = word.lower()       \n            if not word in stop_words:\n                string += word + \" \"\n        HM_data[column][index] = string\n        \nfor index, row in HM_data.iterrows():\n    nlp_preprocessing(row['detail_desc'], index, 'detail_desc')","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:43:08.916631Z","iopub.execute_input":"2022-03-16T11:43:08.91715Z","iopub.status.idle":"2022-03-16T11:43:10.250347Z","shell.execute_reply.started":"2022-03-16T11:43:08.917112Z","shell.execute_reply":"2022-03-16T11:43:10.249483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nltk.stem.porter import *\nstemmer = PorterStemmer()\nprint(stemmer.stem('arguing'))\nprint(stemmer.stem('fishing'))\n","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:43:14.093816Z","iopub.execute_input":"2022-03-16T11:43:14.094081Z","iopub.status.idle":"2022-03-16T11:43:14.10233Z","shell.execute_reply.started":"2022-03-16T11:43:14.094052Z","shell.execute_reply":"2022-03-16T11:43:14.101405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to display an image\ndef display_img(url,ax,fig):  \n    img = Image.open(url) \n    plt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:43:14.891978Z","iopub.execute_input":"2022-03-16T11:43:14.892239Z","iopub.status.idle":"2022-03-16T11:43:14.896874Z","shell.execute_reply.started":"2022-03-16T11:43:14.892211Z","shell.execute_reply":"2022-03-16T11:43:14.896162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to generate the heatmap\ndef plot_heatmap(keys, values, labels, url, text):    \n        gs = gridspec.GridSpec(2, 2, width_ratios=[4,1], height_ratios=[4,1]) \n        fig = plt.figure(figsize=(25,3))\n        # 1st, ploting heat map that represents the count of commonly ocurred words in title2\n        ax = plt.subplot(gs[0])\n        # it displays a cell in white color if the word is intersection(list of words of title1 and list of words of title2), in black if not\n        ax = sns.heatmap(np.array([values]), annot=np.array([labels]))\n        ax.set_xticklabels(keys) # set that axis labels as the words of title\n        ax.set_title(text) # apparel title        \n        # 2nd, plotting image of the the apparel\n        ax = plt.subplot(gs[1])  \n        ax.grid(False)\n        ax.set_xticks([])\n        ax.set_yticks([])  \n        display_img(url, ax, fig)           \n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:43:15.667908Z","iopub.execute_input":"2022-03-16T11:43:15.668565Z","iopub.status.idle":"2022-03-16T11:43:15.678152Z","shell.execute_reply.started":"2022-03-16T11:43:15.668528Z","shell.execute_reply":"2022-03-16T11:43:15.677239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to display heatmap and image        \ndef plot_heatmap_image(doc_id, vec1, vec2, url, text, model):\n      intersection = set(vec1.keys()) & set(vec2.keys())   \n    for i in vec2:\n        if i not in intersection:\n            vec2[i]=0\n   \n    keys = list(vec2.keys())\n    #  if ith word in intersection(lis of words of title1 and list of words of title2): values(i)=count of that word in title2 else values(i)=0 \n    values = [vec2[x] for x in vec2.keys()]   \n    if model == 'bag_of_words':\n        labels = values\n    elif model == 'tfidf':\n        labels = []\n        for x in vec2.keys(): \n            if x in  tfidf_title_vectorizer.vocabulary_:\n                labels.append(tfidf_title_features[doc_id, tfidf_title_vectorizer.vocabulary_[x]])\n            else:\n                labels.append(0)\n    elif model == 'idf':\n        labels = []\n        for x in vec2.keys():  \n            if x in  idf_title_vectorizer.vocabulary_:\n                labels.append(idf_title_features[doc_id, idf_title_vectorizer.vocabulary_[x]])\n            else:\n                labels.append(0)\n    plot_heatmap(keys, values, labels, url, text)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:43:20.093327Z","iopub.execute_input":"2022-03-16T11:43:20.093638Z","iopub.status.idle":"2022-03-16T11:43:20.103008Z","shell.execute_reply.started":"2022-03-16T11:43:20.093607Z","shell.execute_reply":"2022-03-16T11:43:20.102035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#function to convert text into vector\ndef text_to_vector(text):\n    word = re.compile(r'\\w+')\n    words = word.findall(text) \n    return Counter(words) ","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:43:21.735974Z","iopub.execute_input":"2022-03-16T11:43:21.736596Z","iopub.status.idle":"2022-03-16T11:43:21.741205Z","shell.execute_reply.started":"2022-03-16T11:43:21.736546Z","shell.execute_reply":"2022-03-16T11:43:21.740422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to display the result\ndef get_result(doc_id, content_a, content_b, url, model):\n    text1 = content_a\n    text2 = content_b  \n    vector1 = text_to_vector(text1) \n    vector2 = text_to_vector(text2)\n    plot_heatmap_image(doc_id, vector1, vector2, url, text2, model)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T11:43:25.956429Z","iopub.execute_input":"2022-03-16T11:43:25.957051Z","iopub.status.idle":"2022-03-16T11:43:25.962172Z","shell.execute_reply.started":"2022-03-16T11:43:25.957Z","shell.execute_reply":"2022-03-16T11:43:25.961359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\ntitle_vectorizer = CountVectorizer()\ntitle_features   = title_vectorizer.fit_transform(HM_data['detail_desc'])\ntitle_features.get_shape()\n","metadata":{"execution":{"iopub.status.busy":"2022-03-16T12:13:50.863132Z","iopub.execute_input":"2022-03-16T12:13:50.863612Z","iopub.status.idle":"2022-03-16T12:13:50.925485Z","shell.execute_reply.started":"2022-03-16T12:13:50.863573Z","shell.execute_reply":"2022-03-16T12:13:50.924637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data=HM_data.copy()\ndef bag_of_words_model(doc_id, num_results): \n    pairwise_dist = pairwise_distances(title_features,title_features[doc_id])\n  \n    indices = np.argsort(pairwise_dist.flatten())[0:num_results]\n    #pdists will store the smallest distances\n    pdists  = np.sort(pairwise_dist.flatten())[0:num_results]\n    #data frame indices of the 9 smallest distace's\n    df_indices = list(data.index[indices])    \n    for i in range(0,len(indices)):\n        # we will pass 1. doc_id, 2. title1, 3. title2, url, model\n        get_result(indices[i],data['detail_desc'].loc[df_indices[0]], data['detail_desc'].loc[df_indices[i]], data['image_path'].loc[df_indices[i]], 'bag_of_words')\n        print('article_id :',data['article_id'].loc[df_indices[i]])\n        print ('detail_desc:', data['detail_desc'].loc[df_indices[i]])\n        print ('Euclidean similarity with the query image :', pdists[i])\n        print('='*60)\n#call the bag-of-words model for a product to get similar products\nbag_of_words_model(1587, 20)\n# in the output heat map each value represents the tfidf values of the label word, the color represents the intersection with inputs title","metadata":{"execution":{"iopub.status.busy":"2022-03-16T12:15:16.569946Z","iopub.execute_input":"2022-03-16T12:15:16.570196Z","iopub.status.idle":"2022-03-16T12:15:29.6787Z","shell.execute_reply.started":"2022-03-16T12:15:16.570168Z","shell.execute_reply":"2022-03-16T12:15:29.677973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfidf_title_vectorizer = TfidfVectorizer(min_df = 0)\ntfidf_title_features = tfidf_title_vectorizer.fit_transform(data['detail_desc'])","metadata":{"execution":{"iopub.status.busy":"2022-03-16T12:16:26.692556Z","iopub.execute_input":"2022-03-16T12:16:26.693082Z","iopub.status.idle":"2022-03-16T12:16:26.754369Z","shell.execute_reply.started":"2022-03-16T12:16:26.693043Z","shell.execute_reply":"2022-03-16T12:16:26.753702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tfidf_model(doc_id, num_results):   \n    pairwise_dist = pairwise_distances(tfidf_title_features,tfidf_title_features[doc_id]) \n    indices = np.argsort(pairwise_dist.flatten())[0:num_results]  \n    pdists  = np.sort(pairwise_dist.flatten())[0:num_results]  \n    df_indices = list(data.index[indices])\n    for i in range(0,len(indices)):  \n        get_result(indices[i], data['detail_desc'].loc[df_indices[0]], data['detail_desc'].loc[df_indices[i]], data['image_path'].loc[df_indices[i]], 'tfidf')\n        print('article_id :',data['article_id'].loc[df_indices[i]])\n        print ('Eucliden distance from the given image :', pdists[i])\n        print('='*125)\ntfidf_model(1587, 20)\n","metadata":{"execution":{"iopub.status.busy":"2022-03-16T12:17:32.170199Z","iopub.execute_input":"2022-03-16T12:17:32.170475Z","iopub.status.idle":"2022-03-16T12:17:45.832647Z","shell.execute_reply.started":"2022-03-16T12:17:32.170447Z","shell.execute_reply":"2022-03-16T12:17:45.831968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idf_title_vectorizer = CountVectorizer()\nidf_title_features = idf_title_vectorizer.fit_transform(data['detail_desc'])","metadata":{"execution":{"iopub.status.busy":"2022-03-16T12:17:56.554414Z","iopub.execute_input":"2022-03-16T12:17:56.554906Z","iopub.status.idle":"2022-03-16T12:17:56.613983Z","shell.execute_reply.started":"2022-03-16T12:17:56.55487Z","shell.execute_reply":"2022-03-16T12:17:56.613255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def n_containing(word): \n    return sum(1 for blob in data['detail_desc'] if word in blob.split())\ndef idf(word):\n    return math.log(data.shape[0] / (n_containing(word)))","metadata":{"execution":{"iopub.status.busy":"2022-03-16T12:18:19.123203Z","iopub.execute_input":"2022-03-16T12:18:19.123756Z","iopub.status.idle":"2022-03-16T12:18:19.128069Z","shell.execute_reply.started":"2022-03-16T12:18:19.123716Z","shell.execute_reply":"2022-03-16T12:18:19.127368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nidf_title_features  = idf_title_features.astype(np.float)\nfor i in idf_title_vectorizer.vocabulary_.keys():  \n    idf_val = idf(i)\n    # to calculate idf_title_features we will replace the count values with the idf values of the word\n    for j in idf_title_features[:, idf_title_vectorizer.vocabulary_[i]].nonzero()[0]:        \n        idf_title_features[j,idf_title_vectorizer.vocabulary_[i]] = idf_val  ","metadata":{"execution":{"iopub.status.busy":"2022-03-16T12:19:12.789431Z","iopub.execute_input":"2022-03-16T12:19:12.7901Z","iopub.status.idle":"2022-03-16T12:19:22.197084Z","shell.execute_reply.started":"2022-03-16T12:19:12.790063Z","shell.execute_reply":"2022-03-16T12:19:22.196347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def idf_model(doc_id, num_results): \n    pairwise_dist = pairwise_distances(idf_title_features,idf_title_features[doc_id])\n   \n    indices = np.argsort(pairwise_dist.flatten())[0:num_results]\n   \n    pdists  = np.sort(pairwise_dist.flatten())[0:num_results]\n  \n    df_indices = list(data.index[indices])\n    for i in range(0,len(indices)):\n        get_result(indices[i],data['detail_desc'].loc[df_indices[0]], data['detail_desc'].loc[df_indices[i]], data['image_path'].loc[df_indices[i]], 'idf')\n        print('article_id :',data['article_id'].loc[df_indices[i]])\n        print ('euclidean distance from the given image :', pdists[i])\n        print('='*125)        \nidf_model(1587,20)\n","metadata":{"execution":{"iopub.status.busy":"2022-03-16T12:20:05.123113Z","iopub.execute_input":"2022-03-16T12:20:05.12339Z","iopub.status.idle":"2022-03-16T12:20:16.93809Z","shell.execute_reply.started":"2022-03-16T12:20:05.123361Z","shell.execute_reply":"2022-03-16T12:20:16.937311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}