{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**In this notebook I am using embedding of product descriptions to find product similarity**\n\nThis method might give some redundant results becuase many items have the same descriptions. But it might be useful in a hybrid recommendation system or for comparison of different models.\n\n* Embeddings are produced by 'universal-sentence-encoder' found on TensorFlow Hub\n* Distance metric used is dot product\n* Similarity score are stored in a pickle file due to memory constraints 'https://www.kaggle.com/datasets/mohammedobeidat/product-desc-similarity-scores'\n\n","metadata":{}},{"cell_type":"code","source":"import tensorflow_hub as hub\nimport numpy as np\nimport pandas as pd\nimport pickle\nimport warnings\nimport matplotlib.pyplot as plt\nwarnings.filterwarnings('ignore')\n\npath = '../input/h-and-m-personalized-fashion-recommendations/articles.csv'\n \ndf = pd.read_csv(path).astype(str)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-15T08:22:29.535521Z","iopub.execute_input":"2022-05-15T08:22:29.535871Z","iopub.status.idle":"2022-05-15T08:22:38.761648Z","shell.execute_reply.started":"2022-05-15T08:22:29.535787Z","shell.execute_reply":"2022-05-15T08:22:38.760857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#@title Load the Universal Sentence Encoder's TF Hub module\n\nmodule_url = \"https://tfhub.dev/google/universal-sentence-encoder/4\" #@param [\"https://tfhub.dev/google/universal-sentence-encoder/4\", \"https://tfhub.dev/google/universal-sentence-encoder-large/5\"]\nmodel = hub.load(module_url)","metadata":{"execution":{"iopub.status.busy":"2022-05-15T08:22:38.763606Z","iopub.execute_input":"2022-05-15T08:22:38.763908Z","iopub.status.idle":"2022-05-15T08:22:56.709552Z","shell.execute_reply.started":"2022-05-15T08:22:38.763866Z","shell.execute_reply":"2022-05-15T08:22:56.708623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"desc = df['detail_desc'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-05-15T08:22:56.710728Z","iopub.execute_input":"2022-05-15T08:22:56.710977Z","iopub.status.idle":"2022-05-15T08:22:56.769275Z","shell.execute_reply.started":"2022-05-15T08:22:56.710949Z","shell.execute_reply":"2022-05-15T08:22:56.768504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"desc","metadata":{"execution":{"iopub.status.busy":"2022-05-15T08:22:56.770776Z","iopub.execute_input":"2022-05-15T08:22:56.771011Z","iopub.status.idle":"2022-05-15T08:22:58.212150Z","shell.execute_reply.started":"2022-05-15T08:22:56.770984Z","shell.execute_reply":"2022-05-15T08:22:58.211277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeds = model(desc)","metadata":{"execution":{"iopub.status.busy":"2022-05-12T03:49:50.058351Z","iopub.execute_input":"2022-05-12T03:49:50.059134Z","iopub.status.idle":"2022-05-12T03:49:50.064074Z","shell.execute_reply.started":"2022-05-12T03:49:50.059082Z","shell.execute_reply":"2022-05-12T03:49:50.063163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('text_desc_embeddings.pickle', 'wb') as f:\n    pickle.dump(embeds, f)\n    \nwith open('text_desc.pickle', 'wb') as f:\n    pickle.dump(desc, f)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file = open('scores_top20.pkl','wb')\n\nfor embed in embeds: \n    top10 = np.inner(embed, embeds)\n    top10_index = np.argsort(-top10)[:20]\n    top10_score = top10[top10_index]\n\n    pickle.dump([top10_index, top10_score], file)\n\nfile.close()","metadata":{"execution":{"iopub.status.busy":"2022-05-12T03:49:50.065553Z","iopub.execute_input":"2022-05-12T03:49:50.066056Z","iopub.status.idle":"2022-05-12T03:49:50.078383Z","shell.execute_reply.started":"2022-05-12T03:49:50.066Z","shell.execute_reply":"2022-05-12T03:49:50.077581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_top10():\n    \n    top10_indecies = []\n    top10_scores = []\n\n    with open('../input/product-desc-similarity-scores/scores_top10.pkl', 'rb') as f:\n        for i in range(len(desc)):\n            try:\n                row = pickle.load(f)\n                top10_indecies.append(row[0])\n                top10_scores.append(row[1])\n            except:\n                print('Done Loading')\n                \n    return top10_indecies, top10_scores","metadata":{"execution":{"iopub.status.busy":"2022-05-12T03:49:50.079925Z","iopub.execute_input":"2022-05-12T03:49:50.080418Z","iopub.status.idle":"2022-05-12T03:49:52.744541Z","shell.execute_reply.started":"2022-05-12T03:49:50.080383Z","shell.execute_reply":"2022-05-12T03:49:52.742034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top10_indecies, top10_scores = load_top10()","metadata":{"execution":{"iopub.status.busy":"2022-05-12T05:06:51.374402Z","iopub.execute_input":"2022-05-12T05:06:51.375819Z","iopub.status.idle":"2022-05-12T05:06:51.427029Z","shell.execute_reply.started":"2022-05-12T05:06:51.375754Z","shell.execute_reply":"2022-05-12T05:06:51.426168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def recommend(article_id):\n    \n    desc_list = desc.tolist()\n    product_desc = df[df.article_id == sample]['detail_desc'].values[0]\n    desc_index = desc_list.index(product_desc)\n    rcmnds_indecies = top10_indecies[desc_index]\n    rcmnds_scores = top10_scores[desc_index]\n    rcmnds_descs = desc[rcmnds_indecies]\n    map_dict = {i:j for i, j in zip(rcmnds_descs, rcmnds_scores)}\n    rcmnds_article_ids = df[df.detail_desc.isin(rcmnds_descs)]\n    rcmnds_article_ids['score'] = rcmnds_article_ids.detail_desc.map(map_dict)\n    rcmnds_article_ids = rcmnds_article_ids[rcmnds_article_ids.score < 0.99]\n    rcmnds_article_ids = rcmnds_article_ids.sort_values(by='score', ascending=False).drop_duplicates('score')\n    \n    \n    return(rcmnds_article_ids[['article_id', 'score']])","metadata":{"execution":{"iopub.status.busy":"2022-05-12T04:58:51.979859Z","iopub.execute_input":"2022-05-12T04:58:51.980171Z","iopub.status.idle":"2022-05-12T04:58:51.98824Z","shell.execute_reply.started":"2022-05-12T04:58:51.98013Z","shell.execute_reply":"2022-05-12T04:58:51.987494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_items(items):\n    path = \"../input/h-and-m-personalized-fashion-recommendations/images\"\n\n    k = len(items)\n    fig = plt.figure(figsize=(15, 10))\n    for item, i in zip(items, range(1, k+1)):\n        item = '0' + str(item)\n        sub = item[:3]\n        image = path + \"/\"+ sub + \"/\"+ item +\".jpg\"\n        image = plt.imread(image)\n        fig.add_subplot(1, k, i)\n        plt.imshow(image)\n        ","metadata":{"execution":{"iopub.status.busy":"2022-05-12T04:57:02.354206Z","iopub.execute_input":"2022-05-12T04:57:02.354512Z","iopub.status.idle":"2022-05-12T04:57:02.361682Z","shell.execute_reply.started":"2022-05-12T04:57:02.354479Z","shell.execute_reply":"2022-05-12T04:57:02.360385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = df.article_id.iloc[1]\nrcmnds = recommend(sample)","metadata":{"execution":{"iopub.status.busy":"2022-05-12T04:57:02.69296Z","iopub.execute_input":"2022-05-12T04:57:02.693897Z","iopub.status.idle":"2022-05-12T04:57:02.734222Z","shell.execute_reply.started":"2022-05-12T04:57:02.69384Z","shell.execute_reply":"2022-05-12T04:57:02.733405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(rcmnds.sample(6).article_id.values)","metadata":{"execution":{"iopub.status.busy":"2022-05-12T04:57:03.722195Z","iopub.execute_input":"2022-05-12T04:57:03.723066Z","iopub.status.idle":"2022-05-12T04:57:06.483355Z","shell.execute_reply.started":"2022-05-12T04:57:03.723022Z","shell.execute_reply":"2022-05-12T04:57:06.482403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trans = next(pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv', dtype=str, chunksize=10000))\n\ntrans.drop_duplicates(['customer_id', 'article_id'], inplace=True)\ntrans.article_id = trans.article_id.map(lambda x: x[1:])\ngrouped = trans.groupby('customer_id')","metadata":{"execution":{"iopub.status.busy":"2022-05-12T04:54:36.544759Z","iopub.execute_input":"2022-05-12T04:54:36.545586Z","iopub.status.idle":"2022-05-12T04:54:36.582221Z","shell.execute_reply.started":"2022-05-12T04:54:36.545551Z","shell.execute_reply":"2022-05-12T04:54:36.581575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new = []\nfor group in grouped.groups:\n    temp = grouped.get_group(group)\n    if len(temp) >= 12:\n        new.append([group, temp.article_id.values.tolist()[:12]])","metadata":{"execution":{"iopub.status.busy":"2022-05-12T04:57:17.041503Z","iopub.execute_input":"2022-05-12T04:57:17.042162Z","iopub.status.idle":"2022-05-12T04:57:17.306249Z","shell.execute_reply.started":"2022-05-12T04:57:17.042099Z","shell.execute_reply":"2022-05-12T04:57:17.305095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_sim(item, items):\n    \n    item_desc = df.detail_desc[df.article_id == item].values[0]\n    item_embed = model([item_desc])[0]\n    \n    items_desc = df[df.article_id.isin(items)].detail_desc\n    items_embed = model(items_desc)\n    scores = []\n    \n    for i in items_embed:\n        sim = np.dot(i, item_embed)\n        scores.append(sim)\n        \n        \n    return np.mean(scores)","metadata":{"execution":{"iopub.status.busy":"2022-05-12T04:57:18.094181Z","iopub.execute_input":"2022-05-12T04:57:18.09451Z","iopub.status.idle":"2022-05-12T04:57:18.102134Z","shell.execute_reply.started":"2022-05-12T04:57:18.094471Z","shell.execute_reply":"2022-05-12T04:57:18.100913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = []\nscores = []\n\nfor key, value in new:\n    temp = []\n    for item in value[:6]:\n        temp.append(recommend(item))\n    temp2 = pd.concat(temp).sample(6, random_state=42)\n    temp2['actual'] = value[6:]\n    temp3 = []\n    for item in temp2.actual:\n        sim = get_sim(item, temp2.article_id) \n        temp3.append(sim)\n        \n    scores.append(np.mean(temp3))\n    preds.append([key, temp2])","metadata":{"execution":{"iopub.status.busy":"2022-05-12T04:58:56.791399Z","iopub.execute_input":"2022-05-12T04:58:56.791743Z","iopub.status.idle":"2022-05-12T04:59:25.117996Z","shell.execute_reply.started":"2022-05-12T04:58:56.79171Z","shell.execute_reply":"2022-05-12T04:59:25.11676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.mean(scores)","metadata":{"execution":{"iopub.status.busy":"2022-05-12T04:59:25.119964Z","iopub.execute_input":"2022-05-12T04:59:25.12031Z","iopub.status.idle":"2022-05-12T04:59:25.128942Z","shell.execute_reply.started":"2022-05-12T04:59:25.120262Z","shell.execute_reply":"2022-05-12T04:59:25.128154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customer = preds[0][1]","metadata":{"execution":{"iopub.status.busy":"2022-05-12T04:57:25.823222Z","iopub.execute_input":"2022-05-12T04:57:25.823549Z","iopub.status.idle":"2022-05-12T04:57:25.833996Z","shell.execute_reply.started":"2022-05-12T04:57:25.823506Z","shell.execute_reply":"2022-05-12T04:57:25.832967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customer","metadata":{"execution":{"iopub.status.busy":"2022-05-12T04:57:25.83582Z","iopub.execute_input":"2022-05-12T04:57:25.836547Z","iopub.status.idle":"2022-05-12T04:57:25.853333Z","shell.execute_reply.started":"2022-05-12T04:57:25.836508Z","shell.execute_reply":"2022-05-12T04:57:25.852409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(customer.article_id.values)","metadata":{"execution":{"iopub.status.busy":"2022-05-12T04:57:25.854638Z","iopub.execute_input":"2022-05-12T04:57:25.855332Z","iopub.status.idle":"2022-05-12T04:57:29.068675Z","shell.execute_reply.started":"2022-05-12T04:57:25.855284Z","shell.execute_reply":"2022-05-12T04:57:29.067968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(customer.actual.values)","metadata":{"execution":{"iopub.status.busy":"2022-05-12T04:57:29.070592Z","iopub.execute_input":"2022-05-12T04:57:29.071312Z","iopub.status.idle":"2022-05-12T04:57:31.806546Z","shell.execute_reply.started":"2022-05-12T04:57:29.071276Z","shell.execute_reply":"2022-05-12T04:57:31.805683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}