{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import cv2\nimport numpy as np # linear algebra\nimport os\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n\nimport seaborn as sns\nimport plotly.express as px\nfrom os import listdir\nfrom os.path import isfile, join\n\nfrom termcolor import colored\nfrom IPython.display import HTML\nfrom PIL import Image\n\nimport warnings\npd.set_option('display.max_rows', None)\npd.set_option('display.max_columns', None)\npd.set_option('float_format', '{:f}'.format)\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-16T05:36:23.680832Z","iopub.execute_input":"2022-04-16T05:36:23.681602Z","iopub.status.idle":"2022-04-16T05:36:26.508874Z","shell.execute_reply.started":"2022-04-16T05:36:23.681557Z","shell.execute_reply":"2022-04-16T05:36:26.507990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# This notebook is an continuation to my previous notebook on finding similar products.\n\n## Go thorugh it to understand the full cycle: https://www.kaggle.com/code/sussudharsan/h-m-similar-products-recommender-script","metadata":{"execution":{"iopub.status.busy":"2022-04-16T05:36:23.555771Z","iopub.execute_input":"2022-04-16T05:36:23.556760Z","iopub.status.idle":"2022-04-16T05:36:23.584295Z","shell.execute_reply.started":"2022-04-16T05:36:23.556655Z","shell.execute_reply":"2022-04-16T05:36:23.583022Z"}}},{"cell_type":"markdown","source":"In the previous notebook, we built a recommender system to find similar articles.\n\nUsing this recommder, we'll find similar products to what a customer has purcahsed already.\n\nTop 12 products for each customer is then selected usign frequency approach.\n\nThis approach is shown below for a sample population of article and transaction, which gave promising score.\n\nThis can be applied to the entire dataset.\n\n# <span style=\"color:green\">*Upvote if this notebook is useful!*</span>","metadata":{}},{"cell_type":"code","source":"articles = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\ncustomers = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/customers.csv\")\ntransactions = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")\nsample_sub = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-04-16T05:36:43.598364Z","iopub.execute_input":"2022-04-16T05:36:43.598715Z","iopub.status.idle":"2022-04-16T05:38:02.151924Z","shell.execute_reply.started":"2022-04-16T05:36:43.598682Z","shell.execute_reply":"2022-04-16T05:38:02.150944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-16T05:38:02.165808Z","iopub.execute_input":"2022-04-16T05:38:02.166096Z","iopub.status.idle":"2022-04-16T05:38:02.186007Z","shell.execute_reply.started":"2022-04-16T05:38:02.166056Z","shell.execute_reply":"2022-04-16T05:38:02.185082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get only Customers X Articles Purchased from transaction data","metadata":{}},{"cell_type":"code","source":"cust_pur = transactions[['customer_id','article_id']]","metadata":{"execution":{"iopub.status.busy":"2022-04-16T05:38:02.187378Z","iopub.execute_input":"2022-04-16T05:38:02.187623Z","iopub.status.idle":"2022-04-16T05:38:02.658094Z","shell.execute_reply.started":"2022-04-16T05:38:02.187596Z","shell.execute_reply":"2022-04-16T05:38:02.657393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let's recreate the similar product recommender script from this notebook\nhttps://www.kaggle.com/code/sussudharsan/h-m-similar-products-recommender-script","metadata":{}},{"cell_type":"markdown","source":"The below script is run only for a small sample population. This concept can be expanded to the full article and transaction dataset.","metadata":{}},{"cell_type":"code","source":"articles_sub = articles[['article_id','prod_name','product_type_name','product_group_name','graphical_appearance_name','colour_group_name'\n                         ,'perceived_colour_value_name','perceived_colour_master_name','department_name','index_name','index_group_name'\n                         ,'section_name','garment_group_name','detail_desc']]\n\n# Let's remove space in all string columns\nfor i in articles_sub.columns[1:]:\n    articles_sub[i] = articles_sub[i].str.replace(\" \",\"\")\n\n#Combine all info from columns to a single column separated by space\n\ncols = ['prod_name', 'product_type_name', 'product_group_name',\n       'graphical_appearance_name', 'colour_group_name',\n       'perceived_colour_value_name', 'perceived_colour_master_name',\n       'department_name', 'index_name', 'index_group_name', 'section_name',\n       'garment_group_name', 'detail_desc']\narticles_sub['combined'] = articles_sub[cols].apply(lambda row: ' '.join(row.values.astype(str)), axis=1)\n\narticles_final = articles_sub[['article_id','combined']]\n\n#Only 5000 products are taken because of computational issues\narticles_final = articles_final.loc[:1000]\n\n#Import TfIdfVectorizer from scikit-learn\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n\n#Define a TF-IDF Vectorizer Object. Remove all english stop words such as 'the', 'a'\ntfidf = TfidfVectorizer(stop_words='english')\n\n#Replace NaN with an empty string\narticles_final['combined'] = articles_final['combined'].fillna('')\n\n#Construct the required TF-IDF matrix by fitting and transforming the data\ntfidf_matrix = tfidf.fit_transform(articles_final['combined'])\n\n#Output the shape of tfidf_matrix\ntfidf_matrix.shape\n\n# Import linear_kernel\nfrom sklearn.metrics.pairwise import linear_kernel\n\n# Compute the cosine similarity matrix\ncosine_sim = linear_kernel(tfidf_matrix, tfidf_matrix)\n\nindices = pd.Series(articles_final.index, index=articles_final['article_id']).drop_duplicates()\n\n# Function that takes in article_id as input and outputs most similar articles\ndef get_recommendations(title, cosine_sim=cosine_sim):\n    # Get the index of the article that matches the title\n    idx = indices[title]\n\n    # Get the pairwsie similarity scores of all articles\n    sim_scores = list(enumerate(cosine_sim[idx]))\n\n    # Sort the articles based on the similarity scores\n    sim_scores = sorted(sim_scores, key=lambda x: x[1], reverse=True)\n\n    # Get the scores of the 10 most similar articles\n    sim_scores = sim_scores[:12]\n\n    # Get the article indices\n    article_indices = [i[0] for i in sim_scores]\n\n    # Return the top 10 most similar articles\n    return articles_final['article_id'].iloc[article_indices]\n","metadata":{"execution":{"iopub.status.busy":"2022-04-16T05:38:06.887699Z","iopub.execute_input":"2022-04-16T05:38:06.888001Z","iopub.status.idle":"2022-04-16T05:38:11.222384Z","shell.execute_reply.started":"2022-04-16T05:38:06.887972Z","shell.execute_reply":"2022-04-16T05:38:11.221620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let's predict the 12 similar items for every article purchased by a customer","metadata":{}},{"cell_type":"markdown","source":"## For now, take only transactions with article id present in our articles_final dataset.\nThis is because of memory constraint","metadata":{}},{"cell_type":"code","source":"articles_final.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-16T05:38:16.876915Z","iopub.execute_input":"2022-04-16T05:38:16.877225Z","iopub.status.idle":"2022-04-16T05:38:16.888177Z","shell.execute_reply.started":"2022-04-16T05:38:16.877193Z","shell.execute_reply":"2022-04-16T05:38:16.887216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cust_pur = cust_pur[cust_pur['article_id'].isin(articles_final['article_id'])]\ncust_pur.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-16T05:38:22.484391Z","iopub.execute_input":"2022-04-16T05:38:22.485423Z","iopub.status.idle":"2022-04-16T05:38:23.377208Z","shell.execute_reply.started":"2022-04-16T05:38:22.485370Z","shell.execute_reply":"2022-04-16T05:38:23.376618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Subset only 20k rows to build the pipeline\ncust_pur.reset_index(inplace=True,drop=True)\ncust_pur = cust_pur.loc[:20000]","metadata":{"execution":{"iopub.status.busy":"2022-04-16T05:38:36.748244Z","iopub.execute_input":"2022-04-16T05:38:36.748771Z","iopub.status.idle":"2022-04-16T05:38:36.753995Z","shell.execute_reply.started":"2022-04-16T05:38:36.748721Z","shell.execute_reply":"2022-04-16T05:38:36.752770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predicting articles to be purchased by every customer based on their history","metadata":{}},{"cell_type":"code","source":"cust_pur['similar_articles'] = cust_pur['article_id'].apply(lambda x: list(get_recommendations(x)))","metadata":{"execution":{"iopub.status.busy":"2022-04-16T05:38:51.333874Z","iopub.execute_input":"2022-04-16T05:38:51.334210Z","iopub.status.idle":"2022-04-16T05:39:08.299819Z","shell.execute_reply.started":"2022-04-16T05:38:51.334178Z","shell.execute_reply":"2022-04-16T05:39:08.299009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def app_func(dataf):\n  temp = []\n  dataf.reset_index(inplace=True,drop=True)\n  for i in range(dataf.shape[0]):\n    #print(i)\n    #print(i,dataf['similar_articles'][i])\n    temp = temp + dataf['similar_articles'][i]\n  #print('temp',temp)\n  return temp#[ item for elem in temp for item in elem]","metadata":{"execution":{"iopub.status.busy":"2022-04-16T05:39:23.780603Z","iopub.execute_input":"2022-04-16T05:39:23.780928Z","iopub.status.idle":"2022-04-16T05:39:23.787537Z","shell.execute_reply.started":"2022-04-16T05:39:23.780894Z","shell.execute_reply":"2022-04-16T05:39:23.786399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fin = pd.DataFrame(cust_pur.groupby(['customer_id']).apply(app_func))\nfin = fin.reset_index()\nfin.columns = ['customer_id','next_articles']\n\nfrom collections import Counter\nfor i in range(fin.shape[0]):\n    fin['next_articles'][i] = ([element for element,count in Counter(fin['next_articles'][i]).most_common()])[:12]\n\nfor i in range(fin.shape[0]):\n  fin['next_articles'][i] = ' '.join(['0'+ str(x) for x in fin['next_articles'][i]])","metadata":{"execution":{"iopub.status.busy":"2022-04-16T05:39:57.133108Z","iopub.execute_input":"2022-04-16T05:39:57.133904Z","iopub.status.idle":"2022-04-16T05:40:07.813594Z","shell.execute_reply.started":"2022-04-16T05:39:57.133858Z","shell.execute_reply":"2022-04-16T05:40:07.812650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fin.columns = ['customer_id','prediction']","metadata":{"execution":{"iopub.status.busy":"2022-04-16T05:40:20.604315Z","iopub.execute_input":"2022-04-16T05:40:20.605078Z","iopub.status.idle":"2022-04-16T05:40:20.609715Z","shell.execute_reply.started":"2022-04-16T05:40:20.605032Z","shell.execute_reply":"2022-04-16T05:40:20.608792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# The above approach can eb expanded tothe full articles id and transaction dataset","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}