{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importing Libraries","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport numpy as np\nfrom warnings import filterwarnings\nimport nltk\nfrom nltk.corpus import stopwords\nimport re\nimport string\nfrom nltk.stem import WordNetLemmatizer\nfrom nltk import word_tokenize\nfrom nltk.corpus import stopwords\nfrom sklearn.feature_extraction.text import TfidfVectorizer \nfrom sklearn.metrics.pairwise import cosine_similarity\nfilterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:37:41.744288Z","iopub.execute_input":"2022-06-11T16:37:41.744800Z","iopub.status.idle":"2022-06-11T16:37:43.586710Z","shell.execute_reply.started":"2022-06-11T16:37:41.744698Z","shell.execute_reply":"2022-06-11T16:37:43.585882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing Datasets","metadata":{}},{"cell_type":"code","source":"articles = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\ncustomers = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/customers.csv\")\ntransactions = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:37:43.588127Z","iopub.execute_input":"2022-06-11T16:37:43.588359Z","iopub.status.idle":"2022-06-11T16:38:57.941807Z","shell.execute_reply.started":"2022-06-11T16:37:43.588331Z","shell.execute_reply":"2022-06-11T16:38:57.940535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Articles Dataset","metadata":{}},{"cell_type":"markdown","source":"This dataset contains products and related information about them. Rows with null values were removed from the data set.","metadata":{}},{"cell_type":"code","source":"articles = articles.dropna()","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:38:57.943724Z","iopub.execute_input":"2022-06-11T16:38:57.944043Z","iopub.status.idle":"2022-06-11T16:38:58.161924Z","shell.execute_reply.started":"2022-06-11T16:38:57.943996Z","shell.execute_reply":"2022-06-11T16:38:58.161180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles.info()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2022-06-11T16:38:58.163855Z","iopub.execute_input":"2022-06-11T16:38:58.164498Z","iopub.status.idle":"2022-06-11T16:38:58.369459Z","shell.execute_reply.started":"2022-06-11T16:38:58.164460Z","shell.execute_reply":"2022-06-11T16:38:58.368524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Only NLP-related variables were selected from the dataset and all those variables containing text were combined in one column with the name \"text\". Since various numeric values will not be used, they were not selected.","metadata":{}},{"cell_type":"code","source":"articles[\"text\"] = articles[\"prod_name\"].map(str) + \" \" + articles[\"product_type_name\"] +\" \"+ articles[\"product_group_name\"]+ \" \"+ articles['graphical_appearance_name']+\" \"+ articles['colour_group_name'] +\" \"+ articles['perceived_colour_value_name']+ \" \" + articles[\"perceived_colour_master_name\"] +\" \"+ articles[\"department_name\"]+ \" \"+ articles['index_name']+\" \"+articles['index_group_name'] +\" \"+articles['section_name']+ \" \"+ articles['garment_group_name']+\" \"+articles['detail_desc']\narticles.head(2)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2022-06-11T16:38:58.370588Z","iopub.execute_input":"2022-06-11T16:38:58.370810Z","iopub.status.idle":"2022-06-11T16:38:58.947365Z","shell.execute_reply.started":"2022-06-11T16:38:58.370774Z","shell.execute_reply":"2022-06-11T16:38:58.946526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Finally, a dataframe created only includes 'article_id', 'product_code', 'text' columns.","metadata":{}},{"cell_type":"code","source":"df_all = articles[['article_id', 'product_code', 'text']]\n#pd.set_option(\"display.max_colwidth\", -1)","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:38:58.948960Z","iopub.execute_input":"2022-06-11T16:38:58.949417Z","iopub.status.idle":"2022-06-11T16:38:58.979568Z","shell.execute_reply.started":"2022-06-11T16:38:58.949374Z","shell.execute_reply":"2022-06-11T16:38:58.978957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:38:58.981024Z","iopub.execute_input":"2022-06-11T16:38:58.981499Z","iopub.status.idle":"2022-06-11T16:38:58.993180Z","shell.execute_reply.started":"2022-06-11T16:38:58.981455Z","shell.execute_reply":"2022-06-11T16:38:58.992450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The text variable needs to be cleared for NLP implementation. For this reason, the necessary files have been downloaded.","metadata":{}},{"cell_type":"code","source":"nltk.download('punkt')\nnltk.download('stopwords')\nnltk.download('wordnet')\nnltk.download('averaged_perceptron_tagger')","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:38:58.994538Z","iopub.execute_input":"2022-06-11T16:38:58.994773Z","iopub.status.idle":"2022-06-11T16:38:59.279888Z","shell.execute_reply.started":"2022-06-11T16:38:58.994737Z","shell.execute_reply":"2022-06-11T16:38:59.278918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Text cleaning function defined and applied on text variable","metadata":{}},{"cell_type":"code","source":"stop = stopwords.words('english')\nstop_words_ = set(stopwords.words('english'))\nwn = WordNetLemmatizer()\n\ndef black_txt(token):\n    return  token not in stop_words_ and token not in list(string.punctuation)  and len(token)>2   \n  \ndef clean_txt(text):\n  clean_text = []\n  clean_text2 = []\n  text = re.sub(\"'\", \"\",text)\n  text=re.sub(\"(\\\\d|\\\\W)+\",\" \",text) \n  text = text.replace(\"nbsp\", \"\")\n  clean_text = [ wn.lemmatize(word, pos=\"v\") for word in word_tokenize(text.lower()) if black_txt(word)]\n  clean_text2 = [word for word in clean_text if black_txt(word)]\n  return \" \".join(clean_text2)","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:38:59.281411Z","iopub.execute_input":"2022-06-11T16:38:59.281738Z","iopub.status.idle":"2022-06-11T16:38:59.293620Z","shell.execute_reply.started":"2022-06-11T16:38:59.281696Z","shell.execute_reply":"2022-06-11T16:38:59.292996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all['text'] = df_all['text'].apply(clean_txt)","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:38:59.296435Z","iopub.execute_input":"2022-06-11T16:38:59.296989Z","iopub.status.idle":"2022-06-11T16:40:34.533080Z","shell.execute_reply.started":"2022-06-11T16:38:59.296948Z","shell.execute_reply":"2022-06-11T16:40:34.531782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:40:34.534552Z","iopub.execute_input":"2022-06-11T16:40:34.534882Z","iopub.status.idle":"2022-06-11T16:40:34.545907Z","shell.execute_reply.started":"2022-06-11T16:40:34.534839Z","shell.execute_reply":"2022-06-11T16:40:34.545075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Initializing tfidf vectorizer for articles, fitting and transforming the vector","metadata":{}},{"cell_type":"code","source":"tfidf_vectorizer = TfidfVectorizer()\ntfidf_article = tfidf_vectorizer.fit_transform((df_all['text'])) \ntfidf_article","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:40:34.547182Z","iopub.execute_input":"2022-06-11T16:40:34.547402Z","iopub.status.idle":"2022-06-11T16:40:38.907317Z","shell.execute_reply.started":"2022-06-11T16:40:34.547375Z","shell.execute_reply":"2022-06-11T16:40:38.906389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Transactions Dataset","metadata":{}},{"cell_type":"code","source":"transactions = transactions.dropna()","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:40:38.909067Z","iopub.execute_input":"2022-06-11T16:40:38.909470Z","iopub.status.idle":"2022-06-11T16:40:48.383801Z","shell.execute_reply.started":"2022-06-11T16:40:38.909423Z","shell.execute_reply":"2022-06-11T16:40:48.382991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Sorting the dataset by customer id to see all of a customer's purchases","metadata":{}},{"cell_type":"code","source":"transactions =  transactions.sort_values(by='customer_id')\ntransactions.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:40:48.385070Z","iopub.execute_input":"2022-06-11T16:40:48.385280Z","iopub.status.idle":"2022-06-11T16:42:04.787007Z","shell.execute_reply.started":"2022-06-11T16:40:48.385254Z","shell.execute_reply":"2022-06-11T16:42:04.786271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Merging the text&product_id (df_all) dataset with transactions dataset to match article_ids with purchases made by a customer.","metadata":{}},{"cell_type":"code","source":"merged_df = df_all.merge(transactions, how = 'inner', on = ['article_id'])","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:42:04.788135Z","iopub.execute_input":"2022-06-11T16:42:04.788466Z","iopub.status.idle":"2022-06-11T16:42:28.032124Z","shell.execute_reply.started":"2022-06-11T16:42:04.788438Z","shell.execute_reply":"2022-06-11T16:42:28.031190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The text information of all the products purchased by the user are gathered in the same 'text' variable.","metadata":{}},{"cell_type":"code","source":"merged_df2 = merged_df.groupby('customer_id', sort=False)['text'].apply(' '.join).reset_index()\nmerged_df2.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:42:28.033358Z","iopub.execute_input":"2022-06-11T16:42:28.033672Z","iopub.status.idle":"2022-06-11T16:43:44.260058Z","shell.execute_reply.started":"2022-06-11T16:42:28.033642Z","shell.execute_reply":"2022-06-11T16:43:44.259256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Recommendation**","metadata":{"tags":[]}},{"cell_type":"markdown","source":"A random customer_id was chosen to make a reccommendation","metadata":{}},{"cell_type":"code","source":"u = \"000058a12d5b43e67d225668fa1f8d618c13dc232df0cad8ffe7ad4a1091e318\" #customer_id\nindex = np.where(merged_df2['customer_id'] == u)[0][0]\ncust_q = merged_df2.iloc[[index]]\ncust_q","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:43:44.261216Z","iopub.execute_input":"2022-06-11T16:43:44.261430Z","iopub.status.idle":"2022-06-11T16:43:44.925102Z","shell.execute_reply.started":"2022-06-11T16:43:44.261403Z","shell.execute_reply":"2022-06-11T16:43:44.924322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Products user bought before","metadata":{"tags":[]}},{"cell_type":"code","source":"transactions.loc[transactions['customer_id'] == u]","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:43:44.926996Z","iopub.execute_input":"2022-06-11T16:43:44.928123Z","iopub.status.idle":"2022-06-11T16:43:50.335456Z","shell.execute_reply.started":"2022-06-11T16:43:44.928071Z","shell.execute_reply":"2022-06-11T16:43:50.334620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Define a Reccommendation Function ","metadata":{}},{"cell_type":"markdown","source":"Recommendation function includes customer ID, article ID, product code, description and similarity score.","metadata":{}},{"cell_type":"code","source":"def recommendation_product(top, df_all, scores):\n  recommendation = pd.DataFrame(columns = ['customer_id', 'article_id',  'product_code', 'detail_desc', 'score'])\n  count = 0\n  for i in top:\n      recommendation.at[count, 'customer_id'] = u\n      recommendation.at[count, 'article_id'] = df_all['article_id'][i]\n      recommendation.at[count, 'product_code'] = df_all['product_code'][i]\n      recommendation.at[count, 'detail_desc'] = articles['detail_desc'][i]   \n      recommendation.at[count, 'score'] =  scores[count]\n      count += 1\n  return recommendation","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:43:50.336754Z","iopub.execute_input":"2022-06-11T16:43:50.337739Z","iopub.status.idle":"2022-06-11T16:43:50.346009Z","shell.execute_reply.started":"2022-06-11T16:43:50.337688Z","shell.execute_reply":"2022-06-11T16:43:50.345109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Calculating Cosine Similarity for the User","metadata":{}},{"cell_type":"code","source":"user_tfidf = tfidf_vectorizer.transform(cust_q['text'])\ncos_similarity_tfidf = map(lambda x: cosine_similarity(user_tfidf, x),tfidf_article)","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:43:50.349254Z","iopub.execute_input":"2022-06-11T16:43:50.349483Z","iopub.status.idle":"2022-06-11T16:43:50.360715Z","shell.execute_reply.started":"2022-06-11T16:43:50.349457Z","shell.execute_reply":"2022-06-11T16:43:50.360139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output2 = list(cos_similarity_tfidf)","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:43:50.361674Z","iopub.execute_input":"2022-06-11T16:43:50.362256Z","iopub.status.idle":"2022-06-11T16:45:12.520974Z","shell.execute_reply.started":"2022-06-11T16:43:50.362222Z","shell.execute_reply":"2022-06-11T16:45:12.519983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Recommendations with TFIDF","metadata":{}},{"cell_type":"code","source":"top = sorted(range(len(output2)), key=lambda i: output2[i], reverse=True)[:10]\ntf_list_scores = [output2[i][0][0] for i in top]\nrecommendation_product(top, df_all, tf_list_scores)","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:45:12.522591Z","iopub.execute_input":"2022-06-11T16:45:12.522832Z","iopub.status.idle":"2022-06-11T16:45:13.975186Z","shell.execute_reply.started":"2022-06-11T16:45:12.522804Z","shell.execute_reply":"2022-06-11T16:45:13.974287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf_idf_score=pd.DataFrame(recommendation_product(top, df_all, tf_list_scores), columns = ['article_id', 'score'])","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:45:13.976315Z","iopub.execute_input":"2022-06-11T16:45:13.978790Z","iopub.status.idle":"2022-06-11T16:45:13.999725Z","shell.execute_reply.started":"2022-06-11T16:45:13.978753Z","shell.execute_reply":"2022-06-11T16:45:13.998362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reccomendations with CountVectorizer","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\ncount_vectorizer = CountVectorizer()\n\ncount_artid = count_vectorizer.fit_transform((df_all['text'])) #fitting and transforming the vector\ncount_artid","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:45:14.001371Z","iopub.execute_input":"2022-06-11T16:45:14.002258Z","iopub.status.idle":"2022-06-11T16:45:18.251605Z","shell.execute_reply.started":"2022-06-11T16:45:14.002209Z","shell.execute_reply":"2022-06-11T16:45:18.250874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics.pairwise import cosine_similarity\nuser_count = count_vectorizer.transform(cust_q['text'])\ncos_similarity_countv = map(lambda x: cosine_similarity(user_count, x),count_artid)","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:45:18.252657Z","iopub.execute_input":"2022-06-11T16:45:18.252971Z","iopub.status.idle":"2022-06-11T16:45:18.259587Z","shell.execute_reply.started":"2022-06-11T16:45:18.252943Z","shell.execute_reply":"2022-06-11T16:45:18.258627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output3 = list(cos_similarity_countv)","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:45:18.261003Z","iopub.execute_input":"2022-06-11T16:45:18.261458Z","iopub.status.idle":"2022-06-11T16:46:57.979406Z","shell.execute_reply.started":"2022-06-11T16:45:18.261411Z","shell.execute_reply":"2022-06-11T16:46:57.978419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top = sorted(range(len(output3)), key=lambda i: output3[i], reverse=True)[:10]\nlist_scores_cv = [output3[i][0][0] for i in top]\nrecommendation_product(top, df_all, list_scores_cv)","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:46:57.980688Z","iopub.execute_input":"2022-06-11T16:46:57.980942Z","iopub.status.idle":"2022-06-11T16:46:59.314814Z","shell.execute_reply.started":"2022-06-11T16:46:57.980896Z","shell.execute_reply":"2022-06-11T16:46:59.314008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv_score=pd.DataFrame(recommendation_product(top, df_all, list_scores_cv), columns = ['article_id', 'score'])","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:46:59.319127Z","iopub.execute_input":"2022-06-11T16:46:59.319378Z","iopub.status.idle":"2022-06-11T16:46:59.335061Z","shell.execute_reply.started":"2022-06-11T16:46:59.319346Z","shell.execute_reply":"2022-06-11T16:46:59.334346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reccommendations with KNN","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import NearestNeighbors\nKNN = NearestNeighbors(n_neighbors=11)\nKNN.fit(tfidf_article)\nNNs = KNN.kneighbors(user_tfidf, return_distance=True) ","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:46:59.336267Z","iopub.execute_input":"2022-06-11T16:46:59.336468Z","iopub.status.idle":"2022-06-11T16:46:59.484649Z","shell.execute_reply.started":"2022-06-11T16:46:59.336442Z","shell.execute_reply":"2022-06-11T16:46:59.483695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top = NNs[1][0][1:]\nindex_score = NNs[0][0][1:]\nrecommendation_product(top, df_all, index_score)","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:46:59.486021Z","iopub.execute_input":"2022-06-11T16:46:59.486269Z","iopub.status.idle":"2022-06-11T16:46:59.512844Z","shell.execute_reply.started":"2022-06-11T16:46:59.486238Z","shell.execute_reply":"2022-06-11T16:46:59.512081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"knn_score=pd.DataFrame(recommendation_product(top, df_all, index_score), columns = ['article_id', 'score'])","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:46:59.513859Z","iopub.execute_input":"2022-06-11T16:46:59.514077Z","iopub.status.idle":"2022-06-11T16:46:59.535196Z","shell.execute_reply.started":"2022-06-11T16:46:59.514051Z","shell.execute_reply":"2022-06-11T16:46:59.534150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Comparison ","metadata":{}},{"cell_type":"code","source":"tf_idf_score=tf_idf_score.rename(columns={\"score\":\"tf_idf_score\"})\ncv_score=cv_score.rename(columns={\"score\":\"cv_score\"})\nknn_score=knn_score.rename(columns={\"score\":\"knn_score\"})","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:46:59.536440Z","iopub.execute_input":"2022-06-11T16:46:59.536748Z","iopub.status.idle":"2022-06-11T16:46:59.545755Z","shell.execute_reply.started":"2022-06-11T16:46:59.536707Z","shell.execute_reply":"2022-06-11T16:46:59.544820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.concat([tf_idf_score, cv_score, knn_score], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-06-11T16:46:59.547169Z","iopub.execute_input":"2022-06-11T16:46:59.547474Z","iopub.status.idle":"2022-06-11T16:46:59.569000Z","shell.execute_reply.started":"2022-06-11T16:46:59.547432Z","shell.execute_reply":"2022-06-11T16:46:59.568323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It seems that while knn and tf-idf make **almost** the same recommendations, the system based on countvectorizer makes different recommendations.","metadata":{}}]}