{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reading subset of data and restrict only customer who have bought at least three transactions","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport plotly.graph_objects as go\nfrom skimage import io","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:34:29.210442Z","iopub.execute_input":"2022-03-24T14:34:29.210779Z","iopub.status.idle":"2022-03-24T14:34:29.802424Z","shell.execute_reply.started":"2022-03-24T14:34:29.210743Z","shell.execute_reply":"2022-03-24T14:34:29.801690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\narticles = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')\n# Randomely sample 1 Lakh records\nusers = df.sample(n=100000)","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:34:31.795824Z","iopub.execute_input":"2022-03-24T14:34:31.796374Z","iopub.status.idle":"2022-03-24T14:35:43.091972Z","shell.execute_reply.started":"2022-03-24T14:34:31.796326Z","shell.execute_reply":"2022-03-24T14:35:43.090872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Join the user data with article id\ndf = users.merge(articles, on='article_id')\ndf = df[['t_dat', 'customer_id', 'article_id', 'prod_name', 'product_type_name',\n       'product_group_name', \n       'graphical_appearance_name', 'colour_group_name',\n       'perceived_colour_value_name',\n       'perceived_colour_master_name',\n       'department_name', 'index_name',\n       'index_group_name', 'section_name',\n       'garment_group_name', 'detail_desc']]\n\nfeature_subset = ['product_group_name', \n       'graphical_appearance_name', 'colour_group_name',\n       'perceived_colour_value_name',\n       'perceived_colour_master_name',\n       'department_name', 'index_name',\n       'index_group_name', 'section_name',\n       'garment_group_name']","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:35:49.325006Z","iopub.execute_input":"2022-03-24T14:35:49.325284Z","iopub.status.idle":"2022-03-24T14:35:49.790106Z","shell.execute_reply.started":"2022-03-24T14:35:49.325255Z","shell.execute_reply":"2022-03-24T14:35:49.789238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# We will only subset features ignoring transaction date","metadata":{}},{"cell_type":"code","source":"Only_features = df[['customer_id', 'article_id'] + feature_subset]\ndummies_df = pd.get_dummies(Only_features, columns=feature_subset)","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:35:53.741585Z","iopub.execute_input":"2022-03-24T14:35:53.742120Z","iopub.status.idle":"2022-03-24T14:35:54.142033Z","shell.execute_reply.started":"2022-03-24T14:35:53.742074Z","shell.execute_reply":"2022-03-24T14:35:54.141097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dummies_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:35:58.241590Z","iopub.execute_input":"2022-03-24T14:35:58.241904Z","iopub.status.idle":"2022-03-24T14:35:58.265623Z","shell.execute_reply.started":"2022-03-24T14:35:58.241871Z","shell.execute_reply":"2022-03-24T14:35:58.265078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Minimum we will choose minimum a customer has to be doing three transactions\nminimum_transaction = 3\ngroupby_customer = dummies_df.groupby('customer_id')\n\n\nl = []\ncutomer_ids = []\narticle_ids = []\nfor key in groupby_customer.groups.keys():\n    temp = groupby_customer.get_group(key)\n    if temp.article_id.nunique() >= minimum_transaction:\n        l.append(temp.drop('article_id', axis=1).sum(numeric_only=True).values)\n        cutomer_ids.append(key)\n        article_ids.extend(temp.article_id.values.tolist())","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:36:19.210888Z","iopub.execute_input":"2022-03-24T14:36:19.211204Z","iopub.status.idle":"2022-03-24T14:36:54.766057Z","shell.execute_reply.started":"2022-03-24T14:36:19.211166Z","shell.execute_reply":"2022-03-24T14:36:54.765137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user_feature = pd.DataFrame(l, columns = dummies_df.columns[2:])\nnormalized_user_feature = user_feature.div(user_feature.sum(axis=1), axis=0)\nnormalized_user_feature.insert(0, 'customer_id', cutomer_ids)\nnormalized_user_feature = normalized_user_feature.set_index('customer_id')\nnormalized_user_feature","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:37:19.952086Z","iopub.execute_input":"2022-03-24T14:37:19.952370Z","iopub.status.idle":"2022-03-24T14:37:20.269192Z","shell.execute_reply.started":"2022-03-24T14:37:19.952336Z","shell.execute_reply":"2022-03-24T14:37:20.268286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item_feature = dummies_df.drop_duplicates(subset='article_id')\nitem_feature = item_feature[item_feature.article_id.isin(article_ids)].drop('customer_id', axis=1)\nitem_feature = item_feature.set_index('article_id')\nitem_feature","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:37:23.977674Z","iopub.execute_input":"2022-03-24T14:37:23.978500Z","iopub.status.idle":"2022-03-24T14:37:24.135169Z","shell.execute_reply.started":"2022-03-24T14:37:23.978459Z","shell.execute_reply":"2022-03-24T14:37:24.134345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = normalized_user_feature.dot(item_feature.T)\nscores","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:37:26.490545Z","iopub.execute_input":"2022-03-24T14:37:26.491333Z","iopub.status.idle":"2022-03-24T14:37:26.637778Z","shell.execute_reply.started":"2022-03-24T14:37:26.491294Z","shell.execute_reply":"2022-03-24T14:37:26.636911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# We will performing matrix decomposition ","metadata":{}},{"cell_type":"code","source":"\nfrom numpy.linalg import svd\nmatrix = scores.values\n","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:37:29.475958Z","iopub.execute_input":"2022-03-24T14:37:29.476730Z","iopub.status.idle":"2022-03-24T14:37:29.480606Z","shell.execute_reply.started":"2022-03-24T14:37:29.476689Z","shell.execute_reply":"2022-03-24T14:37:29.479761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"u, s, vh = svd(matrix, full_matrices=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:37:32.593880Z","iopub.execute_input":"2022-03-24T14:37:32.594662Z","iopub.status.idle":"2022-03-24T14:37:36.802977Z","shell.execute_reply.started":"2022-03-24T14:37:32.594612Z","shell.execute_reply":"2022-03-24T14:37:36.801952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(u.shape)\nprint(s.shape)\nprint(vh.shape)","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:37:36.805282Z","iopub.execute_input":"2022-03-24T14:37:36.805840Z","iopub.status.idle":"2022-03-24T14:37:36.812388Z","shell.execute_reply.started":"2022-03-24T14:37:36.805792Z","shell.execute_reply":"2022-03-24T14:37:36.811568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reconstructed_vectors = u @ np.diag(s) @ vh\nnp.allclose(reconstructed_vectors,matrix)","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:37:41.570740Z","iopub.execute_input":"2022-03-24T14:37:41.570995Z","iopub.status.idle":"2022-03-24T14:37:42.038804Z","shell.execute_reply.started":"2022-03-24T14:37:41.570968Z","shell.execute_reply":"2022-03-24T14:37:42.037949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# We will now Use Cosine Similarity to get the recommendations","metadata":{}},{"cell_type":"code","source":"# Find the highest similarity\ndef cosine_similarity(v,u):\n    return (v @ u)/ (np.linalg.norm(v) * np.linalg.norm(u))","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:37:45.649037Z","iopub.execute_input":"2022-03-24T14:37:45.649307Z","iopub.status.idle":"2022-03-24T14:37:45.654036Z","shell.execute_reply.started":"2022-03-24T14:37:45.649280Z","shell.execute_reply":"2022-03-24T14:37:45.653149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:37:49.665995Z","iopub.execute_input":"2022-03-24T14:37:49.666664Z","iopub.status.idle":"2022-03-24T14:37:49.695630Z","shell.execute_reply.started":"2022-03-24T14:37:49.666621Z","shell.execute_reply":"2022-03-24T14:37:49.694798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get similarity Scores","metadata":{}},{"cell_type":"code","source":"#We are writing a function when given a column name which is article we will get top 10 recommendations\ndef similarity_score_recommendation(column_value):\n    ranking = {}\n    highest_similarity = -np.inf\n    highest_sim_col = -1\n    for col in range(0,vh.shape[1]):\n        if column_value!=col:\n            similarity = cosine_similarity(vh[:,column_value], vh[:,col])\n            if similarity > highest_similarity:\n                highest_similarity = similarity\n                highest_sim_col = col\n                ranking[col] = highest_similarity\n    sorted_ranking = {k: v for k, v in sorted(ranking.items(), key=lambda item: item[1],reverse=True)[:10]}\n    article_recommendation = []\n    for key in sorted_ranking:\n        article_recommendation.append(scores.columns[key])\n    return article_recommendation","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:37:53.377495Z","iopub.execute_input":"2022-03-24T14:37:53.377770Z","iopub.status.idle":"2022-03-24T14:37:53.384425Z","shell.execute_reply.started":"2022-03-24T14:37:53.377734Z","shell.execute_reply":"2022-03-24T14:37:53.383537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_rcmnd_top_ten(customer_id, scores):\n    cutomer_scores = scores.loc[customer_id]\n    customer_prev_items = groupby_customer.get_group(customer_id)['article_id']\n    recommendations = []\n    for prev_items in customer_prev_items.iteritems():\n        items = prev_items[1]\n        score_idx = scores.columns.get_loc(items)\n        recommendation = similarity_score_recommendation(score_idx)\n        recommendations.extend(recommendation)\n    \n    return list(set(recommendations))[:10]\n        ","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:37:56.114630Z","iopub.execute_input":"2022-03-24T14:37:56.114886Z","iopub.status.idle":"2022-03-24T14:37:56.120241Z","shell.execute_reply.started":"2022-03-24T14:37:56.114857Z","shell.execute_reply":"2022-03-24T14:37:56.119276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def previous_transaction_articles(customer_id, scores):\n    cutomer_scores = scores.loc[customer_id]\n    customer_prev_items = groupby_customer.get_group(customer_id)['article_id']\n    prev_items = []\n    for item in customer_prev_items.iteritems():\n        prev_item = item[1]\n        prev_items.append(prev_item)\n    return prev_items\n        ","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:37:58.887713Z","iopub.execute_input":"2022-03-24T14:37:58.888509Z","iopub.status.idle":"2022-03-24T14:37:58.893688Z","shell.execute_reply.started":"2022-03-24T14:37:58.888466Z","shell.execute_reply":"2022-03-24T14:37:58.893129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We will try some random custoner-id\ncustomer_id = scores.index[356]\nrecommendation = get_rcmnd_top_ten(customer_id, scores)","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:39:39.615934Z","iopub.execute_input":"2022-03-24T14:39:39.616588Z","iopub.status.idle":"2022-03-24T14:39:39.923835Z","shell.execute_reply.started":"2022-03-24T14:39:39.616548Z","shell.execute_reply":"2022-03-24T14:39:39.923206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prev_items = previous_transaction_articles(customer_id, scores)","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:39:41.751719Z","iopub.execute_input":"2022-03-24T14:39:41.752313Z","iopub.status.idle":"2022-03-24T14:39:41.758433Z","shell.execute_reply.started":"2022-03-24T14:39:41.752259Z","shell.execute_reply":"2022-03-24T14:39:41.757699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Now we will plot our Recommendation\n","metadata":{}},{"cell_type":"code","source":"def plot_prev(prev_items):\n    fig = plt.figure(figsize=(20, 10))\n    for item, i in zip(prev_items, range(1, len(prev_items)+1)):\n        item = '0' + str(item)\n        sub = item[:3]\n        image = path + \"/\"+ sub + \"/\"+ item +\".jpg\"\n        image = plt.imread(image)\n        fig.add_subplot(1, 6, i)\n        plt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:38:12.906152Z","iopub.execute_input":"2022-03-24T14:38:12.906431Z","iopub.status.idle":"2022-03-24T14:38:12.911855Z","shell.execute_reply.started":"2022-03-24T14:38:12.906398Z","shell.execute_reply":"2022-03-24T14:38:12.911247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_rcmnd(rcmnds):\n    fig = plt.figure(figsize=(20, 10))\n    for item, i in zip(rcmnds, range(1, len(rcmnds)+1)):\n        item = '0' + str(item)\n        sub = item[:3]\n        image = path + \"/\"+ sub + \"/\"+ item +\".jpg\"\n        image = plt.imread(image)\n        fig.add_subplot(1, 10, i)\n        plt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:38:16.369803Z","iopub.execute_input":"2022-03-24T14:38:16.370608Z","iopub.status.idle":"2022-03-24T14:38:16.375879Z","shell.execute_reply.started":"2022-03-24T14:38:16.370566Z","shell.execute_reply":"2022-03-24T14:38:16.375223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"../input/h-and-m-personalized-fashion-recommendations/images\"","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:38:19.692836Z","iopub.execute_input":"2022-03-24T14:38:19.693176Z","iopub.status.idle":"2022-03-24T14:38:19.697659Z","shell.execute_reply.started":"2022-03-24T14:38:19.693117Z","shell.execute_reply":"2022-03-24T14:38:19.696716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_prev(prev_items)","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:39:48.025504Z","iopub.execute_input":"2022-03-24T14:39:48.026029Z","iopub.status.idle":"2022-03-24T14:39:49.472771Z","shell.execute_reply.started":"2022-03-24T14:39:48.025987Z","shell.execute_reply":"2022-03-24T14:39:49.471830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_rcmnd(recommendation)","metadata":{"execution":{"iopub.status.busy":"2022-03-24T14:39:56.235921Z","iopub.execute_input":"2022-03-24T14:39:56.236271Z","iopub.status.idle":"2022-03-24T14:40:00.543890Z","shell.execute_reply.started":"2022-03-24T14:39:56.236231Z","shell.execute_reply":"2022-03-24T14:40:00.542954Z"},"trusted":true},"execution_count":null,"outputs":[]}]}