{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# H&M Consumer Analytics-RFM Segmentation and Collaborative Filtering","metadata":{"id":"gGziO4HliSQ7"}},{"cell_type":"markdown","source":"### We have applied consumer analytics on H&M Data set\n\n***Overview:\n\n\n* Overview of Data\n* Handling Missing value treatment / Feature Engineering\n \nSolution Approach\nRFM\nRecommender Algorithm","metadata":{"id":"ZzY1c7H5iSRC"}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\n \n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nplt.style.use('seaborn-white')\nsns.set_style(\"whitegrid\")\nsns.despine()\nplt.rc(\"figure\", autolayout=True)\nplt.rc(\"axes\", labelweight=\"bold\", labelsize=\"large\", titleweight=\"bold\", titlesize=14, titlepad=10)\n\nimport matplotlib as mpl\n\nmpl.rcParams['axes.spines.left'] = False\nmpl.rcParams['axes.spines.right'] = False\nmpl.rcParams['axes.spines.top'] = False\nmpl.rcParams['axes.spines.bottom'] = False\nplt.rcParams[\"font.weight\"] = \"bold\"\nplt.rcParams[\"axes.labelweight\"] = \"bold\"","metadata":{"id":"Y9qNgocBiSRS","outputId":"dbfba3d3-2932-48bb-c985-190486aea568","execution":{"iopub.status.busy":"2022-05-06T18:17:52.245047Z","iopub.execute_input":"2022-05-06T18:17:52.245377Z","iopub.status.idle":"2022-05-06T18:17:52.259724Z","shell.execute_reply.started":"2022-05-06T18:17:52.245341Z","shell.execute_reply":"2022-05-06T18:17:52.258836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data import","metadata":{"id":"Nxp_FoTOiSRU"}},{"cell_type":"code","source":"\n#articles = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\", \n #                      encoding=\"ISO-8859-1\", header=0)\ncustomers = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/customers.csv\",\n                        encoding=\"ISO-8859-1\", header=0)\ntransactions =  pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\",\n                           encoding=\"ISO-8859-1\",dtype={'article_id':str}, header=0).drop_duplicates()","metadata":{"id":"UdyjcsSBiSRV","execution":{"iopub.status.busy":"2022-05-06T18:17:52.439101Z","iopub.execute_input":"2022-05-06T18:17:52.439417Z","iopub.status.idle":"2022-05-06T18:19:28.056088Z","shell.execute_reply.started":"2022-05-06T18:17:52.43938Z","shell.execute_reply":"2022-05-06T18:19:28.055131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Missing value imputation with median as we have outliers\ncustomers['age'].fillna(customers['age'].median(), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:19:28.057848Z","iopub.execute_input":"2022-05-06T18:19:28.058089Z","iopub.status.idle":"2022-05-06T18:19:28.106863Z","shell.execute_reply.started":"2022-05-06T18:19:28.058047Z","shell.execute_reply":"2022-05-06T18:19:28.105798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers=customers[['customer_id','age']].drop_duplicates()","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:19:28.108556Z","iopub.execute_input":"2022-05-06T18:19:28.109086Z","iopub.status.idle":"2022-05-06T18:19:29.219725Z","shell.execute_reply.started":"2022-05-06T18:19:28.109035Z","shell.execute_reply":"2022-05-06T18:19:29.218303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"age_bins = [15,26,36,46,56,66,100]\ncustomers['age'] = pd.cut(customers['age'], bins=age_bins, labels=['Below 26','26-35','36-45','46-55', '56-65', 'Above 65'])","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:19:29.223328Z","iopub.execute_input":"2022-05-06T18:19:29.22367Z","iopub.status.idle":"2022-05-06T18:19:29.275931Z","shell.execute_reply.started":"2022-05-06T18:19:29.223637Z","shell.execute_reply":"2022-05-06T18:19:29.275114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers.groupby(\"age\").agg(\"count\" ).round()","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:19:29.277261Z","iopub.execute_input":"2022-05-06T18:19:29.277508Z","iopub.status.idle":"2022-05-06T18:19:29.588779Z","shell.execute_reply.started":"2022-05-06T18:19:29.277479Z","shell.execute_reply":"2022-05-06T18:19:29.587943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# RFM Analysis\n","metadata":{"id":"iK6RuC7piSSb"}},{"cell_type":"code","source":"# import required libraries for clustering\nimport sklearn\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.cluster import KMeans\nfrom sklearn.metrics import silhouette_score\nfrom scipy.cluster.hierarchy import linkage\nfrom scipy.cluster.hierarchy import dendrogram\nfrom scipy.cluster.hierarchy import cut_tree","metadata":{"id":"0Mt2lGg6iSSb","execution":{"iopub.status.busy":"2022-05-06T18:19:29.590637Z","iopub.execute_input":"2022-05-06T18:19:29.590955Z","iopub.status.idle":"2022-05-06T18:19:29.59703Z","shell.execute_reply.started":"2022-05-06T18:19:29.590913Z","shell.execute_reply":"2022-05-06T18:19:29.596046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" \n\ntransactions['InvoiceDate'] = pd.to_datetime(transactions['t_dat'],format='%Y-%m-%d')\n ","metadata":{"id":"QQKqhkkJiSSb","execution":{"iopub.status.busy":"2022-05-06T18:19:29.59892Z","iopub.execute_input":"2022-05-06T18:19:29.599264Z","iopub.status.idle":"2022-05-06T18:19:34.82223Z","shell.execute_reply.started":"2022-05-06T18:19:29.599218Z","shell.execute_reply":"2022-05-06T18:19:34.821325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import datetime as dt","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:19:34.823794Z","iopub.execute_input":"2022-05-06T18:19:34.824014Z","iopub.status.idle":"2022-05-06T18:19:34.828183Z","shell.execute_reply.started":"2022-05-06T18:19:34.823987Z","shell.execute_reply":"2022-05-06T18:19:34.82699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start_date = dt.datetime(2020,6,1)\n\n# Filter transactions by date\ntransactions[\"t_dat\"] = pd.to_datetime(transactions[\"InvoiceDate\"])\ntransactions = transactions.loc[transactions[\"t_dat\"] >= start_date]","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:19:34.829246Z","iopub.execute_input":"2022-05-06T18:19:34.829465Z","iopub.status.idle":"2022-05-06T18:19:38.417953Z","shell.execute_reply.started":"2022-05-06T18:19:34.829432Z","shell.execute_reply":"2022-05-06T18:19:38.415967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.shape","metadata":{"id":"aF90clwKiSSb","execution":{"iopub.status.busy":"2022-05-06T18:19:38.421717Z","iopub.execute_input":"2022-05-06T18:19:38.421985Z","iopub.status.idle":"2022-05-06T18:19:38.428503Z","shell.execute_reply.started":"2022-05-06T18:19:38.421956Z","shell.execute_reply":"2022-05-06T18:19:38.427539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking df's missing value's attribution in %\ndf_null = round(100*(transactions.isna().sum())/len(transactions), 2)\ndf_null","metadata":{"id":"pLU8uCLyiSSc","outputId":"d0f4aefe-2b3e-4a98-b05c-d976dd89b03d","execution":{"iopub.status.busy":"2022-05-06T18:19:38.429874Z","iopub.execute_input":"2022-05-06T18:19:38.430317Z","iopub.status.idle":"2022-05-06T18:19:40.053427Z","shell.execute_reply.started":"2022-05-06T18:19:38.430263Z","shell.execute_reply":"2022-05-06T18:19:40.052521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##Generate Invoice ID as combination of Customer id and Transaction Date.\n#transactions['_ID'] = transactions['customer_id']  + transactions['InvoiceDate'].astype(str) \n\n#transactions['Invoice_id'] = pd.factorize(transactions['_ID'])[0]\n","metadata":{"id":"9Hg-XvvTiSSc","execution":{"iopub.status.busy":"2022-05-06T18:19:40.054699Z","iopub.execute_input":"2022-05-06T18:19:40.05492Z","iopub.status.idle":"2022-05-06T18:19:40.059513Z","shell.execute_reply.started":"2022-05-06T18:19:40.054893Z","shell.execute_reply":"2022-05-06T18:19:40.058402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.head()","metadata":{"id":"YXLJTz4AiSSc","outputId":"49fd7437-a2d2-43a1-d2b2-64f1e25c72c0","execution":{"iopub.status.busy":"2022-05-06T18:19:40.060944Z","iopub.execute_input":"2022-05-06T18:19:40.06131Z","iopub.status.idle":"2022-05-06T18:19:40.080723Z","shell.execute_reply.started":"2022-05-06T18:19:40.061271Z","shell.execute_reply":"2022-05-06T18:19:40.079875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"A_iSC30ciSSc","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#analysis_date = max(transactions['InvoiceDate']) + dt.timedelta(days= 1)\nanalysis_date=dt.datetime(2020,9,23)\nprint((analysis_date).date())","metadata":{"id":"80tJbpvtiSSd","outputId":"5a970194-3aef-4982-86d3-d81b7da699d3","execution":{"iopub.status.busy":"2022-05-06T18:19:40.08211Z","iopub.execute_input":"2022-05-06T18:19:40.082825Z","iopub.status.idle":"2022-05-06T18:19:40.095659Z","shell.execute_reply.started":"2022-05-06T18:19:40.082786Z","shell.execute_reply":"2022-05-06T18:19:40.09485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions['date']=transactions['InvoiceDate']","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:19:40.096957Z","iopub.execute_input":"2022-05-06T18:19:40.097822Z","iopub.status.idle":"2022-05-06T18:19:40.120246Z","shell.execute_reply.started":"2022-05-06T18:19:40.097771Z","shell.execute_reply":"2022-05-06T18:19:40.119133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rfm = transactions.groupby('customer_id').agg({\n    'InvoiceDate': lambda x: (analysis_date - x.max()).days,\n    'date': 'count'\n    ,'price': 'sum'\n})\n#rfm.head()\nrfm.columns=[\"Recency\",\"Frequency\",\"Monetary\"]\nrfm = rfm[rfm[\"Monetary\"] > 0]\n \n #https://www.kaggle.com/code/kanberburak/rfm-analysis/notebook","metadata":{"id":"AEKQ9VhdiSSd","outputId":"dcfe0e46-e912-4834-f2dd-158486b80d25","scrolled":true,"execution":{"iopub.status.busy":"2022-05-06T18:19:40.12158Z","iopub.execute_input":"2022-05-06T18:19:40.122487Z","iopub.status.idle":"2022-05-06T18:21:09.658144Z","shell.execute_reply.started":"2022-05-06T18:19:40.122442Z","shell.execute_reply":"2022-05-06T18:21:09.657245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" transactions.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:21:09.659436Z","iopub.execute_input":"2022-05-06T18:21:09.659686Z","iopub.status.idle":"2022-05-06T18:21:09.674118Z","shell.execute_reply.started":"2022-05-06T18:21:09.659654Z","shell.execute_reply":"2022-05-06T18:21:09.673117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Date from customer's last purchase.The nearest date gets 5 and the furthest date gets 1.\nrfm[\"recency_score\"] = pd.qcut(rfm['Recency'], 5, labels=[5, 4, 3, 2, 1])\n# Total number of purchases.The least frequency gets 1 and the maximum frequency gets 5.\nrfm[\"frequency_score\"] = pd.qcut(rfm[\"Frequency\"].rank(method=\"first\"), 5, labels=[1, 2, 3, 4, 5])\n#Total spend by the customer.The least money gets 1, the most money gets 5.\nrfm[\"monetary_score\"]= pd.qcut(rfm[\"Monetary\"],5,labels=[1,2,3,4,5])\nrfm.head()","metadata":{"id":"pnRGfaemiSSe","outputId":"aa7c0df6-73e0-4cb7-bdaa-6019298e1c3f","execution":{"iopub.status.busy":"2022-05-06T18:21:09.675248Z","iopub.execute_input":"2022-05-06T18:21:09.675481Z","iopub.status.idle":"2022-05-06T18:21:10.051545Z","shell.execute_reply.started":"2022-05-06T18:21:09.675452Z","shell.execute_reply":"2022-05-06T18:21:10.050462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#RFM - The value of 2 different variables that were formed was recorded as a RFM_SCORE\nrfm[\"RFM_SCORE\"] = (rfm[\"recency_score\"].astype(str) + rfm[\"frequency_score\"].astype(str))","metadata":{"id":"a4vyxGgEiSSe","execution":{"iopub.status.busy":"2022-05-06T18:21:10.053351Z","iopub.execute_input":"2022-05-06T18:21:10.053954Z","iopub.status.idle":"2022-05-06T18:21:10.557539Z","shell.execute_reply.started":"2022-05-06T18:21:10.053918Z","shell.execute_reply":"2022-05-06T18:21:10.556361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seg_map = {\n    r'[1-2][1-2]': 'hibernating',\n    r'[1-2][3-4]': 'at_Risk',\n    r'[1-2]5': 'cant_loose',\n    r'3[1-2]': 'about_to_sleep',\n    r'33': 'need_attention',\n    r'[3-4][4-5]': 'loyal_customers',\n    r'41': 'promising',\n    r'51': 'new_customers',\n    r'[4-5][2-3]': 'potential_loyalists',\n    r'5[4-5]': 'champions'\n}\nrfm['segment'] = rfm['RFM_SCORE'].replace(seg_map, regex=True)\nrfm.head()","metadata":{"id":"Ro2LwGlxiSSe","outputId":"49c3de02-fd99-4ac3-c035-02722e7db4d1","execution":{"iopub.status.busy":"2022-05-06T18:21:10.559024Z","iopub.execute_input":"2022-05-06T18:21:10.559294Z","iopub.status.idle":"2022-05-06T18:21:25.227623Z","shell.execute_reply.started":"2022-05-06T18:21:10.55926Z","shell.execute_reply":"2022-05-06T18:21:25.226673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rfm[[\"segment\", \"Recency\",\"Frequency\",\"Monetary\"]].groupby(\"segment\").agg([\"mean\",\"count\",\"max\"]).round()","metadata":{"id":"9ttdWHb3iSSe","outputId":"1c5914c8-6af9-4344-f980-4ec268636b3e","execution":{"iopub.status.busy":"2022-05-06T18:21:25.228985Z","iopub.execute_input":"2022-05-06T18:21:25.229465Z","iopub.status.idle":"2022-05-06T18:21:25.428719Z","shell.execute_reply.started":"2022-05-06T18:21:25.229412Z","shell.execute_reply":"2022-05-06T18:21:25.427781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\n","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:21:25.430124Z","iopub.execute_input":"2022-05-06T18:21:25.430441Z","iopub.status.idle":"2022-05-06T18:21:25.435101Z","shell.execute_reply.started":"2022-05-06T18:21:25.430397Z","shell.execute_reply":"2022-05-06T18:21:25.434269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" \nx = rfm.segment.value_counts()\nfig = px.treemap(x, path=[x.index], values=x)\nfig.update_layout(title_text='Distribution of the RFM Segments', title_x=0.5,\n                  title_font=dict(size=20))\nfig.update_traces(textinfo=\"label+value+percent root\")\nfig.show()","metadata":{"id":"PxWJfAXYiSSe","outputId":"a401b730-69e5-4111-92df-ce217d745761","execution":{"iopub.status.busy":"2022-05-06T18:21:25.436334Z","iopub.execute_input":"2022-05-06T18:21:25.436639Z","iopub.status.idle":"2022-05-06T18:21:25.601458Z","shell.execute_reply.started":"2022-05-06T18:21:25.436598Z","shell.execute_reply":"2022-05-06T18:21:25.600531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Recommend Items Frequently Purchased Together\n","metadata":{"id":"H2OHNccQiSSk"}},{"cell_type":"markdown","source":"Item-Item Based Collaborative Filtering\n\n* Objective-To produce recommendations of Items for Hibernating customer-User 5(from RFM) for their upcoming purchase. \n\nStep 1- Matrix Factorization.\nThe Entries in table are based on time-adjusted count of # of purchase by user- item A bought 5 times on the first day of the train period is inferior to item B bought 4 times on the last day of the train period. This is done by weighted down exponentially by day of purchase\n\nStep 2- See Recommendation as optimization problem, Rating Prediction-make good Recommendation or prediction.\nQuantify Goodness using RMSE:\nLower RMSE =>better recommendation.\nWant to make good recommendation on items that user has not yet seen or purchase before(example – Hibernating customer-User 5 from RFM). Purely based on Popularity of the item. How we do?\n\nLet’s build a system such that it works well on known (User, Product) rating/purchase counts. And hope the system will also predict well the unknown ratings.\n\nDone by optimization method– Epoch. Then use this system to predict/recommend items unknown users\n\nUse Latent Factor Model like SVD to Dimension Reduction, handling nulls.\nNow this is can be assumed as vector space in 2D\nAnd we can calculate the distant of two point using Cosine-get the nearest neighbor.\n\nStep 6-Use this system/model to predict hibernating users-User 5 recommendation.\n\nConcept based on -\n\nhttps://www.youtube.com/watch?v=E8aMcwmqsTg \n\n\n\nhttps://www.analyticsvidhya.com/blog/2021/07/recommendation-system-understanding-the-basic-concepts/#:~:text=A%20recommendation%20system%20is%20a,suggests%20relevant%20items%20to%20users.","metadata":{"id":"aCwziCbEiSSk"}},{"cell_type":"markdown","source":"![](https://github.com/techanalyst84/customer-analytics-project/blob/main/Recommendator%201.jpg?raw=true)","metadata":{}},{"cell_type":"markdown","source":"![](https://github.com/techanalyst84/customer-analytics-project/blob/main/Recommendator%202.JPG?raw=true)","metadata":{}},{"cell_type":"markdown","source":"![](https://github.com/techanalyst84/customer-analytics-project/blob/main/Recommendator%203.JPG?raw=true)","metadata":{"id":"194hOcuxiSSl"}},{"cell_type":"markdown","source":"# Item-Based Collaborative Filtering -using Probabilistic Matrix Factorization\n\n","metadata":{"id":"tQcnd5_eiSSp"}},{"cell_type":"markdown","source":"**Preparing the data** \nWe need to restrict the data respect to a minimum transaction date. In that way, we reduce the dimensionality of the problem and we get rid of transactions that are not important in terms of the time decaying popularity.\n\nAlso, we are getting rid of articles that have not been bought enough. (Minimum 10 purchases are required)\n\n\nhttps://www.kaggle.com/code/luisrodri97/item-based-collaborative-filtering","metadata":{"id":"G5IenImeiSSp"}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport datetime\nfrom tqdm import tqdm","metadata":{"executionInfo":{"elapsed":12,"status":"aborted","timestamp":1651256340248,"user":{"displayName":"Suneeth Sreedharan","userId":"09929491400446734699"},"user_tz":240},"id":"zH6MK-OKiSS_","execution":{"iopub.status.busy":"2022-05-06T18:21:25.602586Z","iopub.execute_input":"2022-05-06T18:21:25.602809Z","iopub.status.idle":"2022-05-06T18:21:25.607826Z","shell.execute_reply.started":"2022-05-06T18:21:25.602775Z","shell.execute_reply":"2022-05-06T18:21:25.606942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rfm=rfm.reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:21:25.608892Z","iopub.execute_input":"2022-05-06T18:21:25.609129Z","iopub.status.idle":"2022-05-06T18:21:25.665166Z","shell.execute_reply.started":"2022-05-06T18:21:25.609094Z","shell.execute_reply":"2022-05-06T18:21:25.664106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntransactions=pd.merge(transactions,rfm[[\"customer_id\",\"segment\"]],how='inner',on='customer_id')\ntraining_segment = ['champions', 'potential_loyalists', 'new_customers','promising','loyal_customers']\ntransactions = transactions[transactions['segment'].isin(training_segment)]\ntransactions=transactions.drop('segment', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:21:25.666522Z","iopub.execute_input":"2022-05-06T18:21:25.66677Z","iopub.status.idle":"2022-05-06T18:21:33.824282Z","shell.execute_reply.started":"2022-05-06T18:21:25.666739Z","shell.execute_reply":"2022-05-06T18:21:33.823235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start_date = datetime.datetime(2020,9,1)\n\n# Filter transactions by date\ntransactions[\"t_dat\"] = pd.to_datetime(transactions[\"InvoiceDate\"])\ntransactions = transactions.loc[transactions[\"t_dat\"] >= start_date]","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:21:33.827211Z","iopub.execute_input":"2022-05-06T18:21:33.827809Z","iopub.status.idle":"2022-05-06T18:21:34.225755Z","shell.execute_reply.started":"2022-05-06T18:21:33.827756Z","shell.execute_reply":"2022-05-06T18:21:34.224613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:21:34.230406Z","iopub.execute_input":"2022-05-06T18:21:34.23069Z","iopub.status.idle":"2022-05-06T18:21:34.643027Z","shell.execute_reply.started":"2022-05-06T18:21:34.230658Z","shell.execute_reply":"2022-05-06T18:21:34.641958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.count()","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:21:34.644654Z","iopub.execute_input":"2022-05-06T18:21:34.644907Z","iopub.status.idle":"2022-05-06T18:21:34.853752Z","shell.execute_reply.started":"2022-05-06T18:21:34.644876Z","shell.execute_reply":"2022-05-06T18:21:34.852967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Filter transactions by number of an article has been bought\narticle_bought_count = transactions[['article_id', 'InvoiceDate']].groupby('article_id').count().reset_index().rename(columns={'InvoiceDate': 'count'})\nmost_bought_articles = article_bought_count[article_bought_count['count']>8]['article_id'].values\ntransactions = transactions[transactions['article_id'].isin(most_bought_articles)]\ntransactions[\"bought\"]=1 ","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:21:34.855079Z","iopub.execute_input":"2022-05-06T18:21:34.855322Z","iopub.status.idle":"2022-05-06T18:21:35.24575Z","shell.execute_reply.started":"2022-05-06T18:21:34.855292Z","shell.execute_reply":"2022-05-06T18:21:35.244861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"Due to the big amount of items, we can not consider the whole matrix in order to train. Therefore, we need to generate some negative samples: transactions that have never occured.\n\n","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"Model will be based on recommendations computed through the time decaying popularity and the most similar items to those items bought the most times by each user. Similarity among items is computed through cosine distance.\n\n","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics.pairwise import cosine_similarity\n\n\nclass ItemBased_RecSys:\n    ''' Collaborative filtering using a custom sim(u,u'). '''\n\n    def __init__(self, positive_transactions, negative_transactions, num_components=10):\n        ''' Constructor '''\n        self.positive_transactions = positive_transactions\n        self.transactions = pd.concat([positive_transactions, negative_transactions])\n        self.customers = self.transactions.customer_id.values\n        self.articles = self.transactions.article_id.values\n        self.bought = self.transactions.bought.values\n        self.num_components = num_components\n\n        self.customer_id2index = {c: i for i, c in enumerate(np.unique(self.customers))}\n        self.article_id2index = {a: i for i, a in enumerate(np.unique(self.articles))}\n        \n    def __sdg__(self):\n        for idx in tqdm(self.training_indices):\n            # Get the current sample\n            customer_id = self.customers[idx]\n            article_id = self.articles[idx]\n            bought = self.bought[idx]\n\n            # Get the index of the user and the article\n            customer_index = self.customer_id2index[customer_id]\n            article_index = self.article_id2index[article_id]\n\n            # Compute the prediction and the error\n            prediction = self.predict_single(customer_index, article_index)\n            error = (bought - prediction) # error\n            \n            # Update latent factors in terms of the learning rate and the observed error\n            self.customers_latent_matrix[customer_index] += self.learning_rate * \\\n                                    (error * self.articles_latent_matrix[article_index] - \\\n                                     self.lmbda * self.customers_latent_matrix[customer_index])\n            self.articles_latent_matrix[article_index] += self.learning_rate * \\\n                                    (error * self.customers_latent_matrix[customer_index] - \\\n                                     self.lmbda * self.articles_latent_matrix[article_index])\n                \n                \n    def fit(self, n_epochs=10, learning_rate=0.001, lmbda=0.1):\n        ''' Compute the matrix factorization R = P x Q '''\n        self.learning_rate = learning_rate\n        self.lmbda = lmbda\n        n_samples = self.transactions.shape[0]\n        \n        # Initialize latent matrices\n        self.customers_latent_matrix = np.random.normal(scale=1., size=(len(np.unique(self.customers)), self.num_components))\n        self.articles_latent_matrix = np.random.normal(scale=1., size=(len(np.unique(self.articles)), self.num_components))\n\n        for epoch in range(n_epochs):\n            print('Epoch: {}'.format(epoch))\n            self.training_indices = np.arange(n_samples)\n            \n            # Shuffle training samples and follow stochastic gradient descent\n            np.random.shuffle(self.training_indices)\n            self.__sdg__()\n\n    def predict_single(self, customer_index, article_index):\n        ''' Make a prediction for an specific user and article '''\n        prediction = np.dot(self.customers_latent_matrix[customer_index], self.articles_latent_matrix[article_index])\n        prediction = np.clip(prediction, 0, 1)\n        \n        return prediction\n\n    def default_recommendation(self):\n        ''' Calculate time decaying popularity '''\n        # Calculate time decaying popularity. This leads to items bought more recently having more weight in the popularity list.\n        # In simple words, item A bought 5 times on the first day of the train period is inferior than item B bought 4 times on the last day of the train period.\n        self.positive_transactions['pop_factor'] = self.positive_transactions['t_dat'].apply(lambda x: 1/(datetime.datetime(2020,9,23) - x).days)\n        transactions_by_article = self.positive_transactions[['article_id', 'pop_factor']].groupby('article_id').sum().reset_index()\n        return transactions_by_article.sort_values(by='pop_factor', ascending=False)['article_id'].values[:12]\n\n\n    def predict(self, customers):\n        ''' Make recommendations '''\n        recommendations = []\n        self.articles_latent_matrix[np.isnan(self.articles_latent_matrix)] = 0\n        # Compute similarity matrix (cosine)\n        similarity_matrix = cosine_similarity(self.articles_latent_matrix, self.articles_latent_matrix, dense_output=False)\n\n        # Convert similarity matrix into a matrix containing the 12 most similar items' index for each item\n        similarity_matrix = np.argsort(similarity_matrix, axis=1)\n        similarity_matrix = similarity_matrix[:, -12:]\n\n        # Get default recommendation (time decay popularity)\n        default_recommendation = self.default_recommendation()\n\n        # Group articles by user and articles to compute the number of times each article has been bought by each user\n        transactions_by_customer = self.positive_transactions[['customer_id', 'article_id', 'bought']].groupby(['customer_id', 'article_id']).count().reset_index()\n        most_bought_article = transactions_by_customer.loc[transactions_by_customer.groupby('customer_id').bought.idxmax()]['article_id'].values\n\n        # Make predictions\n        for customer in tqdm(customers):\n            try:\n                rec_aux1 = []\n                rec_aux2 = []\n                aux = []\n\n                # Retrieve the most bought article by customer\n                user_most_bought_article_id = most_bought_article[self.customer_id2index[customer]]\n\n                # Using the similarity matrix, get the 6 most similar articles\n                rec_aux1 = self.articles[similarity_matrix[self.article_id2index[user_most_bought_article_id]]]\n                # Return the half of the default recommendation\n                rec_aux2 = default_recommendation\n\n                # Merge half of both recommendation lists\n                for rec_idx in range(6):\n                    aux.append(rec_aux2[rec_idx])\n                    aux.append(rec_aux1[rec_idx])\n\n                recommendations.append(' '.join(aux))\n            except:\n                # Return the default recommendation\n                recommendations.append(' '.join(default_recommendation))\n        \n        return pd.DataFrame({\n            'customer_id': customers,\n            'prediction': recommendations,\n        })","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:21:35.247674Z","iopub.execute_input":"2022-05-06T18:21:35.248138Z","iopub.status.idle":"2022-05-06T18:21:35.279955Z","shell.execute_reply.started":"2022-05-06T18:21:35.24809Z","shell.execute_reply":"2022-05-06T18:21:35.278957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"Define your hyperparameters and fit the model. Take into account that there are more customizable parameters in the data processing section.\n\n","metadata":{}},{"cell_type":"code","source":"transactions=pd.merge(transactions,customers,how='inner',on='customer_id')\n","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:21:35.281746Z","iopub.execute_input":"2022-05-06T18:21:35.282134Z","iopub.status.idle":"2022-05-06T18:21:36.526563Z","shell.execute_reply.started":"2022-05-06T18:21:35.282089Z","shell.execute_reply":"2022-05-06T18:21:36.525669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate negative samples\ndef generate_negative_samples(trans):\n    np.random.seed(0)\n    return pd.DataFrame({\n    'article_id': np.random.choice(trans.article_id.unique(), trans.shape[0]),\n    'customer_id': np.random.choice(trans.customer_id.unique(), trans.shape[0]),\n    'bought': np.zeros(trans.shape[0])\n        })","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:21:36.527961Z","iopub.execute_input":"2022-05-06T18:21:36.528241Z","iopub.status.idle":"2022-05-06T18:21:36.535121Z","shell.execute_reply.started":"2022-05-06T18:21:36.52821Z","shell.execute_reply":"2022-05-06T18:21:36.534127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:21:36.536662Z","iopub.execute_input":"2022-05-06T18:21:36.536998Z","iopub.status.idle":"2022-05-06T18:21:36.568696Z","shell.execute_reply.started":"2022-05-06T18:21:36.536951Z","shell.execute_reply":"2022-05-06T18:21:36.567597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.groupby(\"age\").agg(\"count\" ).round()","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:21:36.570401Z","iopub.execute_input":"2022-05-06T18:21:36.570755Z","iopub.status.idle":"2022-05-06T18:21:36.859382Z","shell.execute_reply.started":"2022-05-06T18:21:36.57071Z","shell.execute_reply":"2022-05-06T18:21:36.858346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_input = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv'\n                       ,encoding=\"ISO-8859-1\", dtype={'article_id':str},header=0  ) ","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:21:36.86123Z","iopub.execute_input":"2022-05-06T18:21:36.861513Z","iopub.status.idle":"2022-05-06T18:21:39.959991Z","shell.execute_reply.started":"2022-05-06T18:21:36.861483Z","shell.execute_reply":"2022-05-06T18:21:39.958945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_input=pd.merge(customers_input,customers,how='inner',on='customer_id')\n","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:21:39.961532Z","iopub.execute_input":"2022-05-06T18:21:39.961765Z","iopub.status.idle":"2022-05-06T18:21:41.961875Z","shell.execute_reply.started":"2022-05-06T18:21:39.961737Z","shell.execute_reply":"2022-05-06T18:21:41.96084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rec = ItemBased_RecSys(transactions[transactions['age'].isin(['Below 26'])], generate_negative_samples(transactions[transactions['age'].isin(['Below 26'])]),\n                       num_components=1000)\nrec.fit(n_epochs=10)\nrec_below_26 = rec.predict(customers_input[customers_input['age'].isin(['Below 26'])].customer_id.unique())","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:21:41.963167Z","iopub.execute_input":"2022-05-06T18:21:41.963401Z","iopub.status.idle":"2022-05-06T18:27:43.968931Z","shell.execute_reply.started":"2022-05-06T18:21:41.963373Z","shell.execute_reply":"2022-05-06T18:27:43.967929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rec = ItemBased_RecSys(transactions[transactions['age'].isin(['26-35'])], generate_negative_samples(transactions[transactions['age'].isin(['26-35'])]),\n                       num_components=1000)\nrec.fit(n_epochs=10)\nrec_26_35 = rec.predict(customers_input[customers_input['age'].isin(['26-35'])].customer_id.unique())","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:27:43.970182Z","iopub.execute_input":"2022-05-06T18:27:43.970518Z","iopub.status.idle":"2022-05-06T18:32:33.648619Z","shell.execute_reply.started":"2022-05-06T18:27:43.970483Z","shell.execute_reply":"2022-05-06T18:32:33.647606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rec = ItemBased_RecSys(transactions[transactions['age'].isin(['36-45'])], generate_negative_samples(transactions[transactions['age'].isin(['36-45'])]),\n                       num_components=1000)\nrec.fit(n_epochs=10)\nrec_36_45 = rec.predict(customers_input[customers_input['age'].isin(['36-45'])].customer_id.unique())","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:32:33.650265Z","iopub.execute_input":"2022-05-06T18:32:33.65094Z","iopub.status.idle":"2022-05-06T18:34:38.317286Z","shell.execute_reply.started":"2022-05-06T18:32:33.65089Z","shell.execute_reply":"2022-05-06T18:34:38.316328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rec = ItemBased_RecSys(transactions[transactions['age'].isin(['46-55'])], generate_negative_samples(transactions[transactions['age'].isin(['46-55'])]),\n                       num_components=1000)\nrec.fit(n_epochs=10)\nrec_46_55 = rec.predict(customers_input[customers_input['age'].isin(['46-55'])].customer_id.unique())","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:34:38.318892Z","iopub.execute_input":"2022-05-06T18:34:38.319252Z","iopub.status.idle":"2022-05-06T18:37:44.158028Z","shell.execute_reply.started":"2022-05-06T18:34:38.319204Z","shell.execute_reply":"2022-05-06T18:37:44.157118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rec = ItemBased_RecSys(transactions[transactions['age'].isin(['56-65'])], generate_negative_samples(transactions[transactions['age'].isin(['56-65'])]),\n                       num_components=1000)\nrec.fit(n_epochs=10)\nrec_56_65 = rec.predict(customers_input[customers_input['age'].isin(['56-65'])].customer_id.unique())","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:37:44.159152Z","iopub.execute_input":"2022-05-06T18:37:44.159382Z","iopub.status.idle":"2022-05-06T18:38:53.125433Z","shell.execute_reply.started":"2022-05-06T18:37:44.159354Z","shell.execute_reply":"2022-05-06T18:38:53.124246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rec = ItemBased_RecSys(transactions[transactions['age'].isin(['Above 65'])], generate_negative_samples(transactions[transactions['age'].isin(['Above 65'])]),\n                       num_components=1000)\nrec.fit(n_epochs=10)\nrec_above_65 = rec.predict(customers_input[customers_input['age'].isin(['Above 65'])].customer_id.unique())","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:38:53.126811Z","iopub.execute_input":"2022-05-06T18:38:53.127192Z","iopub.status.idle":"2022-05-06T18:39:08.764321Z","shell.execute_reply.started":"2022-05-06T18:38:53.127154Z","shell.execute_reply":"2022-05-06T18:39:08.763222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recommendations = pd.concat([rec_below_26, rec_26_35])\n\n","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:39:08.76593Z","iopub.execute_input":"2022-05-06T18:39:08.766286Z","iopub.status.idle":"2022-05-06T18:39:08.835817Z","shell.execute_reply.started":"2022-05-06T18:39:08.76624Z","shell.execute_reply":"2022-05-06T18:39:08.834882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recommendations= pd.concat([recommendations, rec_36_45])\n","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:39:08.836815Z","iopub.execute_input":"2022-05-06T18:39:08.837023Z","iopub.status.idle":"2022-05-06T18:39:08.929985Z","shell.execute_reply.started":"2022-05-06T18:39:08.836997Z","shell.execute_reply":"2022-05-06T18:39:08.928858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recommendations= pd.concat([recommendations, rec_46_55])\n","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:39:08.931163Z","iopub.execute_input":"2022-05-06T18:39:08.931414Z","iopub.status.idle":"2022-05-06T18:39:09.053175Z","shell.execute_reply.started":"2022-05-06T18:39:08.931383Z","shell.execute_reply":"2022-05-06T18:39:09.051981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recommendations= pd.concat([recommendations, rec_56_65])\n","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:39:09.05475Z","iopub.execute_input":"2022-05-06T18:39:09.055143Z","iopub.status.idle":"2022-05-06T18:39:09.245674Z","shell.execute_reply.started":"2022-05-06T18:39:09.055094Z","shell.execute_reply":"2022-05-06T18:39:09.244764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recommendations= pd.concat([recommendations, rec_above_65])","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:39:09.247136Z","iopub.execute_input":"2022-05-06T18:39:09.247379Z","iopub.status.idle":"2022-05-06T18:39:09.387615Z","shell.execute_reply.started":"2022-05-06T18:39:09.247351Z","shell.execute_reply":"2022-05-06T18:39:09.386659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recommendations=recommendations.drop_duplicates()","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:39:09.388987Z","iopub.execute_input":"2022-05-06T18:39:09.389261Z","iopub.status.idle":"2022-05-06T18:39:11.207886Z","shell.execute_reply.started":"2022-05-06T18:39:09.38923Z","shell.execute_reply":"2022-05-06T18:39:11.206709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recommendations.head(4)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:39:11.209292Z","iopub.execute_input":"2022-05-06T18:39:11.209637Z","iopub.status.idle":"2022-05-06T18:39:11.220911Z","shell.execute_reply.started":"2022-05-06T18:39:11.2096Z","shell.execute_reply":"2022-05-06T18:39:11.219957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recommendations.count()","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:48:21.083092Z","iopub.execute_input":"2022-05-06T18:48:21.083605Z","iopub.status.idle":"2022-05-06T18:48:21.423104Z","shell.execute_reply.started":"2022-05-06T18:48:21.083546Z","shell.execute_reply":"2022-05-06T18:48:21.422484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recommendations.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T18:39:11.222274Z","iopub.execute_input":"2022-05-06T18:39:11.2226Z","iopub.status.idle":"2022-05-06T18:39:25.248197Z","shell.execute_reply.started":"2022-05-06T18:39:11.222564Z","shell.execute_reply":"2022-05-06T18:39:25.247217Z"},"trusted":true},"execution_count":null,"outputs":[]}]}