{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Overview of this notebook\n1. Group users and articles in a bunch of k-means clusters\n2. Do a simple random forest to take a peek at the features and see what's useful for articles\n    1. In the DF that does this, I set up y = number of times an article was bought\n    2. This is set up as a very simple regression problem just to usee feature importance to see what features the RF was finding useful\n\nThe concept of clustering is key to recommender systems. Using cuML seems to be pretty good compared to CPU based solutions. I have a ways to go on this, but maybe it'll give someone else good ideas too.\n\nBonus... switch between cuML and cuDF and pandas/xgb/scikit/etc where a GPU will help.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \npd.options.plotting.backend = \"matplotlib\"\nimport matplotlib.pyplot as plt\nfrom datetime import datetime, timedelta\nimport gc\nimport cudf\nfrom fastai.tabular.core import add_datepart #fails because of some issue with weeks in cudf?\nimport cupy as cp\nfrom cuml.cluster import KMeans\nfrom cuml.datasets import make_blobs\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-28T19:41:27.139955Z","iopub.execute_input":"2022-04-28T19:41:27.140368Z","iopub.status.idle":"2022-04-28T19:41:32.636731Z","shell.execute_reply.started":"2022-04-28T19:41:27.140286Z","shell.execute_reply":"2022-04-28T19:41:32.635970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load and group data\nCreating a count of how many times items were bought from the cusomer CSV so that we can use it later in the articles.\n\nBasically my idea is to use a simple random forest to predict how many times an article  will sell. First need to build the feature of number of times sold from the transaction data.","metadata":{}},{"cell_type":"code","source":"#some nice ideas on reducing memory: https://www.kaggle.com/c/h-and-m-personalized-fashion-recommendations/discussion/308635\ntransactions = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv', parse_dates=['t_dat'])\ntransactions['customer_id'] = transactions['customer_id'].str[-16:].str.hex_to_int().astype('int64')\ntransactions['article_id'] = transactions.article_id.astype('int32')\ntransactions.t_dat = cudf.to_datetime(transactions.t_dat)\ntransactions = transactions[['t_dat','customer_id','article_id']]\n#transactions.to_parquet('train.pqt',index=False)\nprint( transactions.shape )\ntransactions.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:43:33.695205Z","iopub.execute_input":"2022-04-28T19:43:33.696095Z","iopub.status.idle":"2022-04-28T19:44:18.186580Z","shell.execute_reply.started":"2022-04-28T19:43:33.696028Z","shell.execute_reply":"2022-04-28T19:44:18.185949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = transactions.groupby(['customer_id','article_id'])['t_dat'].agg('count').reset_index()\ntmp.columns = ['customer_id','article_id','ct']\ntmp.tail()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:44:18.189708Z","iopub.execute_input":"2022-04-28T19:44:18.192828Z","iopub.status.idle":"2022-04-28T19:44:18.317574Z","shell.execute_reply.started":"2022-04-28T19:44:18.192778Z","shell.execute_reply":"2022-04-28T19:44:18.316857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions = transactions.merge(tmp,on=['customer_id','article_id'],how='left')\ntransactions = transactions.sort_values(['ct','t_dat'],ascending=False)\ntransactions = transactions.drop_duplicates(['customer_id','article_id'])\ntransactions = transactions.sort_values(['ct','t_dat'],ascending=False)\ntransactions.tail()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:44:18.318716Z","iopub.execute_input":"2022-04-28T19:44:18.318937Z","iopub.status.idle":"2022-04-28T19:44:19.511575Z","shell.execute_reply.started":"2022-04-28T19:44:18.318907Z","shell.execute_reply":"2022-04-28T19:44:19.510895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions['year'] = transactions['t_dat'].dt.year\ntransactions['month'] = transactions['t_dat'].dt.month\ntransactions['day'] = transactions['t_dat'].dt.day\ntransactions['dayofweek'] = transactions['t_dat'].dt.dayofweek\ntransactions['dayofyear'] = transactions['t_dat'].dt.dayofyear\ntransactions['is_month_end'] = transactions['t_dat'].dt.is_month_end\ntransactions['is_month_start'] = transactions['t_dat'].dt.is_month_start\ntransactions.drop(columns=['t_dat'], inplace = True)\n\ntransactions.tail()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:44:19.513627Z","iopub.execute_input":"2022-04-28T19:44:19.513867Z","iopub.status.idle":"2022-04-28T19:44:19.585893Z","shell.execute_reply.started":"2022-04-28T19:44:19.513834Z","shell.execute_reply":"2022-04-28T19:44:19.585083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions['cust_cat']= transactions['customer_id'].astype('category')\ntransactions['cat_codes'] = transactions['cust_cat'].cat.codes \ncust_cat_df = transactions[['customer_id', 'cust_cat', 'cat_codes']] #save them to put them back together later\nprint(cust_cat_df.dtypes)\ncust_cat_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:44:19.587296Z","iopub.execute_input":"2022-04-28T19:44:19.587560Z","iopub.status.idle":"2022-04-28T19:44:20.030901Z","shell.execute_reply.started":"2022-04-28T19:44:19.587526Z","shell.execute_reply":"2022-04-28T19:44:20.030101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.drop(columns=['cust_cat', 'cat_codes'], inplace = True)\ntransactions.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:44:20.032362Z","iopub.execute_input":"2022-04-28T19:44:20.032617Z","iopub.status.idle":"2022-04-28T19:44:20.039864Z","shell.execute_reply.started":"2022-04-28T19:44:20.032584Z","shell.execute_reply":"2022-04-28T19:44:20.039073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Put number sold into articles DF\nSo now all grouped and formatted, put it into the articles DF\n\nAlso need to foramt everythin numerically so that the models that I'm trying to use will take the DF","metadata":{}},{"cell_type":"code","source":"# Get a list of all unique article ids\narticles = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')\narticles.drop(columns=['detail_desc'], inplace = True)\narticles.shape, articles.dtypes\narticles","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:44:20.041289Z","iopub.execute_input":"2022-04-28T19:44:20.041538Z","iopub.status.idle":"2022-04-28T19:44:20.661571Z","shell.execute_reply.started":"2022-04-28T19:44:20.041504Z","shell.execute_reply":"2022-04-28T19:44:20.660911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_names= articles.select_dtypes(include=['object']).columns\ncont_names = articles.select_dtypes(include=['int64']).columns\nobj_names = articles.select_dtypes(include=['object']).columns\n\nfor i in cat_names: articles[i+'_cat']=articles[i].astype('category')\nfor i in obj_names: articles.drop(columns=[i], inplace = True)\n\narticles.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:44:20.665234Z","iopub.execute_input":"2022-04-28T19:44:20.665998Z","iopub.status.idle":"2022-04-28T19:44:20.866548Z","shell.execute_reply.started":"2022-04-28T19:44:20.665960Z","shell.execute_reply":"2022-04-28T19:44:20.865887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:44:20.867895Z","iopub.execute_input":"2022-04-28T19:44:20.868150Z","iopub.status.idle":"2022-04-28T19:44:21.459522Z","shell.execute_reply.started":"2022-04-28T19:44:20.868118Z","shell.execute_reply":"2022-04-28T19:44:21.458882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"times_bought = transactions[['article_id', 'ct']]\ntimes_bought = times_bought.groupby('article_id', as_index = False).sum()\ntimes_bought.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:44:21.462026Z","iopub.execute_input":"2022-04-28T19:44:21.462502Z","iopub.status.idle":"2022-04-28T19:44:21.505807Z","shell.execute_reply.started":"2022-04-28T19:44:21.462464Z","shell.execute_reply":"2022-04-28T19:44:21.505066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles = articles.merge(times_bought,  how='left', on='article_id')\narticles['ct'] = articles['ct'].fillna(0)\narticles.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:44:21.508069Z","iopub.execute_input":"2022-04-28T19:44:21.508291Z","iopub.status.idle":"2022-04-28T19:44:21.991890Z","shell.execute_reply.started":"2022-04-28T19:44:21.508260Z","shell.execute_reply":"2022-04-28T19:44:21.990974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_names= articles.select_dtypes(include=['category']).columns\narticle_cat_df = cudf.DataFrame()\n\nfor i in cat_names: \n    articles[i+'_cat_code'] = articles[i].cat.codes\n    \n    #save them to put them back together later\n    article_cat_df[i] = articles[i]    \n    article_cat_df[i+'cat_code'] =articles[i+'_cat_code']\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:44:21.993226Z","iopub.execute_input":"2022-04-28T19:44:21.993467Z","iopub.status.idle":"2022-04-28T19:44:22.005510Z","shell.execute_reply.started":"2022-04-28T19:44:21.993433Z","shell.execute_reply":"2022-04-28T19:44:22.004756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"article_cat_df","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:44:22.006832Z","iopub.execute_input":"2022-04-28T19:44:22.007681Z","iopub.status.idle":"2022-04-28T19:44:22.782470Z","shell.execute_reply.started":"2022-04-28T19:44:22.007641Z","shell.execute_reply":"2022-04-28T19:44:22.781734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#something with cudf that this needs to be in a loop, very fast anyway\nfor i in cat_names:\n    articles.drop(columns=[i], inplace = True)\narticles.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:44:22.783807Z","iopub.execute_input":"2022-04-28T19:44:22.784225Z","iopub.status.idle":"2022-04-28T19:44:22.792969Z","shell.execute_reply.started":"2022-04-28T19:44:22.784186Z","shell.execute_reply":"2022-04-28T19:44:22.792321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#kmeans can't handle integers, so convert to float\nint64s = articles.select_dtypes(include=['int64']).columns\nfor i in int64s:\n    articles[i] = articles[i].astype(float)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:44:22.794165Z","iopub.execute_input":"2022-04-28T19:44:22.794513Z","iopub.status.idle":"2022-04-28T19:44:22.807179Z","shell.execute_reply.started":"2022-04-28T19:44:22.794476Z","shell.execute_reply":"2022-04-28T19:44:22.806435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Simple random forest to predict # of sales\nMistake from earlier version... we want to do a simple RF BEFORE doing k-means so that we can figure out what features are important to pass to k-means..\n\nAll we're doing here is using RF to predcit the number of sales (the \"ct\" column).","metadata":{}},{"cell_type":"code","source":"# prepare X and y for random forest below\ncols_list = articles.columns\ncols_list = cols_list.to_list()\ncols_list.remove('ct')","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:45:00.975691Z","iopub.execute_input":"2022-04-28T19:45:00.975967Z","iopub.status.idle":"2022-04-28T19:45:00.981495Z","shell.execute_reply.started":"2022-04-28T19:45:00.975938Z","shell.execute_reply":"2022-04-28T19:45:00.980735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert to pandas, use scikit learn random forest to see what features are useful\n# unfortunately cuML doesn't have feature importance in their random forest yet...\nX = articles[cols_list].to_pandas()\ny = articles['ct'].to_pandas()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:45:08.582709Z","iopub.execute_input":"2022-04-28T19:45:08.582961Z","iopub.status.idle":"2022-04-28T19:45:08.606647Z","shell.execute_reply.started":"2022-04-28T19:45:08.582934Z","shell.execute_reply":"2022-04-28T19:45:08.606005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.datasets import make_regression\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom matplotlib import pyplot\n# define dataset\n#X, y = make_regression(n_samples=1000, n_features=10, n_informative=5, random_state=1)\n# define the model\n#model = DecisionTreeRegressor()\nmodel = RandomForestRegressor()\n# fit the model\nmodel.fit(X, y)\n# get importance","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:45:11.213140Z","iopub.execute_input":"2022-04-28T19:45:11.213397Z","iopub.status.idle":"2022-04-28T19:46:39.573924Z","shell.execute_reply.started":"2022-04-28T19:45:11.213370Z","shell.execute_reply":"2022-04-28T19:46:39.573265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fi_plot_df = pd.DataFrame({'cols':X.columns, 'imp':model.feature_importances_}).sort_values('imp', ascending=False)    \nfi_plot_df.plot(kind=\"barh\", x = 'cols')","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:46:39.575420Z","iopub.execute_input":"2022-04-28T19:46:39.575734Z","iopub.status.idle":"2022-04-28T19:46:40.168585Z","shell.execute_reply.started":"2022-04-28T19:46:39.575699Z","shell.execute_reply":"2022-04-28T19:46:40.167882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# so from above, these are the columns that are important to pass to k-means\narticles = articles[['article_id', 'prod_name_cat_cat_code', 'product_code', 'department_no', 'colour_group_name_cat_cat_code', 'ct']]","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:50:43.338167Z","iopub.execute_input":"2022-04-28T19:50:43.338445Z","iopub.status.idle":"2022-04-28T19:50:43.344795Z","shell.execute_reply.started":"2022-04-28T19:50:43.338418Z","shell.execute_reply":"2022-04-28T19:50:43.343899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# K-means cluster\nUsing K-means to create a feature on articles","metadata":{}},{"cell_type":"code","source":"# elbow method to determine the best number of clusters\n# so fast on GPU!\nSum_of_squared_distances = []\nK = range(1, 10)\nfor num_clusters in K :\n kmeans = KMeans(n_clusters=num_clusters)\n kmeans.fit(articles)\n Sum_of_squared_distances.append(kmeans.inertia_)\nplt.plot(K,Sum_of_squared_distances,'bx-')\nplt.xlabel('Values of K') \nplt.ylabel('Sum of squared distances/Inertia') \nplt.title('Elbow Method For Optimal k')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:51:55.854461Z","iopub.execute_input":"2022-04-28T19:51:55.854709Z","iopub.status.idle":"2022-04-28T19:51:57.036232Z","shell.execute_reply.started":"2022-04-28T19:51:55.854683Z","shell.execute_reply":"2022-04-28T19:51:57.035558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# elbow looks like around 4\n# It was 12 in an an earlier version!\nkmeans_float = KMeans(n_clusters=4)\nkmeans_fit = kmeans_float.fit(articles)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:52:35.952206Z","iopub.execute_input":"2022-04-28T19:52:35.953123Z","iopub.status.idle":"2022-04-28T19:52:35.989633Z","shell.execute_reply.started":"2022-04-28T19:52:35.953066Z","shell.execute_reply":"2022-04-28T19:52:35.988929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"labels:\")\nprint(kmeans_float.labels_)\nprint(\"cluster_centers:\")\nprint(kmeans_float.cluster_centers_)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:52:36.374182Z","iopub.execute_input":"2022-04-28T19:52:36.374917Z","iopub.status.idle":"2022-04-28T19:52:36.610480Z","shell.execute_reply.started":"2022-04-28T19:52:36.374870Z","shell.execute_reply":"2022-04-28T19:52:36.609637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kmeans_float.fit_predict(articles)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:52:36.820363Z","iopub.execute_input":"2022-04-28T19:52:36.820604Z","iopub.status.idle":"2022-04-28T19:52:36.862368Z","shell.execute_reply.started":"2022-04-28T19:52:36.820577Z","shell.execute_reply":"2022-04-28T19:52:36.861711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = kmeans_float.labels_\n\n#Glue back to originaal data\narticles['clusters'] = labels","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:52:37.741553Z","iopub.execute_input":"2022-04-28T19:52:37.741805Z","iopub.status.idle":"2022-04-28T19:52:37.745963Z","shell.execute_reply.started":"2022-04-28T19:52:37.741774Z","shell.execute_reply":"2022-04-28T19:52:37.745296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles.clusters.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:52:38.288470Z","iopub.execute_input":"2022-04-28T19:52:38.288954Z","iopub.status.idle":"2022-04-28T19:52:38.306401Z","shell.execute_reply.started":"2022-04-28T19:52:38.288917Z","shell.execute_reply":"2022-04-28T19:52:38.305681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles.tail()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:52:40.666381Z","iopub.execute_input":"2022-04-28T19:52:40.667094Z","iopub.status.idle":"2022-04-28T19:52:40.698663Z","shell.execute_reply.started":"2022-04-28T19:52:40.667042Z","shell.execute_reply":"2022-04-28T19:52:40.697980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles.to_parquet('articles.parquet', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:52:41.042144Z","iopub.execute_input":"2022-04-28T19:52:41.042869Z","iopub.status.idle":"2022-04-28T19:52:41.142977Z","shell.execute_reply.started":"2022-04-28T19:52:41.042835Z","shell.execute_reply":"2022-04-28T19:52:41.141957Z"},"trusted":true},"execution_count":null,"outputs":[]}]}