{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Thank you BEN LEBOVITZ for sharing your Notebook https://www.kaggle.com/code/beezus666/k-means-and-feature-importance-for-articles","metadata":{}},{"cell_type":"markdown","source":"# Overview of this notebook\n1. Group users and articles in a bunch of k-means clusters\n2. Do a simple random forest to take a peek at the features and see what's useful for articles\n    1. In the DF that does this, I set up y = number of times an article was bought\n    2. This is set up as a very simple regression problem just to usee feature importance to see what features the RF was finding useful\n\nThe concept of clustering is key to recommender systems. Using cuML seems to be pretty good compared to CPU based solutions. I have a ways to go on this, but maybe it'll give someone else good ideas too.\n\nBonus... switch between cuML and cuDF and pandas/xgb/scikit/etc where a GPU will help.","metadata":{}},{"cell_type":"markdown","source":"# Overview of this notebook\n1. Implement a simple Random Forest to take a peek at the features and see what's useful for articles\n    1. Create Dataset that y = number of times an article was bought\n    2. This is set up as a very simple regression problem just to usee feature importance to see what features the RF was finding useful\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \npd.options.plotting.backend = \"matplotlib\"\nimport matplotlib.pyplot as plt\nfrom datetime import datetime, timedelta\nimport gc\nimport cudf\nfrom fastai.tabular.core import add_datepart #fails because of some issue with weeks in cudf?\nimport cupy as cp\nfrom cuml.cluster import KMeans\nfrom cuml.datasets import make_blobs\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-09T20:04:54.126233Z","iopub.execute_input":"2022-06-09T20:04:54.127055Z","iopub.status.idle":"2022-06-09T20:04:59.789297Z","shell.execute_reply.started":"2022-06-09T20:04:54.126510Z","shell.execute_reply":"2022-06-09T20:04:59.788563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load and group data\n1. Creating a count of how many times items were bought from the cusomer CSV so that we can use it later in the articles.\n\n2. Main idea is to use a simple Random Forest to predict how many times an article  will sell. \n","metadata":{}},{"cell_type":"markdown","source":"First need to build the feature of number of times sold from the transaction data.","metadata":{}},{"cell_type":"code","source":"#some nice ideas on reducing memory: https://www.kaggle.com/c/h-and-m-personalized-fashion-recommendations/discussion/308635\ntransactions = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv', parse_dates=['t_dat'])\ntransactions['customer_id'] = transactions['customer_id'].str[-16:].str.hex_to_int().astype('int64')\ntransactions['article_id'] = transactions.article_id.astype('int32')\ntransactions.t_dat = cudf.to_datetime(transactions.t_dat)\ntransactions = transactions[['t_dat','customer_id','article_id']]\n#transactions.to_parquet('train.pqt',index=False)\nprint( transactions.shape )\ntransactions.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:10:20.323604Z","iopub.execute_input":"2022-06-09T20:10:20.324011Z","iopub.status.idle":"2022-06-09T20:10:23.329832Z","shell.execute_reply.started":"2022-06-09T20:10:20.323973Z","shell.execute_reply":"2022-06-09T20:10:23.329130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = transactions.groupby(['customer_id','article_id'])['t_dat'].agg('count').reset_index()\ntmp.columns = ['customer_id','article_id','ct']\ntmp.head(4)","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:08:49.641015Z","iopub.execute_input":"2022-06-09T20:08:49.641279Z","iopub.status.idle":"2022-06-09T20:08:49.759911Z","shell.execute_reply.started":"2022-06-09T20:08:49.641244Z","shell.execute_reply":"2022-06-09T20:08:49.759214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions = transactions.merge(tmp,on=['customer_id','article_id'],how='left')\ntransactions = transactions.sort_values(['ct','t_dat'],ascending=False)\ntransactions = transactions.drop_duplicates(['customer_id','article_id'])\ntransactions = transactions.sort_values(['ct','t_dat'],ascending=False)\n","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:10:25.081074Z","iopub.execute_input":"2022-06-09T20:10:25.081705Z","iopub.status.idle":"2022-06-09T20:10:26.236246Z","shell.execute_reply.started":"2022-06-09T20:10:25.081659Z","shell.execute_reply":"2022-06-09T20:10:26.235539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.sample(3)","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:10:33.287660Z","iopub.execute_input":"2022-06-09T20:10:33.289773Z","iopub.status.idle":"2022-06-09T20:10:33.491397Z","shell.execute_reply.started":"2022-06-09T20:10:33.289731Z","shell.execute_reply":"2022-06-09T20:10:33.490570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**transformed t_date into specific columns like Year, Month, Day, Day of Week, Day of Year, is month end, is month sart**","metadata":{}},{"cell_type":"code","source":"transactions['year'] = transactions['t_dat'].dt.year\ntransactions['month'] = transactions['t_dat'].dt.month\ntransactions['day'] = transactions['t_dat'].dt.day\ntransactions['dayofweek'] = transactions['t_dat'].dt.dayofweek\ntransactions['dayofyear'] = transactions['t_dat'].dt.dayofyear\ntransactions['is_month_end'] = transactions['t_dat'].dt.is_month_end\ntransactions['is_month_start'] = transactions['t_dat'].dt.is_month_start\ntransactions.drop(columns=['t_dat'], inplace = True)\n\ntransactions.tail()","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:10:45.006556Z","iopub.execute_input":"2022-06-09T20:10:45.007231Z","iopub.status.idle":"2022-06-09T20:10:45.086815Z","shell.execute_reply.started":"2022-06-09T20:10:45.007195Z","shell.execute_reply":"2022-06-09T20:10:45.086143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We use Customer Id and encoded it, save it in cust_cat_df","metadata":{}},{"cell_type":"code","source":"transactions['cust_cat']= transactions['customer_id'].astype('category')\ntransactions['cat_codes'] = transactions['cust_cat'].cat.codes \ncust_cat_df = transactions[['customer_id', 'cust_cat', 'cat_codes']] #save them to put them back together later\nprint(cust_cat_df.dtypes)\ncust_cat_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:12:48.530933Z","iopub.execute_input":"2022-06-09T20:12:48.531387Z","iopub.status.idle":"2022-06-09T20:12:49.013120Z","shell.execute_reply.started":"2022-06-09T20:12:48.531352Z","shell.execute_reply":"2022-06-09T20:12:49.012428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.drop(columns=['cust_cat', 'cat_codes'], inplace = True)\ntransactions.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:13:44.761957Z","iopub.execute_input":"2022-06-09T20:13:44.762213Z","iopub.status.idle":"2022-06-09T20:13:44.769469Z","shell.execute_reply.started":"2022-06-09T20:13:44.762186Z","shell.execute_reply":"2022-06-09T20:13:44.768197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Put number sold into articles DF\nSo now all grouped and formatted, put it into the articles DF\n\nAlso need to foramt everythin numerically so that the models that I'm trying to use will take the DF","metadata":{}},{"cell_type":"code","source":"# Get a list of all unique article ids\narticles = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')\narticles.drop(columns=['detail_desc'], inplace = True)\narticles.shape, articles.dtypes\narticles","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:18:13.794121Z","iopub.execute_input":"2022-06-09T20:18:13.794380Z","iopub.status.idle":"2022-06-09T20:18:14.318507Z","shell.execute_reply.started":"2022-06-09T20:18:13.794352Z","shell.execute_reply":"2022-06-09T20:18:14.317837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:18:50.447409Z","iopub.execute_input":"2022-06-09T20:18:50.447684Z","iopub.status.idle":"2022-06-09T20:18:50.568125Z","shell.execute_reply.started":"2022-06-09T20:18:50.447653Z","shell.execute_reply":"2022-06-09T20:18:50.567335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_names= articles.select_dtypes(include=['object']).columns\ncont_names = articles.select_dtypes(include=['int64']).columns\nobj_names = articles.select_dtypes(include=['object']).columns\n\nfor i in cat_names: articles[i+'_cat']=articles[i].astype('category')\nfor i in obj_names: articles.drop(columns=[i], inplace = True)\n\narticles.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:18:56.606921Z","iopub.execute_input":"2022-06-09T20:18:56.607444Z","iopub.status.idle":"2022-06-09T20:18:56.802917Z","shell.execute_reply.started":"2022-06-09T20:18:56.607405Z","shell.execute_reply":"2022-06-09T20:18:56.802108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:20:47.499381Z","iopub.execute_input":"2022-06-09T20:20:47.499697Z","iopub.status.idle":"2022-06-09T20:20:48.013605Z","shell.execute_reply.started":"2022-06-09T20:20:47.499667Z","shell.execute_reply":"2022-06-09T20:20:48.012797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"times_bought = transactions[['article_id', 'ct']]\ntimes_bought = times_bought.groupby('article_id', as_index = False).sum()\ntimes_bought.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:20:54.159477Z","iopub.execute_input":"2022-06-09T20:20:54.159921Z","iopub.status.idle":"2022-06-09T20:20:54.207050Z","shell.execute_reply.started":"2022-06-09T20:20:54.159884Z","shell.execute_reply":"2022-06-09T20:20:54.206183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles = articles.merge(times_bought,  how='left', on='article_id')\narticles['ct'] = articles['ct'].fillna(0)\narticles.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:21:14.493456Z","iopub.execute_input":"2022-06-09T20:21:14.493759Z","iopub.status.idle":"2022-06-09T20:21:14.980205Z","shell.execute_reply.started":"2022-06-09T20:21:14.493726Z","shell.execute_reply":"2022-06-09T20:21:14.979500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles.select_dtypes(include=['category']).columns","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:21:50.294496Z","iopub.execute_input":"2022-06-09T20:21:50.295051Z","iopub.status.idle":"2022-06-09T20:21:50.304162Z","shell.execute_reply.started":"2022-06-09T20:21:50.295013Z","shell.execute_reply":"2022-06-09T20:21:50.303524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_names= articles.select_dtypes(include=['category']).columns\narticle_cat_df = cudf.DataFrame()\n\nfor i in cat_names: \n    articles[i+'_cat_code'] = articles[i].cat.codes\n    \n    #save them to put them back together later\n    article_cat_df[i] = articles[i]    \n    article_cat_df[i+'cat_code'] =articles[i+'_cat_code']","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:21:55.796676Z","iopub.execute_input":"2022-06-09T20:21:55.796944Z","iopub.status.idle":"2022-06-09T20:21:55.810955Z","shell.execute_reply.started":"2022-06-09T20:21:55.796914Z","shell.execute_reply":"2022-06-09T20:21:55.809685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:22:34.549205Z","iopub.execute_input":"2022-06-09T20:22:34.549468Z","iopub.status.idle":"2022-06-09T20:22:34.670462Z","shell.execute_reply.started":"2022-06-09T20:22:34.549440Z","shell.execute_reply":"2022-06-09T20:22:34.669680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"article_cat_df","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:22:45.526242Z","iopub.execute_input":"2022-06-09T20:22:45.526491Z","iopub.status.idle":"2022-06-09T20:22:46.288107Z","shell.execute_reply.started":"2022-06-09T20:22:45.526463Z","shell.execute_reply":"2022-06-09T20:22:46.287448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles.columns","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:23:21.115092Z","iopub.execute_input":"2022-06-09T20:23:21.115350Z","iopub.status.idle":"2022-06-09T20:23:21.121465Z","shell.execute_reply.started":"2022-06-09T20:23:21.115321Z","shell.execute_reply":"2022-06-09T20:23:21.120784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#something with cudf that this needs to be in a loop, very fast anyway\nfor i in cat_names:\n    articles.drop(columns=[i], inplace = True)\narticles.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:23:40.964579Z","iopub.execute_input":"2022-06-09T20:23:40.965072Z","iopub.status.idle":"2022-06-09T20:23:40.974347Z","shell.execute_reply.started":"2022-06-09T20:23:40.965036Z","shell.execute_reply":"2022-06-09T20:23:40.973600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#kmeans can't handle integers, so convert to float\nint64s = articles.select_dtypes(include=['int64']).columns\nfor i in int64s:\n    articles[i] = articles[i].astype(float)","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:23:52.198258Z","iopub.execute_input":"2022-06-09T20:23:52.198522Z","iopub.status.idle":"2022-06-09T20:23:52.212049Z","shell.execute_reply.started":"2022-06-09T20:23:52.198494Z","shell.execute_reply":"2022-06-09T20:23:52.211318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:24:37.038806Z","iopub.execute_input":"2022-06-09T20:24:37.039081Z","iopub.status.idle":"2022-06-09T20:24:37.051720Z","shell.execute_reply.started":"2022-06-09T20:24:37.039050Z","shell.execute_reply":"2022-06-09T20:24:37.051040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Simple random forest to predict # of sales\nMistake from earlier version... we want to do a simple RF BEFORE doing k-means so that we can figure out what features are important to pass to k-means..\n\nAll we're doing here is using RF to predcit the number of sales (the \"ct\" column).","metadata":{}},{"cell_type":"code","source":"# prepare X and y for random forest below\ncols_list = articles.columns\ncols_list = cols_list.to_list()\ncols_list.remove('ct')","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:24:45.666109Z","iopub.execute_input":"2022-06-09T20:24:45.666370Z","iopub.status.idle":"2022-06-09T20:24:45.671567Z","shell.execute_reply.started":"2022-06-09T20:24:45.666341Z","shell.execute_reply":"2022-06-09T20:24:45.670865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert to pandas, use scikit learn random forest to see what features are useful\n# unfortunately cuML doesn't have feature importance in their random forest yet...\nX = articles[cols_list].to_pandas()\ny = articles['ct'].to_pandas()","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:24:55.256765Z","iopub.execute_input":"2022-06-09T20:24:55.257031Z","iopub.status.idle":"2022-06-09T20:24:55.281905Z","shell.execute_reply.started":"2022-06-09T20:24:55.257001Z","shell.execute_reply":"2022-06-09T20:24:55.281239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:26:33.818060Z","iopub.execute_input":"2022-06-09T20:26:33.818552Z","iopub.status.idle":"2022-06-09T20:26:33.824047Z","shell.execute_reply.started":"2022-06-09T20:26:33.818513Z","shell.execute_reply":"2022-06-09T20:26:33.823266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:52:35.175793Z","iopub.execute_input":"2022-06-09T20:52:35.176121Z","iopub.status.idle":"2022-06-09T20:52:35.182254Z","shell.execute_reply.started":"2022-06-09T20:52:35.176084Z","shell.execute_reply":"2022-06-09T20:52:35.181500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.33, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:47:33.858905Z","iopub.execute_input":"2022-06-09T20:47:33.859156Z","iopub.status.idle":"2022-06-09T20:47:33.890937Z","shell.execute_reply.started":"2022-06-09T20:47:33.859128Z","shell.execute_reply":"2022-06-09T20:47:33.890241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.datasets import make_regression\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom matplotlib import pyplot\n\nmodel = RandomForestRegressor()\n# fit the model\nmodel.fit(X_train, y_train)\n# get importance","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:53:04.065935Z","iopub.execute_input":"2022-06-09T20:53:04.066432Z","iopub.status.idle":"2022-06-09T20:54:02.071001Z","shell.execute_reply.started":"2022-06-09T20:53:04.066394Z","shell.execute_reply":"2022-06-09T20:54:02.070292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:54:02.072674Z","iopub.execute_input":"2022-06-09T20:54:02.072930Z","iopub.status.idle":"2022-06-09T20:54:03.381958Z","shell.execute_reply.started":"2022-06-09T20:54:02.072897Z","shell.execute_reply":"2022-06-09T20:54:03.381147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum((y_test - y_pred) ** 2) / len(y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:56:47.192361Z","iopub.execute_input":"2022-06-09T20:56:47.192709Z","iopub.status.idle":"2022-06-09T20:56:47.208960Z","shell.execute_reply.started":"2022-06-09T20:56:47.192669Z","shell.execute_reply":"2022-06-09T20:56:47.208327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Mean Squared Error (MSE):', metrics.mean_squared_error(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:54:57.676543Z","iopub.execute_input":"2022-06-09T20:54:57.676871Z","iopub.status.idle":"2022-06-09T20:54:57.683697Z","shell.execute_reply.started":"2022-06-09T20:54:57.676838Z","shell.execute_reply":"2022-06-09T20:54:57.682904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fi_plot_df = pd.DataFrame({'cols':X.columns, 'imp':model.feature_importances_}).sort_values('imp', ascending=False)    \nfi_plot_df.plot(kind=\"barh\", x = 'cols')","metadata":{"execution":{"iopub.status.busy":"2022-06-09T20:30:04.615458Z","iopub.execute_input":"2022-06-09T20:30:04.615726Z","iopub.status.idle":"2022-06-09T20:30:05.172097Z","shell.execute_reply.started":"2022-06-09T20:30:04.615697Z","shell.execute_reply":"2022-06-09T20:30:05.171448Z"},"trusted":true},"execution_count":null,"outputs":[]}]}