{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# H&M Recommendation Customer Clustering by Kmeans","metadata":{}},{"cell_type":"markdown","source":"Thank you for looking my notebook.  \nThis notebook is made with reference to　[By HechtJP -Rule-Base-by-Customer-Age](https://www.kaggle.com/code/hechtjp/h-m-eda-rule-base-by-customer-age).\n\nThis notebook is my first public notebook.\nIf this notebook helps you, I would appreciate your comments and up votes.","metadata":{}},{"cell_type":"markdown","source":"## 1.Library","metadata":{}},{"cell_type":"code","source":"import sys, warnings, time, os, copy, gc, re, random, pickle, cudf\nwarnings.filterwarnings('ignore')\nfrom IPython.display import display\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\n# pd.set_option('display.max_rows', 50)\n# pd.set_option('display.max_columns', None)\n# pd.set_option(\"display.max_colwidth\", 10000)\nimport seaborn as sns\nsns.set()\nfrom pandas.io.json import json_normalize\nfrom pprint import pprint\nfrom pathlib import Path\nfrom tqdm import tqdm\ntqdm.pandas()\nfrom collections import Counter\nfrom datetime import datetime, timedelta\nimport cudf\n\nfrom sklearn.cluster import KMeans\nfrom sklearn import preprocessing","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:07:41.0093Z","iopub.execute_input":"2022-04-04T13:07:41.010061Z","iopub.status.idle":"2022-04-04T13:07:42.838759Z","shell.execute_reply.started":"2022-04-04T13:07:41.009936Z","shell.execute_reply":"2022-04-04T13:07:42.837711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.Clutering","metadata":{}},{"cell_type":"markdown","source":"There are preprocessing and clustering here.\n- preprocessing\n|columns|conversion|\n|:-:|:-:|\n|fashion_news_frequency|{np.nan :0, 'None':0, 'Monthly':1, 'Regularly':2}|\n|club_member_status|{np.nan :0, 'PRE-CREATE':1, 'ACTIVE':2, 'LEFT CLUB':-1}|\n|age|{np.nan:-1}|\n|FN|{np.nan:0}|\n|Active|{np.nan:0}|\n\n- clustering\n    - Normarization\n        1. StandardScaler\n        1. minMax\n        1. None\n    - parameters\n        1. random_state:2022\n        1. clusters:12\n","metadata":{}},{"cell_type":"code","source":"class Clustering_HandM():\n    def customers_preprocessing(self, customers, dropcol=['postal_code'] , **kwargs):\n        customers = customers.drop(dropcol, axis=1)\n        customers_col = list(customers.columns)\n        \n        if 'fashion_news_frequency' in customers_col :\n            customers['fashion_news_frequency'] = customers['fashion_news_frequency'].replace('NONE','None')\n            customers['fashion_news_frequency'] = customers['fashion_news_frequency'].replace({np.nan :0, 'None':0, 'Monthly':1, 'Regularly':2})\n            \n        if 'club_member_status' in customers_col:\n            customers['club_member_status'] = customers['club_member_status'].replace({np.nan :0, 'PRE-CREATE':1, 'ACTIVE':2, 'LEFT CLUB':-1})\n            \n        if 'age' in customers_col:\n            customers['age'] = customers['age'].fillna(-1)\n            \n        if 'FN' in customers_col:\n            customers['FN'] = customers['FN'].fillna(0)\n\n        if 'Active' in customers_col:\n            customers['Active'] = customers['Active'].fillna(0)\n            \n            print(f'###NULL DESCRIPTION###\\n{customers.isnull().sum()}')\n            \n        return customers\n    \n    def clustering(self, df, predcol, usecol, normmethod='StandardScaler', clusters=12, DEBUG=False):\n        \n        X = np.array(df[usecol])\n        \n        if normmethod == 'StandardScaler':\n            nm = preprocessing.StandardScaler()\n            X = nm.fit_transform(X)\n        elif normmethod == 'minMax':\n            nm = preprocessing.MinMaxScaler()\n            X = nm.fit_transform(X)\n        print(f'NormarlizationMethod:{normmethod}')\n        \n        km = KMeans(n_clusters=clusters, random_state=2022)\n        km.fit(X)\n        print('Distortion: %.2f'% km.inertia_)\n\n        pred = km.labels_\n        df_pred = pd.DataFrame(pred, columns=['pred'])\n        df_pred = pd.concat([df, df_pred], axis=1)\n        \n        df_norm = pd.DataFrame(X, columns=usecol)\n        print(df_norm.describe())\n\n\n        if DEBUG:\n            df_norm = pd.concat([df[predcol], df_norm], axis=1)\n            return df_pred, df_norm\n        else:\n            return df_pred\n","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:07:42.840854Z","iopub.execute_input":"2022-04-04T13:07:42.841227Z","iopub.status.idle":"2022-04-04T13:07:42.860098Z","shell.execute_reply.started":"2022-04-04T13:07:42.841169Z","shell.execute_reply":"2022-04-04T13:07:42.858992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEBUG = False\nPATH_INPUT = r'../input/h-and-m-personalized-fashion-recommendations/'","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:07:42.864031Z","iopub.execute_input":"2022-04-04T13:07:42.864343Z","iopub.status.idle":"2022-04-04T13:07:42.878303Z","shell.execute_reply.started":"2022-04-04T13:07:42.864309Z","shell.execute_reply":"2022-04-04T13:07:42.877004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers = pd.read_csv(PATH_INPUT + 'customers.csv')\n\nclst = Clustering_HandM()\ncustomers = clst.customers_preprocessing(customers)\n\nusecol = ['club_member_status', 'fashion_news_frequency', 'age', 'FN', 'Active']\npredcol = ['customer_id']\n\ndfCustomers = clst.clustering(customers, predcol=predcol, usecol=usecol, )","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:07:42.881996Z","iopub.execute_input":"2022-04-04T13:07:42.883084Z","iopub.status.idle":"2022-04-04T13:08:12.512615Z","shell.execute_reply.started":"2022-04-04T13:07:42.883032Z","shell.execute_reply":"2022-04-04T13:08:12.510408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.Comparison of Age_bins and ClusteringPred","metadata":{}},{"cell_type":"code","source":"listBin = [-1, 19, 29, 39, 49, 59, 69, 119]\ndfCustomers['age_bins'] = pd.cut(dfCustomers['age'], listBin)","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:08:12.514643Z","iopub.execute_input":"2022-04-04T13:08:12.515031Z","iopub.status.idle":"2022-04-04T13:08:12.564567Z","shell.execute_reply.started":"2022-04-04T13:08:12.514982Z","shell.execute_reply":"2022-04-04T13:08:12.563525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.crosstab(dfCustomers['pred'], dfCustomers['age_bins'])","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:08:12.56645Z","iopub.execute_input":"2022-04-04T13:08:12.566803Z","iopub.status.idle":"2022-04-04T13:08:12.747741Z","shell.execute_reply.started":"2022-04-04T13:08:12.566747Z","shell.execute_reply":"2022-04-04T13:08:12.746727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfCustomers = dfCustomers.drop(['age_bins'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:08:12.749603Z","iopub.execute_input":"2022-04-04T13:08:12.750205Z","iopub.status.idle":"2022-04-04T13:08:12.841778Z","shell.execute_reply.started":"2022-04-04T13:08:12.750157Z","shell.execute_reply":"2022-04-04T13:08:12.840736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4.EDA of recent popular articles of each Pred\n[Reference](https://www.kaggle.com/code/hechtjp/h-m-eda-rule-base-by-customer-age)","metadata":{}},{"cell_type":"code","source":"dfTransactions = cudf.read_csv(PATH_INPUT + 'transactions_train.csv',  \n                               usecols=['t_dat', 'customer_id', 'article_id'],\n                               dtype={'article_id': 'int32', 't_dat': 'string', 'customer_id': 'string'})\ndfTransactions['t_dat'] = cudf.to_datetime(dfTransactions['t_dat'])\ndfTransactions.set_index('t_dat', inplace=True)\ndfTransactions.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:08:12.843631Z","iopub.execute_input":"2022-04-04T13:08:12.844015Z","iopub.status.idle":"2022-04-04T13:08:15.878089Z","shell.execute_reply.started":"2022-04-04T13:08:12.843969Z","shell.execute_reply":"2022-04-04T13:08:15.87695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfRecent = dfTransactions.loc['2020-09-01' : '2020-09-21']\ndfRecent.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:08:15.880143Z","iopub.execute_input":"2022-04-04T13:08:15.880523Z","iopub.status.idle":"2022-04-04T13:08:16.551529Z","shell.execute_reply.started":"2022-04-04T13:08:15.880474Z","shell.execute_reply":"2022-04-04T13:08:16.550486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfRecent = dfRecent.to_pandas()\n# dfRecent = dfRecent.merge(dfCustomers[['customer_id', 'age_bins']], on='customer_id', how='inner')\ndfRecent = dfRecent.merge(dfCustomers[['customer_id', 'pred']], on='customer_id', how='inner')\ndfRecent.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:08:16.553107Z","iopub.execute_input":"2022-04-04T13:08:16.553971Z","iopub.status.idle":"2022-04-04T13:08:18.064095Z","shell.execute_reply.started":"2022-04-04T13:08:16.553897Z","shell.execute_reply":"2022-04-04T13:08:18.062994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfRecent = dfRecent.groupby(['pred', 'article_id']).count().reset_index().rename(columns={'customer_id': 'counts'})\nlistUniBins = dfRecent['pred'].unique().tolist()\n\ndict100 = {}\nfor uniBin in listUniBins:\n    # dfTemp = dfRecent[dfRecent['age_bins'] == uniBin]\n    dfTemp = dfRecent[dfRecent['pred'] == uniBin]\n    dfTemp = dfTemp.sort_values(by='counts', ascending=False)\n    dict100[uniBin] = dfTemp.head(100)['article_id'].values.tolist()\n\ndf100 = pd.DataFrame([dict100]).T.rename(columns={0:'top100'})","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:08:18.069838Z","iopub.execute_input":"2022-04-04T13:08:18.072619Z","iopub.status.idle":"2022-04-04T13:08:18.389369Z","shell.execute_reply.started":"2022-04-04T13:08:18.072564Z","shell.execute_reply":"2022-04-04T13:08:18.388333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index in df100.index:\n    df100[index] = [len(set(df100.at[index, 'top100']) & set(df100.at[x, 'top100']))/100 for x in df100.index]\n\ndf100 = df100.drop(columns='top100')\nplt.figure(figsize=(10, 6))\nsns.heatmap(df100, annot=True, cbar=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:08:18.395294Z","iopub.execute_input":"2022-04-04T13:08:18.398108Z","iopub.status.idle":"2022-04-04T13:08:19.551441Z","shell.execute_reply.started":"2022-04-04T13:08:18.398056Z","shell.execute_reply":"2022-04-04T13:08:19.550271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5.Prediction\n[H&M EDA & Rule Base by Customer Age](https://www.kaggle.com/code/hechtjp/h-m-eda-rule-base-by-customer-age)\n[H&M: Faster Trending Products Weekly](https://www.kaggle.com/code/hervind/h-m-faster-trending-products-weekly/notebook)","metadata":{}},{"cell_type":"code","source":"N = 12\nlistUniBins = dfCustomers['pred'].unique().tolist()","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:08:19.555306Z","iopub.execute_input":"2022-04-04T13:08:19.555611Z","iopub.status.idle":"2022-04-04T13:08:19.572478Z","shell.execute_reply.started":"2022-04-04T13:08:19.555576Z","shell.execute_reply":"2022-04-04T13:08:19.571022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for uniBin in listUniBins:\n    df  = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv',\n                            usecols= ['t_dat', 'customer_id', 'article_id'], \n                            dtype={'article_id': 'int32', 't_dat': 'string', 'customer_id': 'string'})\n    if str(uniBin) == 'nan':\n        dfCustomersTemp = dfCustomers[dfCustomers['pred'].isnull()]\n    else:\n        dfCustomersTemp = dfCustomers[dfCustomers['pred'] == uniBin]\n    \n    dfCustomersTemp = dfCustomersTemp.drop(['pred'], axis=1)\n    dfCustomersTemp = cudf.from_pandas(dfCustomersTemp)\n    \n    df = df.merge(dfCustomersTemp[['customer_id', 'age']], on='customer_id', how='inner')\n    print(f'The shape of scope transaction for {uniBin} is {df.shape}. \\n')\n          \n    df ['customer_id'] = df ['customer_id'].str[-16:].str.hex_to_int().astype('int64')\n    df['t_dat'] = cudf.to_datetime(df['t_dat'])\n    last_ts = df['t_dat'].max()\n\n    tmp = df[['t_dat']].copy().to_pandas()\n    tmp['dow'] = tmp['t_dat'].dt.dayofweek\n    tmp['ldbw'] = tmp['t_dat'] - pd.TimedeltaIndex(tmp['dow'] - 1, unit='D')\n    tmp.loc[tmp['dow'] >=2 , 'ldbw'] = tmp.loc[tmp['dow'] >=2 , 'ldbw'] + pd.TimedeltaIndex(np.ones(len(tmp.loc[tmp['dow'] >=2])) * 7, unit='D')\n\n    df['ldbw'] = tmp['ldbw'].values\n    \n    weekly_sales = df.drop('customer_id', axis=1).groupby(['ldbw', 'article_id']).count().reset_index()\n    weekly_sales = weekly_sales.rename(columns={'t_dat': 'count'})\n    \n    df = df.merge(weekly_sales, on=['ldbw', 'article_id'], how = 'left')\n    \n    weekly_sales = weekly_sales.reset_index().set_index('article_id')\n\n    df = df.merge(\n        weekly_sales.loc[weekly_sales['ldbw']==last_ts, ['count']],\n        on='article_id', suffixes=(\"\", \"_targ\"))\n\n    df['count_targ'].fillna(0, inplace=True)\n    del weekly_sales\n    \n    df['quotient'] = df['count_targ'] / df['count']\n    \n    target_sales = df.drop('customer_id', axis=1).groupby('article_id')['quotient'].sum()\n    general_pred = target_sales.nlargest(N).index.to_pandas().tolist()\n    general_pred = ['0' + str(article_id) for article_id in general_pred]\n    general_pred_str =  ' '.join(general_pred)\n    del target_sales\n    \n    purchase_dict = {}\n\n    tmp = df.copy().to_pandas()\n    tmp['x'] = ((last_ts - tmp['t_dat']) / np.timedelta64(1, 'D')).astype(int)\n    tmp['dummy_1'] = 1 \n    tmp['x'] = tmp[[\"x\", \"dummy_1\"]].max(axis=1)\n\n    a, b, c, d = 2.5e4, 1.5e5, 2e-1, 1e3\n    tmp['y'] = a / np.sqrt(tmp['x']) + b * np.exp(-c*tmp['x']) - d\n\n    tmp['dummy_0'] = 0 \n    tmp['y'] = tmp[[\"y\", \"dummy_0\"]].max(axis=1)\n    tmp['value'] = tmp['quotient'] * tmp['y'] \n\n    tmp = tmp.groupby(['customer_id', 'article_id']).agg({'value': 'sum'})\n    tmp = tmp.reset_index()\n\n    tmp = tmp.loc[tmp['value'] > 0]\n    tmp['rank'] = tmp.groupby(\"customer_id\")[\"value\"].rank(\"dense\", ascending=False)\n    tmp = tmp.loc[tmp['rank'] <= 12]\n\n    purchase_df = tmp.sort_values(['customer_id', 'value'], ascending = False).reset_index(drop = True)\n    purchase_df['prediction'] = '0' + purchase_df['article_id'].astype(str) + ' '\n    purchase_df = purchase_df.groupby('customer_id').agg({'prediction': sum}).reset_index()\n    purchase_df['prediction'] = purchase_df['prediction'].str.strip()\n    purchase_df = cudf.DataFrame(purchase_df)\n    \n    sub  = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv',\n                            usecols= ['customer_id'], \n                            dtype={'customer_id': 'string'})\n    \n    numCustomers = sub.shape[0]\n    \n    sub = sub.merge(dfCustomersTemp[['customer_id', 'age']], on='customer_id', how='inner')\n\n    sub['customer_id2'] = sub['customer_id'].str[-16:].str.hex_to_int().astype('int64')\n\n    sub = sub.merge(purchase_df, left_on = 'customer_id2', right_on = 'customer_id', how = 'left',\n                   suffixes = ('', '_ignored'))\n\n    sub = sub.to_pandas()\n    sub['prediction'] = sub['prediction'].fillna(general_pred_str)\n    sub['prediction'] = sub['prediction'] + ' ' +  general_pred_str\n    sub['prediction'] = sub['prediction'].str.strip()\n    sub['prediction'] = sub['prediction'].str[:131]\n    sub = sub[['customer_id', 'prediction']]\n    sub.to_csv(f'submission_' + str(uniBin) + '.csv',index=False)\n    print(f'Saved prediction for {uniBin}. The shape is {sub.shape}. \\n')\n    print('-'*50)\nprint('Finished.\\n')\nprint('='*50)","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:08:19.574603Z","iopub.execute_input":"2022-04-04T13:08:19.575138Z","iopub.status.idle":"2022-04-04T13:09:35.970129Z","shell.execute_reply.started":"2022-04-04T13:08:19.575087Z","shell.execute_reply":"2022-04-04T13:09:35.968649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6.Submission","metadata":{}},{"cell_type":"code","source":"for i, uniBin in enumerate(listUniBins):\n    dfTemp  = cudf.read_csv(f'submission_' + str(uniBin) + '.csv')\n    if i == 0:\n        dfSub = dfTemp\n    else:\n        dfSub = cudf.concat([dfSub, dfTemp], axis=0)\n\nassert dfSub.shape[0] == numCustomers, f'The number of dfSub rows is not correct. {dfSub.shape[0]} vs {numCustomers}.'\n\ndfSub.to_csv(f'submission.csv', index=False)\nprint(f'Saved submission.csv.')","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:09:35.972088Z","iopub.execute_input":"2022-04-04T13:09:35.972386Z","iopub.status.idle":"2022-04-04T13:09:37.260261Z","shell.execute_reply.started":"2022-04-04T13:09:35.972344Z","shell.execute_reply":"2022-04-04T13:09:37.259128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfCheck = cudf.read_csv('./submission.csv')\ndfCheck.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:09:37.261785Z","iopub.execute_input":"2022-04-04T13:09:37.262666Z","iopub.status.idle":"2022-04-04T13:09:37.473057Z","shell.execute_reply.started":"2022-04-04T13:09:37.262616Z","shell.execute_reply":"2022-04-04T13:09:37.471971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}