{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":".\n\n\n#  H&M-personalized-fashion-recommendations - Competition\n\n## <font color=green>Implement a simple Customer Segmentation Analysis for H&M Fashion</font>\n\n***\n\n> **This is a WORK-IN-PROGRESS -   9/19/2022 - please check back for the final version...**\n\n> Calculating the total sales for each Customer can take a few minutes - Please be patient...\n\n\nThis analysis is bit off-topic from the competition goal, but understanding the underlying Market is critical.  \n\nLet's use the SKLearn Kmeans Model ( Unsupervised Learning ) to see what the computer thinks our Customer Segments should be...\n\nHope this simple example helps!\n\nAll the best,\nMike Pastor\n\n\nCompetition Home Page:  \n[H&M Personalized Fashion Recommendations\nProvide product recommendations based on previous purchases](hhttps://www.kaggle.com/competitions/h-and-m-personalized-fashion-recommendations)\n","metadata":{}},{"cell_type":"markdown","source":".\n\n#  Let's start by loading our libraries and setting up the environment...\n","metadata":{}},{"cell_type":"code","source":"#################################################################\n#       Clustering_main.py\n#\n#       Implement a Market Segmentation Analysis for H&M\n#\n#           Mike Pastor  9/17/2022\n#\n# Kaggle Project Link\n# https://www.kaggle.com/competitions/h-and-m-personalized-fashion-recommendations/overview\n\n\n#  Our critical runtime parameters\n#\nHM_DATASET_PATH='../input/h-and-m-personalized-fashion-recommendations'\n%matplotlib inline\n\n# importing the required libraries\n#\nimport pandas as pd\nimport numpy as np\nfrom datetime import datetime\nimport matplotlib.pyplot as plt\n\nimport sklearn\nfrom sklearn.cluster import KMeans\nfrom sklearn.decomposition import PCA\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\n\nimport seaborn as sns\nsns.set() ## this is for sns graph styling\nnp.set_printoptions(precision=2)\n\n#  Track the overall time for train & submission preparation\nglobal_now = datetime.now()\nglobal_current_time = global_now.strftime(\"%H:%M:%S\")\nprint(\"##############  Starting up...  - Current Time =\", global_current_time)\nprint('sklearn Version: %s' % sklearn.__version__)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-09-19T23:35:08.152250Z","iopub.execute_input":"2022-09-19T23:35:08.152618Z","iopub.status.idle":"2022-09-19T23:35:09.624729Z","shell.execute_reply.started":"2022-09-19T23:35:08.152547Z","shell.execute_reply":"2022-09-19T23:35:09.623131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":",\n\n\n#  Now let's read in the  H&M Customer dataset...\n","metadata":{}},{"cell_type":"code","source":"\n##############################################################################\n##     importing the data\n#\ncustomers_data = pd.read_csv(HM_DATASET_PATH+'/customers.csv')\nprint(\"CUSTOMERS dataset == \", customers_data.shape)\n\n# Also import the Transaction data  for TOTAL SALES per customer Feature \ntransactions_data = pd.read_csv(HM_DATASET_PATH+'/transactions_train.csv')\nprint(\"TRANSACTIONS dataset == \", transactions_data.shape)\n\n#  Scale the data back to near dollars per the Kaggle discussions\ntransactions_data['price'] = transactions_data['price'] * 590\n\n# Computer the total sales per Customer - GrossSales \n#\nsalesDataset = transactions_data.groupby('customer_id')['price'].sum().reset_index()\nsalesDataset = pd.DataFrame( salesDataset )\nsalesDataset = salesDataset[ ['customer_id', 'price'] ]\nsalesDataset =  salesDataset.rename( columns={'price': 'GrossSales'})\n\ncustomers_data = pd.merge(customers_data, salesDataset, how=\"outer\", on=[\"customer_id\"])\nprint(\"  FINAL  customers_data  Dataset == \", customers_data.shape )\nprint( \"FINAL  customers_data  Dataset columns ==\", customers_data.columns )\nprint(\"#######################  Cust Total gross Sales = \", customers_data['GrossSales'].sum() )\n\n","metadata":{"execution":{"iopub.status.busy":"2022-09-19T23:35:09.626867Z","iopub.execute_input":"2022-09-19T23:35:09.627245Z","iopub.status.idle":"2022-09-19T23:36:42.074786Z","shell.execute_reply.started":"2022-09-19T23:35:09.627212Z","shell.execute_reply":"2022-09-19T23:36:42.073926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":".\n\n\n#  Now let's prepare the data\n","metadata":{}},{"cell_type":"code","source":"\n#  Remove the Customer ID and Postal code\n#\nX = customers_data[ [ 'FN', 'Active', 'age', 'fashion_news_frequency', 'club_member_status', 'GrossSales' ] ].copy()\n#  \n\n\n# # Save this for later\nXcolumnlist = X.columns\n\n# Fix the Categorical variables by creating a Data Dictionary\n#     We could also apply one-hot-encoding here, but this maintains the understandability..\n#\n#\nprint( \"X == \", X.columns )\n\ncleanup_nums = {\"fashion_news_frequency\":     {\"Monthly\": 4, \"Regularly\": 2, \"NONE\": 0, \"None\": 0, \"\": -1},\n                \"club_member_status\": {\"ACTIVE\": 4, \"PRE-CREATE\": 2, \"LEFT CLUB\": 0, \"\": -1 }}\n\nX = X.replace(cleanup_nums)\n\n\n############################################################################\n#\n#     Get rid of the Numerical nulls - replace with ZERO for this analysis\n#\nimputer = SimpleImputer(missing_values=np.nan, strategy='constant', fill_value=0)\nimputer = imputer.fit(X)\nX = imputer.transform(X)\n\n\n# Standardize the data  - TBD\n#\n# scaler = StandardScaler()\n# X = scaler.fit_transform(X)\n\nprint(\"Customer data has been Prepared...\")","metadata":{"execution":{"iopub.status.busy":"2022-09-19T23:36:42.075913Z","iopub.execute_input":"2022-09-19T23:36:42.076679Z","iopub.status.idle":"2022-09-19T23:36:43.299928Z","shell.execute_reply.started":"2022-09-19T23:36:42.076632Z","shell.execute_reply":"2022-09-19T23:36:43.298699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":".\n\n\n#  Run the SKLearn  KMeans model... \n","metadata":{}},{"cell_type":"code","source":"\n#  Run the SKLearn  KMeans model...  ###########################################################\n#\n#    k-means++ gives us better cluster starting points (i.e. based on probability - not random)\n#\n#    Start with 4 clusters...\n\nmodel = KMeans(n_clusters = 4, init = 'k-means++', random_state = 42)\n\nmodel.fit(X)\n\nprint(\"Cluster Centers ######################\")\nprint(model.cluster_centers_)\n","metadata":{"execution":{"iopub.status.busy":"2022-09-19T23:36:43.302200Z","iopub.execute_input":"2022-09-19T23:36:43.302996Z","iopub.status.idle":"2022-09-19T23:36:59.048007Z","shell.execute_reply.started":"2022-09-19T23:36:43.302959Z","shell.execute_reply":"2022-09-19T23:36:59.047044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":".\n\n#  What is the best number of Clusters for our Kmeans Model?\n\n","metadata":{}},{"cell_type":"code","source":"#  How many clusters are best?\n\n#      Within Clusters Sum of Squares(WCSS) and Elbow Method\n\nprint(\"wcss STARTING...\")\nwcss = []\nfor i in range(1,6):\n    kmeans_pca = KMeans(n_clusters = i, init = 'k-means++', random_state = 42)\n    kmeans_pca.fit(X)\n    wcss.append(kmeans_pca.inertia_)\n    print(\"wcss Scenario ADDED...\")\n\nplt.figure(figsize = (10,8))\nplt.plot(range(1, 6), wcss, marker = 'o', linestyle = '-.',color='purple')\n\nplt.xlabel('Number of Clusters')\nplt.ylabel('WCSS')\nplt.title('Mike''s K-means Clustering')\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-09-19T23:36:59.049182Z","iopub.execute_input":"2022-09-19T23:36:59.050377Z","iopub.status.idle":"2022-09-19T23:37:50.801403Z","shell.execute_reply.started":"2022-09-19T23:36:59.050349Z","shell.execute_reply":"2022-09-19T23:37:50.800053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  The 'Elbow' is at 2 Clusters - This is best but we will use 4 Clusters here\n\n##   Let's analyze the Segment assignments for each Customer...\n","metadata":{}},{"cell_type":"code","source":"\n#  # We create a new data frame with the original features\n#      and add a new column with the assigned clusters for each point.\n#\nnewX = pd.DataFrame( X ).copy()\nnewX.columns = Xcolumnlist   # Set above\nnewX['Assigned Cluster'] = model.labels_\n","metadata":{"execution":{"iopub.status.busy":"2022-09-19T23:37:50.803218Z","iopub.execute_input":"2022-09-19T23:37:50.803544Z","iopub.status.idle":"2022-09-19T23:37:50.816640Z","shell.execute_reply.started":"2022-09-19T23:37:50.803508Z","shell.execute_reply":"2022-09-19T23:37:50.815035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":".\n#  Let's start the analysis...\n\n## Assign names to the Segments - \n##   Since the Kmeans algorithm appears to have split the dataset on Gross Sales lines, let's use those groups for the Segment names:\n\n \n> Copper- $145\n\n> Gold- $6276\n\n> Silver- $2515\n\n> Bronze- $918\n\n","metadata":{}},{"cell_type":"code","source":"\nnew_SUMMARY = newX.groupby(['Assigned Cluster']).mean()\nprint(\"####    SUMMARY - Average values for each feature ######## \\n\")\n#print( new_SUMMARY )\n#\nnew_SUMMARY = new_SUMMARY.rename({0:'Copper- $145',\n                         1:'Gold- $6276',\n                         2:'Silver- $2515',\n                         3:'Bronze- $918' })\n\n#\nprint( new_SUMMARY.head() )","metadata":{"execution":{"iopub.status.busy":"2022-09-19T23:37:50.819008Z","iopub.execute_input":"2022-09-19T23:37:50.819477Z","iopub.status.idle":"2022-09-19T23:37:50.874178Z","shell.execute_reply.started":"2022-09-19T23:37:50.819443Z","shell.execute_reply":"2022-09-19T23:37:50.872883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":".\n\n\n#  Let's view the Segments by size..","metadata":{}},{"cell_type":"code","source":"\ncounts = np.zeros(4)\ncounts[0] = newX[newX['Assigned Cluster'] == 0  ].count()[0]\ncounts[1] = newX[newX['Assigned Cluster'] == 1 ].count()[0]\ncounts[2] = newX[newX['Assigned Cluster'] == 2  ].count()[0]\ncounts[3] = newX[newX['Assigned Cluster'] == 3  ].count()[0]\nprint( counts )\n\n# mylabels = [\"Young 21+\", \"Senior 62+\", \"Older 48+\", \"Middle 31+\"]\nmylabels = [\"Copper-$145\", \"Gold-$6276\", \"Silver-$2515\", \"Bronze-$918\"]\n\nplt.pie( counts, labels=mylabels  )\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-09-19T23:37:50.876041Z","iopub.execute_input":"2022-09-19T23:37:50.876378Z","iopub.status.idle":"2022-09-19T23:37:51.047557Z","shell.execute_reply.started":"2022-09-19T23:37:50.876353Z","shell.execute_reply":"2022-09-19T23:37:51.046837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  The Customer Base is dominated by the 'Copper- $145' Segment with about 1.1 million Customers\n\n### This fact could be determined by 'eye-balling' the data, but the Kmeans Model can independently confirm these matters.\n\n'\n\n##  Let's also view the Segments by Total Gross Sales for each assigned group...","metadata":{}},{"cell_type":"code","source":"salesArray = np.zeros(4)\nsalesArray[0] = newX[newX['Assigned Cluster'] == 0  ]['GrossSales'].sum()\nsalesArray[1] = newX[newX['Assigned Cluster'] == 1  ]['GrossSales'].sum()\nsalesArray[2] = newX[newX['Assigned Cluster'] == 2  ]['GrossSales'].sum()\nsalesArray[3] = newX[newX['Assigned Cluster'] == 3  ]['GrossSales'].sum()\n\n\n# mylabels = [\"Young 21+\", \"Senior 62+\", \"Older 48+\", \"Middle 31+\"]\nmylabels = [\"Copper-$145\", \"Gold-$6276\", \"Silver-$2515\", \"Bronze-$918\"]\n\nprint( mylabels[0], \" == \", \"${:0,.0f}\".format(salesArray[0] ) )\nprint( mylabels[1], \" == \", \"${:0,.0f}\".format(salesArray[1] ) )\nprint( mylabels[2], \" == \", \"${:0,.0f}\".format(salesArray[2] ) )\nprint( mylabels[3], \" == \", \"${:0,.0f}\".format(salesArray[3] ) )\n\nplt.pie( salesArray, labels=mylabels  )\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-19T23:37:51.048604Z","iopub.execute_input":"2022-09-19T23:37:51.049114Z","iopub.status.idle":"2022-09-19T23:37:51.181811Z","shell.execute_reply.started":"2022-09-19T23:37:51.049084Z","shell.execute_reply":"2022-09-19T23:37:51.180752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n##  The 'Bronze' Segment leads with the highest Total Gross Sales (about $197m)\n'\n##  What else could we learn from the Segments that Kmeans assigned?\n","metadata":{}},{"cell_type":"code","source":"print( new_SUMMARY )\n","metadata":{"execution":{"iopub.status.busy":"2022-09-19T23:37:51.185347Z","iopub.execute_input":"2022-09-19T23:37:51.185741Z","iopub.status.idle":"2022-09-19T23:37:51.193361Z","shell.execute_reply.started":"2022-09-19T23:37:51.185703Z","shell.execute_reply":"2022-09-19T23:37:51.192457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  The 'Gold- $6276' Segment produces the most 'Activity' and is most likely to read the 'Fashion News'\n###  However, they are also the smallest group with about 6800 members.\n\n\n## The 'Copper- $145' Segment with the most Customers (1.1 million) is also the least likely to read the 'Fashion News'\n> ##    Adjust the marketing strategy?\n\n","metadata":{}},{"cell_type":"markdown","source":".\n\n# Let's view a scatter plot of the Age versus total Gross Sales per Customer\n##  with the Assigned Segment LABEL points in color...\n","metadata":{}},{"cell_type":"code","source":"from matplotlib.lines import Line2D\n\n#  Scatter plot\n\nx_axis = newX['age']\ny_axis = newX['GrossSales']\nplt.xlabel = 'Age'\nplt.ylabel = 'GrossSales'\nplt.figure(figsize = (8,6))\n\nsns.scatterplot(x=x_axis, y=y_axis,\\\n                hue = newX['Assigned Cluster'], palette = ['g', 'r', 'c', 'm'], legend=False)\n\ncustom = [Line2D([], [], marker='.', color='g', linestyle='None'),\n          Line2D([], [], marker='.', color='r', linestyle='None'),\n          Line2D([], [], marker='.', color='c', linestyle='None'),\n          Line2D([], [], marker='.', color='m', linestyle='None')]\nplt.legend(custom, ['Copper', 'Gold', 'Silver', 'Bronze' ], loc='upper right')\n\nplt.title('H&M Fashion - Customer Segmentation via K-means')\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-09-20T00:27:50.379328Z","iopub.execute_input":"2022-09-20T00:27:50.379715Z","iopub.status.idle":"2022-09-20T00:28:17.227649Z","shell.execute_reply.started":"2022-09-20T00:27:50.379688Z","shell.execute_reply":"2022-09-20T00:28:17.226005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  The Segments assigned by Kmeans are clearly shown on this graph","metadata":{}},{"cell_type":"code","source":"\n# Mission Complete!\n\n##################################################################################\nglobal_later = datetime.now()\nglobal_later_time = global_later.strftime(\"%H:%M:%S\")\n# print(\"DONE - train_data.csv  - Current Time =\", global_later_time)\nprint(\"Customer Segmentation Example - Total EXECUTION Time =\", (global_later - global_now) )\nprint(\"Done...\")\n","metadata":{"execution":{"iopub.status.busy":"2022-09-19T23:38:17.748585Z","iopub.execute_input":"2022-09-19T23:38:17.748996Z","iopub.status.idle":"2022-09-19T23:38:17.755464Z","shell.execute_reply.started":"2022-09-19T23:38:17.748971Z","shell.execute_reply":"2022-09-19T23:38:17.754563Z"},"trusted":true},"execution_count":null,"outputs":[]}]}