{"metadata":{"environment":{"name":"rapids-gpu.0-18.m65","type":"gcloud","uri":"gcr.io/deeplearning-platform-release/rapids-gpu.0-18:m65"},"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"},{"sourceType":"kernelVersion","sourceId":151983018},{"sourceType":"kernelVersion","sourceId":151982430}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Data Analysis","metadata":{"tags":[]}},{"cell_type":"markdown","source":"In this notebook, I have tried to explore the most common questions that can provide value to the model developement process.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:43:55.144129Z","iopub.execute_input":"2023-11-23T12:43:55.145561Z","iopub.status.idle":"2023-11-23T12:43:55.706696Z","shell.execute_reply.started":"2023-11-23T12:43:55.145507Z","shell.execute_reply":"2023-11-23T12:43:55.704942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install kaleido --q","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:43:55.710844Z","iopub.execute_input":"2023-11-23T12:43:55.711787Z","iopub.status.idle":"2023-11-23T12:44:11.460553Z","shell.execute_reply.started":"2023-11-23T12:43:55.711729Z","shell.execute_reply":"2023-11-23T12:44:11.458106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from eda_functions_python_class import EdaBusinessCase as eda","metadata":{"execution":{"iopub.status.busy":"2023-11-23T13:02:30.426032Z","iopub.execute_input":"2023-11-23T13:02:30.426653Z","iopub.status.idle":"2023-11-23T13:02:33.11693Z","shell.execute_reply.started":"2023-11-23T13:02:30.426588Z","shell.execute_reply":"2023-11-23T13:02:33.115294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nfrom time import time\nimport plotly.express as px","metadata":{"execution":{"iopub.status.busy":"2023-11-23T13:02:36.364928Z","iopub.execute_input":"2023-11-23T13:02:36.365744Z","iopub.status.idle":"2023-11-23T13:02:36.956859Z","shell.execute_reply.started":"2023-11-23T13:02:36.365695Z","shell.execute_reply":"2023-11-23T13:02:36.955506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.io as pio\n# pio.renderers.default = \"svg\"","metadata":{"execution":{"iopub.status.busy":"2023-11-23T13:02:39.446965Z","iopub.execute_input":"2023-11-23T13:02:39.448271Z","iopub.status.idle":"2023-11-23T13:02:39.454927Z","shell.execute_reply.started":"2023-11-23T13:02:39.448218Z","shell.execute_reply":"2023-11-23T13:02:39.453262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_base = '/kaggle/input/h-and-m-personalized-fashion-recommendations/'","metadata":{"execution":{"iopub.status.busy":"2023-11-23T13:02:43.504649Z","iopub.execute_input":"2023-11-23T13:02:43.50517Z","iopub.status.idle":"2023-11-23T13:02:43.512044Z","shell.execute_reply.started":"2023-11-23T13:02:43.505126Z","shell.execute_reply":"2023-11-23T13:02:43.510614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reference = eda(path_base)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T13:02:45.508881Z","iopub.execute_input":"2023-11-23T13:02:45.509411Z","iopub.status.idle":"2023-11-23T13:06:33.55261Z","shell.execute_reply.started":"2023-11-23T13:02:45.509373Z","shell.execute_reply":"2023-11-23T13:06:33.550897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Daily aggregates","metadata":{"tags":[]}},{"cell_type":"code","source":"daily_aggregate = reference.daily_aggregate()","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:48:05.339263Z","iopub.execute_input":"2023-11-23T12:48:05.33988Z","iopub.status.idle":"2023-11-23T12:48:42.553487Z","shell.execute_reply.started":"2023-11-23T12:48:05.339829Z","shell.execute_reply":"2023-11-23T12:48:42.551568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.line(data_frame=daily_aggregate, \n       x='t_dat_', \n       y=['no_unique_custs', 'n_transactions', 'n_unique_articles'],\n       facet_row='sales_channel_id_',\n        labels={'t_dat_': 'date'},\n        title='Daily aggregate of customers, articles and transactions')\n\nfig.show(width = 1200, height = 800)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:48:42.555844Z","iopub.execute_input":"2023-11-23T12:48:42.556509Z","iopub.status.idle":"2023-11-23T12:48:43.510273Z","shell.execute_reply.started":"2023-11-23T12:48:42.556468Z","shell.execute_reply":"2023-11-23T12:48:43.508002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.line(data_frame=daily_aggregate, \n              x='t_dat_', \n              y=['tx_per_cust', 'u_articles_per_cust'],\n              line_dash='sales_channel_id_',\n              labels={'t_dat_': 'date'},\n              title='Daily aggregate of transaction and unique articles bought per customer')\n\nfig.show(width = 1200, height = 800)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:48:43.512448Z","iopub.execute_input":"2023-11-23T12:48:43.512978Z","iopub.status.idle":"2023-11-23T12:48:43.688694Z","shell.execute_reply.started":"2023-11-23T12:48:43.512936Z","shell.execute_reply":"2023-11-23T12:48:43.686925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observation\n- Channel 1 is store, 2 is online.\n- No sales from march 20 to may 6 in stores.\n- On average more unique articles are bought in stores then online.\n- Online has more transactions per customer (how does it affect modelling?)\n- During covid restrictions number of transactions per customer were increased (Probably due to closed stores) but number of unique item purchases went down.","metadata":{}},{"cell_type":"code","source":"## book keeping\ndel(daily_aggregate)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:48:43.69142Z","iopub.execute_input":"2023-11-23T12:48:43.691953Z","iopub.status.idle":"2023-11-23T12:48:43.920986Z","shell.execute_reply.started":"2023-11-23T12:48:43.691903Z","shell.execute_reply":"2023-11-23T12:48:43.919212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"article_subset = reference.article_subset_info()\narticle_subset.head(-1)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:48:43.923388Z","iopub.execute_input":"2023-11-23T12:48:43.92388Z","iopub.status.idle":"2023-11-23T12:48:44.885146Z","shell.execute_reply.started":"2023-11-23T12:48:43.923841Z","shell.execute_reply":"2023-11-23T12:48:44.883621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Too many options here, therefore, gonna stick to `index_group_name` (this is better), `index_name`, `product_group_name`, `perceived_colour_value_name` only.","metadata":{}},{"cell_type":"markdown","source":"### Distribution of article sale over days","metadata":{"tags":[]}},{"cell_type":"code","source":"df_season = reference.season_analysis()","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:48:44.887461Z","iopub.execute_input":"2023-11-23T12:48:44.887914Z","iopub.status.idle":"2023-11-23T12:51:59.548562Z","shell.execute_reply.started":"2023-11-23T12:48:44.887878Z","shell.execute_reply":"2023-11-23T12:51:59.546909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(df_season, x='t_dat', facet_col='tx_year',\n            title='Distribution of number of days an article is purchased', labels={'t_dat': 'Date'})\nfig.show(width = 1200, height = 800)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:51:59.555355Z","iopub.execute_input":"2023-11-23T12:51:59.556239Z","iopub.status.idle":"2023-11-23T12:51:59.760858Z","shell.execute_reply.started":"2023-11-23T12:51:59.556166Z","shell.execute_reply":"2023-11-23T12:51:59.759366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_season.groupby(['tx_year'])['t_dat'].agg(['mean', 'median'])","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:51:59.762676Z","iopub.execute_input":"2023-11-23T12:51:59.763116Z","iopub.status.idle":"2023-11-23T12:51:59.796833Z","shell.execute_reply.started":"2023-11-23T12:51:59.763077Z","shell.execute_reply":"2023-11-23T12:51:59.795712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del(df_season)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:51:59.799034Z","iopub.execute_input":"2023-11-23T12:51:59.799505Z","iopub.status.idle":"2023-11-23T12:51:59.80636Z","shell.execute_reply.started":"2023-11-23T12:51:59.799463Z","shell.execute_reply":"2023-11-23T12:51:59.804692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observation\n- The idea behind this graph is to understand if there are articles that are availbale for a shorter period of time as seasonality plays an important part in garment retails.\n- Most articles are purchased for shorter period of time. (maybe we can filter predictions to recommend recently purchased articles such as in last 36 days)","metadata":{}},{"cell_type":"markdown","source":"## Unique available products per year (based on product name)","metadata":{"tags":[]}},{"cell_type":"code","source":"df_prod_per_year = reference.products_per_year()","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:51:59.808445Z","iopub.execute_input":"2023-11-23T12:51:59.808942Z","iopub.status.idle":"2023-11-23T12:52:11.674776Z","shell.execute_reply.started":"2023-11-23T12:51:59.808899Z","shell.execute_reply":"2023-11-23T12:52:11.672566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(df_prod_per_year.astype({'tx_year': str}), \n       x='tx_year', \n       y='n_unique_prods', \n       title='Distribution of unique products over years',\n      labels={'n_unique_prods': 'number of unique products per year', 'tx_year': 'year'})\n\nfig.show(width = 1200, height = 800)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:52:11.677027Z","iopub.execute_input":"2023-11-23T12:52:11.677634Z","iopub.status.idle":"2023-11-23T12:52:11.817504Z","shell.execute_reply.started":"2023-11-23T12:52:11.677582Z","shell.execute_reply":"2023-11-23T12:52:11.815585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Available transaction dates per year\n- 2018 is second part of the year while 2020 has first 9 months of the year. ","metadata":{"tags":[]}},{"cell_type":"code","source":"available_dates = reference.return_available_dates()\navailable_dates.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:52:11.819729Z","iopub.execute_input":"2023-11-23T12:52:11.820441Z","iopub.status.idle":"2023-11-23T12:52:23.452931Z","shell.execute_reply.started":"2023-11-23T12:52:11.820379Z","shell.execute_reply":"2023-11-23T12:52:23.451503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del (df_prod_per_year, available_dates)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:52:23.454947Z","iopub.execute_input":"2023-11-23T12:52:23.455477Z","iopub.status.idle":"2023-11-23T12:52:23.644312Z","shell.execute_reply.started":"2023-11-23T12:52:23.455428Z","shell.execute_reply":"2023-11-23T12:52:23.642938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Distribution of products wrt to purchases over months\n\n- By this analysis, I want to check if there is a seasonality wrt the purchased products. i.e. are certain products purchased throughout the year?\n- This graph shows the number of unique months a product is purchased over a year.\n\n\nAfter running `product_type_name`, `product_group_name`, `section_name` etc. we know these columns are not granular enough to be separated by sale season or month. They provide more genereric overview of the garment.","metadata":{"tags":[]}},{"cell_type":"code","source":"df_prod_season2 = reference.prod_season()\ndf_prod_season2 = df_prod_season2[df_prod_season2.tx_month > 0]","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:52:23.646273Z","iopub.execute_input":"2023-11-23T12:52:23.647511Z","iopub.status.idle":"2023-11-23T12:55:01.226163Z","shell.execute_reply.started":"2023-11-23T12:52:23.647444Z","shell.execute_reply":"2023-11-23T12:55:01.224696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(df_prod_season2, \n       x='tx_month', \n       y= 'prod_year_prop_sale', \n       facet_col='tx_year', \n       barmode='group',\n       labels={'tx_year': 'year', 'tx_month': 'number of purchase month', 'prod_year_prop_sale': 'proportion of unique articles available during the year'})\nfig.show(width = 1200, height = 800)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:55:01.228149Z","iopub.execute_input":"2023-11-23T12:55:01.228692Z","iopub.status.idle":"2023-11-23T12:55:01.348306Z","shell.execute_reply.started":"2023-11-23T12:55:01.228643Z","shell.execute_reply":"2023-11-23T12:55:01.346306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observation\n- For 2018 we only have 4 months of data therefore, most articles were purchased all 4 months.\n- Most garments/items were bought for few month or in a season.\n- Introducing season based on the months would be easier to comprehend.\n- Month is an important feature to be considered for modelling and ranking.","metadata":{"tags":[]}},{"cell_type":"code","source":"del(df_prod_season2)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:55:01.350846Z","iopub.execute_input":"2023-11-23T12:55:01.351583Z","iopub.status.idle":"2023-11-23T12:55:01.360224Z","shell.execute_reply.started":"2023-11-23T12:55:01.351538Z","shell.execute_reply":"2023-11-23T12:55:01.35834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Does garment color and other dimension have any effect on seasonality\n- seasonality defined as month of sale","metadata":{"tags":[]}},{"cell_type":"markdown","source":"### Using `graphical_appearance_name`","metadata":{}},{"cell_type":"code","source":"df_color = reference.colour_analysis1()","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:55:01.362943Z","iopub.execute_input":"2023-11-23T12:55:01.363711Z","iopub.status.idle":"2023-11-23T12:55:52.386775Z","shell.execute_reply.started":"2023-11-23T12:55:01.363656Z","shell.execute_reply":"2023-11-23T12:55:52.384335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(df_color, \n           x='tx_month', \n           y='n_unique_customers',\n           color='graphical_appearance_name', \n           facet_row='tx_year',\n           labels={'tx_month': 'month', 'n_unique_customers': 'Unique customers', 'tx_year': 'year'})\nfig.show(width = 1200, height = 800)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:55:52.388491Z","iopub.execute_input":"2023-11-23T12:55:52.388922Z","iopub.status.idle":"2023-11-23T12:55:52.852352Z","shell.execute_reply.started":"2023-11-23T12:55:52.388888Z","shell.execute_reply":"2023-11-23T12:55:52.850688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Observation\n- \"All over pattern\" articles has the most unique customer interested in during 6th and 7th month. (can be described as popularity)\n- Too many values nothing much of relevance here.","metadata":{}},{"cell_type":"code","source":"del df_color\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:55:52.854481Z","iopub.execute_input":"2023-11-23T12:55:52.854938Z","iopub.status.idle":"2023-11-23T12:55:53.029663Z","shell.execute_reply.started":"2023-11-23T12:55:52.854902Z","shell.execute_reply":"2023-11-23T12:55:53.027846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Using `perceived_colour_value_name`","metadata":{}},{"cell_type":"code","source":"df_color = reference.colour_analysis2()","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:55:53.031435Z","iopub.execute_input":"2023-11-23T12:55:53.031875Z","iopub.status.idle":"2023-11-23T12:56:36.228363Z","shell.execute_reply.started":"2023-11-23T12:55:53.031841Z","shell.execute_reply":"2023-11-23T12:56:36.226835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(df_color, \n           x='tx_month', \n           y='n_unique_customers',\n           color='perceived_colour_value_name', \n           facet_row='tx_year',\n           labels={'tx_month': 'month', 'n_unique_customers': 'Unique customers', 'tx_year': 'year'})\nfig.show(width = 1200, height = 800)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:56:36.230257Z","iopub.execute_input":"2023-11-23T12:56:36.23068Z","iopub.status.idle":"2023-11-23T12:56:36.472781Z","shell.execute_reply.started":"2023-11-23T12:56:36.230645Z","shell.execute_reply":"2023-11-23T12:56:36.471317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Observation\n- More customers are interested in dusty light and light items during first half of the year.\n- No one wants bright color\n- Dark perceived items are more popular.","metadata":{}},{"cell_type":"markdown","source":"### Perceived color distribution wrt garment groups","metadata":{"tags":[]}},{"cell_type":"code","source":"df_color = reference.colour_analysis3()","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:56:36.474789Z","iopub.execute_input":"2023-11-23T12:56:36.475241Z","iopub.status.idle":"2023-11-23T12:56:52.562974Z","shell.execute_reply.started":"2023-11-23T12:56:36.475172Z","shell.execute_reply":"2023-11-23T12:56:52.561597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(df_color.sort_values(by='garment_group_name', ascending=False), \n           y='garment_group_name', \n           x='n_unique_articles',\n           color='perceived_colour_value_name', \n           facet_col='tx_year',\n           labels={'garment_group_name': 'Garment Group', 'n_unique_articles': 'Unique articles', 'tx_year': 'year', \n                  'perceived_colour_value_name': 'perceived color name'})\nfig.show(width = 1200, height = 800)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:56:52.565181Z","iopub.execute_input":"2023-11-23T12:56:52.566725Z","iopub.status.idle":"2023-11-23T12:56:52.834478Z","shell.execute_reply.started":"2023-11-23T12:56:52.566679Z","shell.execute_reply":"2023-11-23T12:56:52.832847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Popular segments wrt sale\n- price not a good indicator as we have partial data for 2018 and 2020.\n- Ladieswear is making the most money for the business. \n(perhaps a model can be tuned towards this segment or we proceed with caution for this bias in the dataset)","metadata":{"tags":[]}},{"cell_type":"code","source":"df_color = reference.color_analysis4()\ndf_color.sort_values(by='garment_group_name', ascending=False, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:56:52.836165Z","iopub.execute_input":"2023-11-23T12:56:52.836587Z","iopub.status.idle":"2023-11-23T12:57:02.4319Z","shell.execute_reply.started":"2023-11-23T12:56:52.836552Z","shell.execute_reply":"2023-11-23T12:57:02.429595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(df_color, \n       y='garment_group_name', \n       x='total_sale_amount', \n       facet_col='tx_year',\n       color='index_group_name',\n       barmode='group',\n       labels={'total_sale_amount': 'Total sale amount', 'garment_group_name': 'Garment Group', 'tx_year': 'year'})\n\nfig.show(width = 1200, height = 800)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:57:02.434414Z","iopub.execute_input":"2023-11-23T12:57:02.43488Z","iopub.status.idle":"2023-11-23T12:57:02.635169Z","shell.execute_reply.started":"2023-11-23T12:57:02.434844Z","shell.execute_reply":"2023-11-23T12:57:02.633398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del(df_color)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:57:02.63693Z","iopub.execute_input":"2023-11-23T12:57:02.637394Z","iopub.status.idle":"2023-11-23T12:57:02.890305Z","shell.execute_reply.started":"2023-11-23T12:57:02.637356Z","shell.execute_reply":"2023-11-23T12:57:02.888711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Customer buying patters","metadata":{}},{"cell_type":"markdown","source":"### How often a customer is buying in a given period? weekly","metadata":{}},{"cell_type":"code","source":"purchase_freq_ = reference.purchase_frequency()","metadata":{"execution":{"iopub.status.busy":"2023-11-23T13:23:45.067989Z","iopub.execute_input":"2023-11-23T13:23:45.069772Z","iopub.status.idle":"2023-11-23T13:23:45.542451Z","shell.execute_reply.started":"2023-11-23T13:23:45.069708Z","shell.execute_reply":"2023-11-23T13:23:45.540413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## filtering to removes 0s added by grouping\npurchase_freq_ = purchase_freq_[(purchase_freq_.n_transactions > 0) & (purchase_freq_.n_customers > 0)]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"purchase_freq_.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.area(data_frame=purchase_freq_[purchase_freq_.n_transactions < purchase_freq_.n_transactions.quantile(q=0.25, interpolation='nearest')],\n      x='tx_week', \n      y='n_customers', \n      facet_row='tx_year',\n      facet_col='sales_channel_id',\n      color='n_transactions',\n      labels={'n_customers': 'Number of customers', 'tx_year': 'year', 'n_transactions': 'number of tx', 'tx_week': 'week'},\n      title='Distribution of customers by number of transactions over weeks')\n\nfig.show(width = 1200, height = 800)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Observation\n- Only showing results where the number of transactions per week is within first quantile. \n- This reduces the results to display in the plot as we can see from the table most customers are generating fewer transactions per week\n- Most customers are making 1-3 transactions per week. (Model evaluation of 12 items per week/bucket seems a bit harsh or maybe the test/validation data is baised towards customers making more purchases? TBD) Maybe a successful business model needs to predict/rank 3 items most likely to be purchased next?","metadata":{}},{"cell_type":"code","source":"del(purchase_freq_)\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Distribution of repeating customers (weekly purchase)\n- The grapgh shows the distribution of customers over number of purchase weeks per year and sale channel id.\n- Long tail dstribution: most customers have made purchases (or were active) for 1-3 weeks. (Model needs to generalise well for customers with fewer transactions )","metadata":{}},{"cell_type":"code","source":"a = time()\nrep_customer = reference.repeating_customer()\n## aggregation results in 0 accumulation \nrep_customer = rep_customer[rep_customer.tx_week > 0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(rep_customer, x='tx_week', facet_col='sales_channel_id', facet_row='tx_year', \n      labels={'count': 'Number of customers', 'tx_year': 'year', 'tx_week': 'week'})\nfig.show(width = 1200, height = 800)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(time() - a)\ndel (rep_customer)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Distribution of customers over transactions  \nNumber of customers making transactions per year\n- Very few cusotmers make large number of transactions per year. (possible outliers)","metadata":{}},{"cell_type":"code","source":"a= time()\ndf_purchase_freq = reference.purchase_freq()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(data_frame=df_purchase_freq, \n       y='customer_id', \n       x='t_dat',\n       labels={'t_dat': 'number of transactions', 'customer_id': 'number of customers', 'tx_year': 'year'},\n             title='Distribution of customers over purchase transactions',\n       facet_row='tx_year')\n\nfig.update_traces(marker_color='red')\nfig.show(width = 1200, height = 800)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(time() - a)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del(df_purchase_freq)\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Average gap between purchases","metadata":{}},{"cell_type":"code","source":"df_results = reference.purchase_gap()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_results1 = df_results.groupby(['average_gap_bw_purchase_days', 'tx_year']).count().reset_index()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.line(data_frame=df_results1, \n           x ='average_gap_bw_purchase_days',\n           y = 'customer_id',\n           color='tx_year',\n       title='Average gap between two consecutive purchases',\n       labels={'average_gap_bw_purchase_days': 'days', 'customer_id': 'Number of customers'})\n\nfig.show(width = 1200, height = 800)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Observation\n- The graph shows the difference in days between two consecutive purchases by a customer. Averaged over all customer per year.\n- Average gap of 0 days reflect cusotmers with single purchase day (also shown in the previous graph for purchase frequency)\n- Therefore, the model need to generalise better on customers with fewer transactions. (hence, it will make sense to evaluate these segments separately)","metadata":{}},{"cell_type":"code","source":"del(df_results)\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### What products and garment groups are bought together?\nFollowing observations are based on random sample of 1% transactions","metadata":{}},{"cell_type":"code","source":"test_df = reference.bought_together()\ntest_df.sort_values(by='n_unique_custs', inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.scatter(data_frame=test_df, \n           y='garment_group_name_together', \n           x='n_unique_custs', \n           size='total_purchase_days',\n          title='Distribution of garment groups - daily bucket (bought together)')\n\nfig.show(width = 1200, height = 800)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observation\n- Garment groups with top 2% unique customers are shown in the graph.\n- The size refers to the number of transaction days for each garment group.\n- X: number of unique customers making the purchase. Y: garment group bought together.\n- Most customers and purchases include single garment group. (what this means for modelling? we can use the items bought together to rank the predictions. Since most customers buys from one garment group this might not provide significant gain in performance TBD)","metadata":{}},{"cell_type":"code","source":"del (test_df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Customer rebuying frequency \n- After the first purchase what's the next two purchases (displayed by the garment group)","metadata":{}},{"cell_type":"code","source":"new_df_purchase = reference.rebuying_frequency()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(data_frame=new_df_purchase.sort_values(by='next_garment_group_name'),\n           color='garment_group_name',\n           y='next_garment_group_name',\n       labels={'t_dat': 'number of transactions', 'garment_group_name': 'Purchased group', 'next_garment_group_name': 'Next purchases'},\n           x='t_dat')\n\nfig.show(width = 1200, height = 800)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del (new_df_purchase)\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}