{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Overview\n\nThis is such a fun dataset for exploration! It has images, it has product details, and it has transaction history. I made this visualization snippet to get more familiar with the collection of articles / items from H&M, as well as get a feel of what the products are and perhaps a glance on customer profile.\n\nBig thanks to the following notebook(s) that gave a lot of ideas and implementation steps for the snippets:\n- https://www.kaggle.com/vanguarde/h-m-eda-first-look","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\nfrom tqdm.notebook import tqdm\nimport plotly.express as px\nimport matplotlib.image as mpimg\n\nimport warnings \nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-20T09:37:28.290117Z","iopub.execute_input":"2022-02-20T09:37:28.290608Z","iopub.status.idle":"2022-02-20T09:37:31.115089Z","shell.execute_reply.started":"2022-02-20T09:37:28.290376Z","shell.execute_reply":"2022-02-20T09:37:31.113821Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Overview of Data","metadata":{}},{"cell_type":"code","source":"articles = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\ncustomers = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/customers.csv\")\ntransactions = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-02-20T09:37:31.117357Z","iopub.execute_input":"2022-02-20T09:37:31.117675Z","iopub.status.idle":"2022-02-20T09:38:52.331138Z","shell.execute_reply.started":"2022-02-20T09:37:31.117642Z","shell.execute_reply":"2022-02-20T09:38:52.329899Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(articles.columns)\narticles.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-20T09:38:52.332742Z","iopub.execute_input":"2022-02-20T09:38:52.333258Z","iopub.status.idle":"2022-02-20T09:38:52.408484Z","shell.execute_reply.started":"2022-02-20T09:38:52.333202Z","shell.execute_reply":"2022-02-20T09:38:52.407003Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(customers.columns)\ncustomers.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-20T09:38:56.761907Z","iopub.execute_input":"2022-02-20T09:38:56.763162Z","iopub.status.idle":"2022-02-20T09:38:56.787573Z","shell.execute_reply.started":"2022-02-20T09:38:56.763095Z","shell.execute_reply":"2022-02-20T09:38:56.786665Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(transactions.columns)\ntransactions.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-20T09:38:56.788825Z","iopub.execute_input":"2022-02-20T09:38:56.789504Z","iopub.status.idle":"2022-02-20T09:38:56.813131Z","shell.execute_reply.started":"2022-02-20T09:38:56.789461Z","shell.execute_reply":"2022-02-20T09:38:56.812275Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_SAMPLE = 1000000\ntransactionsSample = transactions.sample(n=N_SAMPLE)","metadata":{"execution":{"iopub.status.busy":"2022-02-20T09:38:56.814461Z","iopub.execute_input":"2022-02-20T09:38:56.814734Z","iopub.status.idle":"2022-02-20T09:39:00.565620Z","shell.execute_reply.started":"2022-02-20T09:38:56.814702Z","shell.execute_reply":"2022-02-20T09:39:00.564480Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"article_volume = transactionsSample.groupby('article_id')['t_dat'].count().sort_values(ascending=False).reset_index()\narticle_volume.columns = ['article_id','volume']\narticle_volume.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-20T09:39:00.567011Z","iopub.execute_input":"2022-02-20T09:39:00.567330Z","iopub.status.idle":"2022-02-20T09:39:00.801607Z","shell.execute_reply.started":"2022-02-20T09:39:00.567292Z","shell.execute_reply":"2022-02-20T09:39:00.800357Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_withVolume = pd.merge(articles,article_volume,on=['article_id'],how='left')\narticles_withVolume.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-02-20T09:39:00.803611Z","iopub.execute_input":"2022-02-20T09:39:00.803878Z","iopub.status.idle":"2022-02-20T09:39:01.119599Z","shell.execute_reply.started":"2022-02-20T09:39:00.803847Z","shell.execute_reply":"2022-02-20T09:39:01.117812Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"jupyter":{"source_hidden":true}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Treemap to get better feel of different product categories and how large they are\nComparing treemap based on # articles and sales volume","metadata":{}},{"cell_type":"code","source":"articles['ones'] = 1.0  # to count number of rows\npx.treemap(articles, path=['index_group_name','index_name','product_group_name', 'product_type_name'],\n                values='ones', title='Tree Map based on Article ID')\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-20T09:42:26.293154Z","iopub.execute_input":"2022-02-20T09:42:26.293964Z","iopub.status.idle":"2022-02-20T09:42:28.954453Z","shell.execute_reply.started":"2022-02-20T09:42:26.293905Z","shell.execute_reply":"2022-02-20T09:42:28.953675Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.treemap(articles_withVolume, path=['index_group_name','index_name','product_group_name', 'product_type_name'],\n                values='volume', title='Tree Map based on Sales Volume')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-20T09:41:55.692666Z","iopub.execute_input":"2022-02-20T09:41:55.694869Z","iopub.status.idle":"2022-02-20T09:41:58.342491Z","shell.execute_reply.started":"2022-02-20T09:41:55.694748Z","shell.execute_reply":"2022-02-20T09:41:58.341561Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.treemap(articles_withVolume, path=['index_group_name','section_name','product_group_name', 'product_type_name'],\n                values='volume', title='Tree Map based on Sales Volume')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-20T09:43:58.406538Z","iopub.execute_input":"2022-02-20T09:43:58.407479Z","iopub.status.idle":"2022-02-20T09:44:01.159561Z","shell.execute_reply.started":"2022-02-20T09:43:58.407404Z","shell.execute_reply":"2022-02-20T09:44:01.158720Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Functions","metadata":{}},{"cell_type":"markdown","source":"### Function: Given an article_id, get the volume and price history (weekly)","metadata":{}},{"cell_type":"code","source":"def getArticlePriceHistory(article_id):\n    dfTrxArticle = transactions[transactions.article_id == article_id]\n    dfTrxArticle['priceK'] = dfTrxArticle.price * 1000\n    dfTrxArticle['t_dat'] = pd.to_datetime(dfTrxArticle['t_dat'])\n    series_mean = dfTrxArticle[['t_dat', 'priceK']].groupby(pd.Grouper(key=\"t_dat\", freq='W')).mean()\n    series_stdev = dfTrxArticle[['t_dat', 'priceK']].groupby(pd.Grouper(key=\"t_dat\", freq='W')).std().fillna(0)\n    series_volume = dfTrxArticle[['t_dat', 'priceK']].groupby(pd.Grouper(key=\"t_dat\", freq='W')).count().fillna(0)\n    dfArticlePriceHistory = pd.DataFrame({'price_avg':series_mean['priceK'],'price_std':series_stdev['priceK'],'volume':series_volume['priceK']},index=series_volume.index)\n    dfArticlePriceHistory['lower'] = dfArticlePriceHistory.price_avg - 2 * dfArticlePriceHistory.price_std\n    dfArticlePriceHistory['upper'] = dfArticlePriceHistory.price_avg + 2 * dfArticlePriceHistory.price_std\n    return dfArticlePriceHistory","metadata":{"execution":{"iopub.status.busy":"2022-02-20T09:39:03.873859Z","iopub.execute_input":"2022-02-20T09:39:03.874301Z","iopub.status.idle":"2022-02-20T09:39:03.887004Z","shell.execute_reply.started":"2022-02-20T09:39:03.874263Z","shell.execute_reply":"2022-02-20T09:39:03.885384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Function: Given an article_id, get the img object","metadata":{}},{"cell_type":"code","source":"def getImgFromArticle(article_id):\n    subfolder = '0'+str(article_id)[:2]\n    filename = '0'+str(article_id)+'.jpg'\n    filename_root = '../input/h-and-m-personalized-fashion-recommendations/images/'\n    filename_path = filename_root + subfolder + '/' + filename\n    img = mpimg.imread(filename_path)\n    return img","metadata":{"execution":{"iopub.status.busy":"2022-02-20T09:39:03.888865Z","iopub.execute_input":"2022-02-20T09:39:03.889176Z","iopub.status.idle":"2022-02-20T09:39:03.910300Z","shell.execute_reply.started":"2022-02-20T09:39:03.889138Z","shell.execute_reply":"2022-02-20T09:39:03.909366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Function: Get the article info as dictionary for an article_id","metadata":{}},{"cell_type":"code","source":"def getArticleInfo(article_id):\n    dictArticleInfo = articles[articles.article_id==article_id].reset_index().iloc[0].to_dict()\n    return dictArticleInfo","metadata":{"execution":{"iopub.status.busy":"2022-02-20T09:39:03.911773Z","iopub.execute_input":"2022-02-20T09:39:03.912225Z","iopub.status.idle":"2022-02-20T09:39:03.926870Z","shell.execute_reply.started":"2022-02-20T09:39:03.912170Z","shell.execute_reply":"2022-02-20T09:39:03.925548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Function: Visualize one set of info group for an article_id","metadata":{}},{"cell_type":"code","source":"def visualizeRowArticle(article_id):\n    fig, axes = plt.subplots(1,2,figsize=(15,5))\n    try:\n        imgSample = getImgFromArticle(article_id)\n        axes[0].imshow(imgSample)\n        axes[0].set_title('Product Image')\n    except:\n        axes[0].set_title('Product Image is Missing')\n    dfArticlePriceHistory = getArticlePriceHistory(article_id)\n    dictArticleInfo = getArticleInfo(article_id)   \n    axes[1].plot(dfArticlePriceHistory.price_avg, color='red', label='Prices')\n    axes[1].fill_between(dfArticlePriceHistory.index, dfArticlePriceHistory.lower, dfArticlePriceHistory.upper, color='grey',alpha=0.1)\n    axes[1].set_ylabel('Price')\n    axes[1].legend(loc=2) # upper left\n    ax1_twin = axes[1].twinx()\n    ax1_twin.bar(x=dfArticlePriceHistory.index,height=dfArticlePriceHistory.volume, color='blue', label='Volume')\n    ax1_twin.set_ylabel('Volume')        \n    ax1_twin.legend(loc=1) # upper right\n    axes[1].set_title('Historical Price Chart')\n    plt.suptitle(dictArticleInfo['prod_name'] + ':\\n' + dictArticleInfo['detail_desc'],horizontalalignment='left',x=0.1, y=1.05)","metadata":{"execution":{"iopub.status.busy":"2022-02-20T09:52:47.692709Z","iopub.execute_input":"2022-02-20T09:52:47.693141Z","iopub.status.idle":"2022-02-20T09:52:47.708559Z","shell.execute_reply.started":"2022-02-20T09:52:47.693102Z","shell.execute_reply":"2022-02-20T09:52:47.707222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Function: Divider title between sections of plotting","metadata":{}},{"cell_type":"code","source":"def createDividerTitle(title='Chart',color='mistyrose'):\n    fig,axes = plt.subplots(figsize=(20,1), facecolor=color)\n    axes.axis('off')\n    plt.text(0.01,0.5,title,dict(size=20))","metadata":{"execution":{"iopub.status.busy":"2022-02-20T10:23:31.771983Z","iopub.execute_input":"2022-02-20T10:23:31.772287Z","iopub.status.idle":"2022-02-20T10:23:31.778911Z","shell.execute_reply.started":"2022-02-20T10:23:31.772255Z","shell.execute_reply":"2022-02-20T10:23:31.777862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Run functions to explore a couple of products (articles)","metadata":{}},{"cell_type":"code","source":"listArticle = [736489010,505221004,610776002]\nfor article_id in listArticle:\n    visualizeRowArticle(article_id)","metadata":{"execution":{"iopub.status.busy":"2022-02-20T09:52:55.512883Z","iopub.execute_input":"2022-02-20T09:52:55.513367Z","iopub.status.idle":"2022-02-20T09:52:59.436131Z","shell.execute_reply.started":"2022-02-20T09:52:55.513314Z","shell.execute_reply":"2022-02-20T09:52:59.434991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualize top selling articles per year","metadata":{}},{"cell_type":"code","source":"transactionsSample = transactions.sample(n=100000)\ntransactionsSample['t_dat'] = pd.to_datetime(transactionsSample['t_dat']) \ntransactionsSample['year'] = pd.DatetimeIndex(transactionsSample['t_dat']).year\ntransactionsSample.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-20T10:05:53.769807Z","iopub.execute_input":"2022-02-20T10:05:53.770114Z","iopub.status.idle":"2022-02-20T10:05:56.566872Z","shell.execute_reply.started":"2022-02-20T10:05:53.770084Z","shell.execute_reply":"2022-02-20T10:05:56.564514Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"groupedTrx = transactionsSample.groupby(['year','article_id'])['customer_id'].count().reset_index()\ngroupedTrx.columns = ['year','article_id','count']","metadata":{"execution":{"iopub.status.busy":"2022-02-20T09:39:09.216357Z","iopub.execute_input":"2022-02-20T09:39:09.216884Z","iopub.status.idle":"2022-02-20T09:39:09.239298Z","shell.execute_reply.started":"2022-02-20T09:39:09.216826Z","shell.execute_reply":"2022-02-20T09:39:09.238505Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"years = groupedTrx.year.unique()\nfor year in years:\n    dfYear = groupedTrx[groupedTrx.year==year]\n    dfYear = dfYear.sort_values(by='count',ascending=False)\n    topArticleId = dfYear[:10].article_id.values    \n    titleText = \"Top Articles in Year {}\".format(year)\n    createDividerTitle(title=titleText,color='mistyrose')\n    for article_id in topArticleId:\n        visualizeRowArticle(article_id)","metadata":{"execution":{"iopub.status.busy":"2022-02-20T10:24:26.884152Z","iopub.execute_input":"2022-02-20T10:24:26.884736Z","iopub.status.idle":"2022-02-20T10:25:08.510853Z","shell.execute_reply.started":"2022-02-20T10:24:26.884688Z","shell.execute_reply":"2022-02-20T10:25:08.509505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualize top products for each Index category, based on 2020 Sales","metadata":{}},{"cell_type":"code","source":"transactionsSample.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-20T10:04:54.712423Z","iopub.execute_input":"2022-02-20T10:04:54.712794Z","iopub.status.idle":"2022-02-20T10:04:54.729346Z","shell.execute_reply.started":"2022-02-20T10:04:54.712756Z","shell.execute_reply":"2022-02-20T10:04:54.728375Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get top index name\narticle_volume2020 = transactionsSample[transactionsSample['year']==2020].groupby('article_id')['t_dat'].count().sort_values(ascending=False).reset_index()\narticle_volume2020.columns = ['article_id','volume']\narticles_withVolume2020 = pd.merge(articles,article_volume2020,on=['article_id'],how='left')\ndfTopIndex = articles_withVolume2020.groupby('index_name')['volume'].sum().sort_values(ascending=False)\ndfTopIndex.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-02-20T10:07:46.089385Z","iopub.execute_input":"2022-02-20T10:07:46.090844Z","iopub.status.idle":"2022-02-20T10:07:46.438975Z","shell.execute_reply.started":"2022-02-20T10:07:46.090694Z","shell.execute_reply":"2022-02-20T10:07:46.438026Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index = 'Menswear'\ndfTopArticleID = articles_withVolume2020[articles_withVolume2020.index_name==index].sort_values(by='volume',ascending=False)\nlistTopArticleID = dfTopArticleID.head(5)['article_id'].values\nlistTopArticleID","metadata":{"execution":{"iopub.status.busy":"2022-02-20T10:15:54.753541Z","iopub.execute_input":"2022-02-20T10:15:54.753931Z","iopub.status.idle":"2022-02-20T10:15:54.798756Z","shell.execute_reply.started":"2022-02-20T10:15:54.753895Z","shell.execute_reply":"2022-02-20T10:15:54.798045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index = 'Ladieswear'\ndfTopArticleID = articles_withVolume2020[articles_withVolume2020.index_name==index].sort_values(by='volume',ascending=False)\nlistTopArticleID = dfTopArticleID.head(5)['article_id'].values\nlistTopArticleID","metadata":{"execution":{"iopub.status.busy":"2022-02-20T10:16:03.820999Z","iopub.execute_input":"2022-02-20T10:16:03.821593Z","iopub.status.idle":"2022-02-20T10:16:03.873162Z","shell.execute_reply.started":"2022-02-20T10:16:03.821551Z","shell.execute_reply":"2022-02-20T10:16:03.871779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_withVolume2020.index_name.unique()","metadata":{"execution":{"iopub.status.busy":"2022-02-20T10:26:10.495339Z","iopub.execute_input":"2022-02-20T10:26:10.495805Z","iopub.status.idle":"2022-02-20T10:26:10.517085Z","shell.execute_reply.started":"2022-02-20T10:26:10.495753Z","shell.execute_reply":"2022-02-20T10:26:10.516145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colorIndex = {}\nfor index in articles_withVolume2020.index_name.unique():\n    colorIndex[index] = 'lightgrey'\ncolorIndex['Ladieswear'] = 'lightcoral'\ncolorIndex['Ladies ACcessories'] = 'lightcoral'\ncolorIndex['Menswear'] = 'lightskyblue'\ncolorIndex['Lingeries/Tights'] = 'purple'","metadata":{"execution":{"iopub.status.busy":"2022-02-20T10:28:02.741606Z","iopub.execute_input":"2022-02-20T10:28:02.742005Z","iopub.status.idle":"2022-02-20T10:28:02.759390Z","shell.execute_reply.started":"2022-02-20T10:28:02.741968Z","shell.execute_reply":"2022-02-20T10:28:02.757979Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index in dfTopIndex.head(10).index:\n    dfTopArticleID = articles_withVolume2020[articles_withVolume2020.index_name==index].sort_values(by='volume',ascending=False)\n    listTopArticleID = dfTopArticleID.head(5)['article_id'].values\n    titleText = \"Top Articles in Index {} for Year 2020\".format(index)\n    createDividerTitle(title=titleText,color=colorIndex[index])\n    for article_id in listTopArticleID:\n        visualizeRowArticle(article_id)       ","metadata":{"execution":{"iopub.status.busy":"2022-02-20T10:28:12.009595Z","iopub.execute_input":"2022-02-20T10:28:12.009924Z","iopub.status.idle":"2022-02-20T10:29:20.222190Z","shell.execute_reply.started":"2022-02-20T10:28:12.009882Z","shell.execute_reply":"2022-02-20T10:29:20.221190Z"},"trusted":true},"execution_count":null,"outputs":[]}]}