{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## EDA for H&M dataset  \n\n### It also enables panadas dataframe optimization for reducing memory usage\n\nReference Notebook Link: https://www.kaggle.com/vanguarde/h-m-eda-first-look/notebook  \n\nThe dataset comprises of 5 segregations, please find the details below:  \n1. images: This folder consists of images for most of the articles that are being sold in H&M however all articls don't have respective images.  \n2. articles.csv: This csv file consists of metadata for each of the article which are getting sold under H&M banner. It has total 25 columns for each article including metadata information such as product type, product name, product description, etc.  \n3. customers.csv: This csv file consists of metadata for each customer id. It has total 7 columns and includes general information regarding a speicific customer such as H&M club member status, age, etc.  \n4. transactions_train.csv: This csv file consists of transaction information of any specific article id which any of the customer id has purchased and at what price point the transactions were made and also when was the transaction made.  \n5. sample_submission.csv: This csv files consists of customer ids for which the predictions are to be made as to what are the list of article ids that a specific customer id will purchase within a span of one week i.e. 7 days after the training date ends.","metadata":{}},{"cell_type":"code","source":"# Importing Libraries\nimport re\nimport gc\nimport string\nimport spacy\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom textwrap import wrap\nfrom wordcloud import WordCloud\nfrom sklearn.feature_extraction.text import CountVectorizer\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:12:59.042622Z","iopub.execute_input":"2022-03-03T20:12:59.043293Z","iopub.status.idle":"2022-03-03T20:13:08.707699Z","shell.execute_reply.started":"2022-03-03T20:12:59.043162Z","shell.execute_reply":"2022-03-03T20:13:08.707134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Optimize dataframes\ndef reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2    \n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)    \n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose: print('Mem. usage decreased to {:5.2f} Mb ({:.1f}% reduction)'.format(end_mem, 100 * (start_mem - end_mem) / start_mem))\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:13:08.709137Z","iopub.execute_input":"2022-03-03T20:13:08.709447Z","iopub.status.idle":"2022-03-03T20:13:08.722112Z","shell.execute_reply.started":"2022-03-03T20:13:08.709422Z","shell.execute_reply":"2022-03-03T20:13:08.721388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_data = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\ncustomers_data = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/customers.csv\")\ntransactions_data = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:13:08.723544Z","iopub.execute_input":"2022-03-03T20:13:08.723757Z","iopub.status.idle":"2022-03-03T20:14:20.290436Z","shell.execute_reply.started":"2022-03-03T20:13:08.723733Z","shell.execute_reply":"2022-03-03T20:14:20.289732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Optimizing the dataframes\narticles_data = reduce_mem_usage(articles_data)\ncustomers_data = reduce_mem_usage(customers_data)\ntransactions_data = reduce_mem_usage(transactions_data)","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:20.291555Z","iopub.execute_input":"2022-03-03T20:14:20.291752Z","iopub.status.idle":"2022-03-03T20:14:20.962419Z","shell.execute_reply.started":"2022-03-03T20:14:20.291729Z","shell.execute_reply":"2022-03-03T20:14:20.961451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:20.966315Z","iopub.execute_input":"2022-03-03T20:14:20.966559Z","iopub.status.idle":"2022-03-03T20:14:21.167431Z","shell.execute_reply.started":"2022-03-03T20:14:20.966531Z","shell.execute_reply":"2022-03-03T20:14:21.166622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Exploring Articles Metadata","metadata":{}},{"cell_type":"code","source":"# List all columns present in article metadata\narticles_data.columns","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:21.168951Z","iopub.execute_input":"2022-03-03T20:14:21.169152Z","iopub.status.idle":"2022-03-03T20:14:21.178658Z","shell.execute_reply.started":"2022-03-03T20:14:21.169127Z","shell.execute_reply":"2022-03-03T20:14:21.177863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:21.17962Z","iopub.execute_input":"2022-03-03T20:14:21.179854Z","iopub.status.idle":"2022-03-03T20:14:21.213077Z","shell.execute_reply.started":"2022-03-03T20:14:21.179821Z","shell.execute_reply":"2022-03-03T20:14:21.212318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(15, 8))\nax = sns.histplot(data=articles_data, y='index_name', color='blue')\nax.set_xlabel('count by index name')\nax.set_ylabel('index name')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:21.214568Z","iopub.execute_input":"2022-03-03T20:14:21.215029Z","iopub.status.idle":"2022-03-03T20:14:21.621946Z","shell.execute_reply.started":"2022-03-03T20:14:21.214989Z","shell.execute_reply":"2022-03-03T20:14:21.621235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the above horizontal bar graph, its easily visible that out of 10 categories of products, the count of ladieswear articles is the highest in the metadata and is more than 25000 in number followed by Divided, Menswear, etc.","metadata":{}},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(15, 8))\nax = sns.histplot(data=articles_data, y='garment_group_name', color='orange', hue='index_group_name', multiple=\"stack\")\nax.set_xlabel('count by garment group')\nax.set_ylabel('garment group')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:21.623045Z","iopub.execute_input":"2022-03-03T20:14:21.623237Z","iopub.status.idle":"2022-03-03T20:14:22.543305Z","shell.execute_reply.started":"2022-03-03T20:14:21.623214Z","shell.execute_reply":"2022-03-03T20:14:22.542595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The garments grouped by index: Jersey fancy is the most frequent garment, especially for women and baby/children. The next by number is accessories which is followed by Jersey Basic, Under-nightwear, etc","metadata":{}},{"cell_type":"markdown","source":"Lets look up to the combination counts for index group name and index name","metadata":{}},{"cell_type":"code","source":"articles_data.groupby(['index_group_name', 'index_name']).count()['article_id']","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:22.544449Z","iopub.execute_input":"2022-03-03T20:14:22.544704Z","iopub.status.idle":"2022-03-03T20:14:22.736298Z","shell.execute_reply.started":"2022-03-03T20:14:22.544659Z","shell.execute_reply":"2022-03-03T20:14:22.735695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets look up to the combination counts for product group name and product name","metadata":{}},{"cell_type":"code","source":"pd.options.display.max_rows = None\narticles_data.groupby(['product_group_name', 'product_type_name']).count()['article_id']","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:22.737392Z","iopub.execute_input":"2022-03-03T20:14:22.738009Z","iopub.status.idle":"2022-03-03T20:14:22.923412Z","shell.execute_reply.started":"2022-03-03T20:14:22.737964Z","shell.execute_reply":"2022-03-03T20:14:22.922541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract number of unique columns values under each column\nfor col_name in articles_data.columns:\n    un_n = articles_data[col_name].nunique()\n    print(f'No of unique {col_name}: {un_n}')","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:22.925099Z","iopub.execute_input":"2022-03-03T20:14:22.925466Z","iopub.status.idle":"2022-03-03T20:14:23.095454Z","shell.execute_reply.started":"2022-03-03T20:14:22.925423Z","shell.execute_reply":"2022-03-03T20:14:23.094584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Exploring Textual description of articles\n\n### Steps inolved:\n#### Preprocessing Steps:\n1. Check missing values and remove them\n2. Expand Contractions\n3. Lowercase the descriptions\n4. Remove digits and words containing digits \n5. Remove punctuations\n\n#### EDA for Textual Description:\n1. Stopwords Removal  \n2. Lemmatization  \n3. Creating Document Matrix","metadata":{}},{"cell_type":"code","source":"articles_desc_data = articles_data[['article_id','detail_desc']]","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:23.096589Z","iopub.execute_input":"2022-03-03T20:14:23.096841Z","iopub.status.idle":"2022-03-03T20:14:23.10248Z","shell.execute_reply.started":"2022-03-03T20:14:23.09679Z","shell.execute_reply":"2022-03-03T20:14:23.101741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Preprocessing Steps\n\n#### 1. Check missing values and remove them\n","metadata":{}},{"cell_type":"code","source":"# Counting number of null detailed descriptions for articles\narticles_desc_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:23.105969Z","iopub.execute_input":"2022-03-03T20:14:23.106155Z","iopub.status.idle":"2022-03-03T20:14:23.129423Z","shell.execute_reply.started":"2022-03-03T20:14:23.106131Z","shell.execute_reply":"2022-03-03T20:14:23.128596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_desc_data.dropna(inplace=True)\narticles_desc_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:23.130745Z","iopub.execute_input":"2022-03-03T20:14:23.131409Z","iopub.status.idle":"2022-03-03T20:14:23.173329Z","shell.execute_reply.started":"2022-03-03T20:14:23.131365Z","shell.execute_reply":"2022-03-03T20:14:23.172469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_desc_data['detail_desc'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:23.174663Z","iopub.execute_input":"2022-03-03T20:14:23.175366Z","iopub.status.idle":"2022-03-03T20:14:23.220822Z","shell.execute_reply.started":"2022-03-03T20:14:23.17533Z","shell.execute_reply":"2022-03-03T20:14:23.220063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 2. Remove Contractions","metadata":{}},{"cell_type":"code","source":"# Dictionary of English Contractions\ncontractions_dict = { \"ain't\": \"are not\",\"'s\":\" is\",\"aren't\": \"are not\",\n                     \"can't\": \"cannot\",\"can't've\": \"cannot have\",\n                     \"'cause\": \"because\",\"could've\": \"could have\",\"couldn't\": \"could not\",\n                     \"couldn't've\": \"could not have\", \"didn't\": \"did not\",\"doesn't\": \"does not\",\n                     \"don't\": \"do not\",\"hadn't\": \"had not\",\"hadn't've\": \"had not have\",\n                     \"hasn't\": \"has not\",\"haven't\": \"have not\",\"he'd\": \"he would\",\n                     \"he'd've\": \"he would have\",\"he'll\": \"he will\", \"he'll've\": \"he will have\",\n                     \"how'd\": \"how did\",\"how'd'y\": \"how do you\",\"how'll\": \"how will\",\n                     \"I'd\": \"I would\", \"I'd've\": \"I would have\",\"I'll\": \"I will\",\n                     \"I'll've\": \"I will have\",\"I'm\": \"I am\",\"I've\": \"I have\", \"isn't\": \"is not\",\n                     \"it'd\": \"it would\",\"it'd've\": \"it would have\",\"it'll\": \"it will\",\n                     \"it'll've\": \"it will have\", \"let's\": \"let us\",\"ma'am\": \"madam\",\n                     \"mayn't\": \"may not\",\"might've\": \"might have\",\"mightn't\": \"might not\", \n                     \"mightn't've\": \"might not have\",\"must've\": \"must have\",\"mustn't\": \"must not\",\n                     \"mustn't've\": \"must not have\", \"needn't\": \"need not\",\n                     \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\",\"oughtn't\": \"ought not\",\n                     \"oughtn't've\": \"ought not have\",\"shan't\": \"shall not\",\"sha'n't\": \"shall not\",\n                     \"shan't've\": \"shall not have\",\"she'd\": \"she would\",\"she'd've\": \"she would have\",\n                     \"she'll\": \"she will\", \"she'll've\": \"she will have\",\"should've\": \"should have\",\n                     \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\",\"so've\": \"so have\",\n                     \"that'd\": \"that would\",\"that'd've\": \"that would have\", \"there'd\": \"there would\",\n                     \"there'd've\": \"there would have\", \"they'd\": \"they would\",\n                     \"they'd've\": \"they would have\",\"they'll\": \"they will\",\n                     \"they'll've\": \"they will have\", \"they're\": \"they are\",\"they've\": \"they have\",\n                     \"to've\": \"to have\",\"wasn't\": \"was not\",\"we'd\": \"we would\",\n                     \"we'd've\": \"we would have\",\"we'll\": \"we will\",\"we'll've\": \"we will have\",\n                     \"we're\": \"we are\",\"we've\": \"we have\", \"weren't\": \"were not\",\"what'll\": \"what will\",\n                     \"what'll've\": \"what will have\",\"what're\": \"what are\", \"what've\": \"what have\",\n                     \"when've\": \"when have\",\"where'd\": \"where did\", \"where've\": \"where have\",\n                     \"who'll\": \"who will\",\"who'll've\": \"who will have\",\"who've\": \"who have\",\n                     \"why've\": \"why have\",\"will've\": \"will have\",\"won't\": \"will not\",\n                     \"won't've\": \"will not have\", \"would've\": \"would have\",\"wouldn't\": \"would not\",\n                     \"wouldn't've\": \"would not have\",\"y'all\": \"you all\", \"y'all'd\": \"you all would\",\n                     \"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\n                     \"y'all've\": \"you all have\", \"you'd\": \"you would\",\"you'd've\": \"you would have\",\n                     \"you'll\": \"you will\",\"you'll've\": \"you will have\", \"you're\": \"you are\",\n                     \"you've\": \"you have\"}\n\n# Regular expression for finding contractions\ncontractions_re=re.compile('(%s)' % '|'.join(contractions_dict.keys()))","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:23.221976Z","iopub.execute_input":"2022-03-03T20:14:23.222323Z","iopub.status.idle":"2022-03-03T20:14:23.239223Z","shell.execute_reply.started":"2022-03-03T20:14:23.222297Z","shell.execute_reply":"2022-03-03T20:14:23.23865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function for expanding contractions\ndef expand_contractions(text,contractions_dict=contractions_dict):\n  def replace(match):\n    return contractions_dict[match.group(0)]\n  return contractions_re.sub(replace, text)\n\n# Expanding Contractions in the reviews\narticles_desc_data['detail_desc']=articles_desc_data['detail_desc'].apply(lambda x:expand_contractions(x))","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:23.240117Z","iopub.execute_input":"2022-03-03T20:14:23.240581Z","iopub.status.idle":"2022-03-03T20:14:27.535985Z","shell.execute_reply.started":"2022-03-03T20:14:23.240549Z","shell.execute_reply":"2022-03-03T20:14:27.535176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The expand_contractions function uses regular expressions to map the contractions in the text to their expanded forms from the dictionary. ","metadata":{}},{"cell_type":"markdown","source":"#### 3. Lowercase the decriptions","metadata":{}},{"cell_type":"code","source":"articles_desc_data['detail_desc_cleaned']=articles_desc_data['detail_desc'].apply(lambda x: x.lower())","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:27.537093Z","iopub.execute_input":"2022-03-03T20:14:27.537304Z","iopub.status.idle":"2022-03-03T20:14:27.601107Z","shell.execute_reply.started":"2022-03-03T20:14:27.537277Z","shell.execute_reply":"2022-03-03T20:14:27.600127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 4. Remove digits and words containing digits","metadata":{}},{"cell_type":"code","source":"articles_desc_data['detail_desc_cleaned']=articles_desc_data['detail_desc_cleaned'].apply(lambda x: re.sub('\\w*\\d\\w*','', x))","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:27.602553Z","iopub.execute_input":"2022-03-03T20:14:27.603465Z","iopub.status.idle":"2022-03-03T20:14:30.781548Z","shell.execute_reply.started":"2022-03-03T20:14:27.603419Z","shell.execute_reply":"2022-03-03T20:14:30.7805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 5. Remove punctuations and spaces","metadata":{}},{"cell_type":"code","source":"articles_desc_data['detail_desc_cleaned']=articles_desc_data['detail_desc_cleaned'].apply(lambda x: re.sub('[%s]' % re.escape(string.punctuation), '', x))\narticles_desc_data['detail_desc_cleaned']=articles_desc_data['detail_desc_cleaned'].apply(lambda x: re.sub(' +',' ',x))","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:30.782859Z","iopub.execute_input":"2022-03-03T20:14:30.783292Z","iopub.status.idle":"2022-03-03T20:14:32.830378Z","shell.execute_reply.started":"2022-03-03T20:14:30.783253Z","shell.execute_reply":"2022-03-03T20:14:32.829525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_desc_data['detail_desc_cleaned'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:32.831493Z","iopub.execute_input":"2022-03-03T20:14:32.831769Z","iopub.status.idle":"2022-03-03T20:14:32.88492Z","shell.execute_reply.started":"2022-03-03T20:14:32.831742Z","shell.execute_reply":"2022-03-03T20:14:32.884085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### EDA for Textual Description\n\n#### \n1. Stopwords Removal \n2. Lemmatization  \n3. Finally, we will utilize word cloud library to represent the data\n\nWe will use spacy to remove the stopwords and present the lemma form of toekns in article descriptions","metadata":{}},{"cell_type":"code","source":"# Loading english spacy model\nnlp = spacy.load('en_core_web_sm',disable=['parser', 'ner'])\n\nlemmatized_desc_list = [(lambda x: ' '.join([token.lemma_ for token in list(nlp(x)) if (token.is_stop==False)]))(x) for x in articles_desc_data['detail_desc_cleaned'].unique()]\nlen(lemmatized_desc_list)","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:14:32.886123Z","iopub.execute_input":"2022-03-03T20:14:32.88636Z","iopub.status.idle":"2022-03-03T20:17:24.799666Z","shell.execute_reply.started":"2022-03-03T20:14:32.886331Z","shell.execute_reply":"2022-03-03T20:17:24.798834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lemmatized_desc_list[:5]","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:24.801091Z","iopub.execute_input":"2022-03-03T20:17:24.801383Z","iopub.status.idle":"2022-03-03T20:17:24.807383Z","shell.execute_reply.started":"2022-03-03T20:17:24.801343Z","shell.execute_reply":"2022-03-03T20:17:24.806627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extracting only unique cleaned and lemmatized description for fitting it to Document Term Matrix\nprint('Shape of overall articles dataframe: %s' %(str(articles_desc_data.shape[0])))\nprint('Total no of unique article descriptions: %s' %(len(lemmatized_desc_list)))","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:24.808871Z","iopub.execute_input":"2022-03-03T20:17:24.809177Z","iopub.status.idle":"2022-03-03T20:17:24.819495Z","shell.execute_reply.started":"2022-03-03T20:17:24.809112Z","shell.execute_reply":"2022-03-03T20:17:24.818942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating Document Term Matrix\ncv=CountVectorizer(analyzer='word')\ndata=cv.fit_transform(lemmatized_desc_list)\ndf_dtm = pd.DataFrame(data.toarray(), columns=cv.get_feature_names())\ndf_dtm.index = lemmatized_desc_list\ndf_dtm.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:24.820602Z","iopub.execute_input":"2022-03-03T20:17:24.821088Z","iopub.status.idle":"2022-03-03T20:17:26.188319Z","shell.execute_reply.started":"2022-03-03T20:17:24.821056Z","shell.execute_reply":"2022-03-03T20:17:26.187646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_dtm.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:26.189388Z","iopub.execute_input":"2022-03-03T20:17:26.189593Z","iopub.status.idle":"2022-03-03T20:17:26.206592Z","shell.execute_reply.started":"2022-03-03T20:17:26.189569Z","shell.execute_reply":"2022-03-03T20:17:26.205645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function for generating word clouds\ndef generate_wordcloud(data,title):\n#     wc = WordCloud(width=400, height=330, max_words=150,colormap=\"Dark2\").generate_from_frequencies(data)\n    wc = WordCloud(width=400, height=330, max_words=150, background_color ='white', min_font_size=4).generate_from_frequencies(data)\n    plt.figure(figsize=(15,8))\n    plt.imshow(wc, interpolation='bilinear')\n    plt.axis(\"off\")\n    plt.title('\\n'.join(wrap(title,60)),fontsize=13)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:26.207652Z","iopub.execute_input":"2022-03-03T20:17:26.207919Z","iopub.status.idle":"2022-03-03T20:17:26.218302Z","shell.execute_reply.started":"2022-03-03T20:17:26.207891Z","shell.execute_reply":"2022-03-03T20:17:26.217624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transposing document term matrix\ndf_dtm=df_dtm.transpose()\ndf_dtm.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:26.21963Z","iopub.execute_input":"2022-03-03T20:17:26.21987Z","iopub.status.idle":"2022-03-03T20:17:26.253214Z","shell.execute_reply.started":"2022-03-03T20:17:26.219835Z","shell.execute_reply":"2022-03-03T20:17:26.252506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting word cloud for each unique first 10 articles\nfor index,product in enumerate(df_dtm.columns[:10]):\n  generate_wordcloud(df_dtm[product].sort_values(ascending=False),product)","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:26.254416Z","iopub.execute_input":"2022-03-03T20:17:26.254604Z","iopub.status.idle":"2022-03-03T20:17:28.142141Z","shell.execute_reply.started":"2022-03-03T20:17:26.25458Z","shell.execute_reply":"2022-03-03T20:17:28.141553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the world clouds above, we can observer that certain words like naroow, shoulder, belt, pocket, jersey, soft, etc are widely being used under descriptions. This proves that most of the article description that H&M uses describes the physical aspects of the article as well as ","metadata":{}},{"cell_type":"code","source":"# Clearing out local variables for freeing up RAM\ndel contractions_re, articles_desc_data, nlp, cv, data, df_dtm, lemmatized_desc_list\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:28.143058Z","iopub.execute_input":"2022-03-03T20:17:28.143709Z","iopub.status.idle":"2022-03-03T20:17:28.391283Z","shell.execute_reply.started":"2022-03-03T20:17:28.143678Z","shell.execute_reply":"2022-03-03T20:17:28.390355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Exploring Customer Metadata","metadata":{}},{"cell_type":"code","source":"# List all columns present in article metadata\ncustomers_data.columns","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:28.392411Z","iopub.execute_input":"2022-03-03T20:17:28.392707Z","iopub.status.idle":"2022-03-03T20:17:28.403479Z","shell.execute_reply.started":"2022-03-03T20:17:28.392675Z","shell.execute_reply":"2022-03-03T20:17:28.402702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:28.404559Z","iopub.execute_input":"2022-03-03T20:17:28.404941Z","iopub.status.idle":"2022-03-03T20:17:28.421663Z","shell.execute_reply.started":"2022-03-03T20:17:28.404909Z","shell.execute_reply":"2022-03-03T20:17:28.421138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting number of null detailed descriptions for customers\ncustomers_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:28.422715Z","iopub.execute_input":"2022-03-03T20:17:28.423061Z","iopub.status.idle":"2022-03-03T20:17:29.047719Z","shell.execute_reply.started":"2022-03-03T20:17:28.423033Z","shell.execute_reply":"2022-03-03T20:17:29.047043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_style(\"darkgrid\")\nf, ax = plt.subplots(figsize=(15,8))\nax = sns.histplot(data=customers_data, x='age', bins=50, color='blue')\nax.set_xlabel('Distribution of the customers age')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:29.048849Z","iopub.execute_input":"2022-03-03T20:17:29.049054Z","iopub.status.idle":"2022-03-03T20:17:29.628157Z","shell.execute_reply.started":"2022-03-03T20:17:29.049029Z","shell.execute_reply":"2022-03-03T20:17:29.627232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the above bar plot, its clear that the customer category which shops at H&M are from age range of 21 to 25 which is mostly the early 20s and relevant customers are also seen within age range of 44 to 50. The reasons may be:  \n1. Youth styling is better at H&M  \n2. Since, dress styles which cater to older category which are mostly subtle in nature are also widely available.","metadata":{}},{"cell_type":"code","source":"sns.set_style(\"darkgrid\")\nf, ax = plt.subplots(figsize=(10,5))\nax = sns.histplot(data=customers_data, x='club_member_status', color='blue')\nax.set_xlabel('Distribution of club member status')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:29.629735Z","iopub.execute_input":"2022-03-03T20:17:29.630214Z","iopub.status.idle":"2022-03-03T20:17:31.578268Z","shell.execute_reply.started":"2022-03-03T20:17:29.630174Z","shell.execute_reply":"2022-03-03T20:17:31.577511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_data.loc[customers_data.club_member_status.isin(['ACTIVE'])].shape[0]/customers_data.shape[0] * 100","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:31.579701Z","iopub.execute_input":"2022-03-03T20:17:31.580686Z","iopub.status.idle":"2022-03-03T20:17:31.819592Z","shell.execute_reply.started":"2022-03-03T20:17:31.580632Z","shell.execute_reply":"2022-03-03T20:17:31.818636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Almost 92.7 i.e. 93 % of the customers are active members of H&M.","metadata":{}},{"cell_type":"code","source":"sns.set_style(\"darkgrid\")\nf, ax = plt.subplots(figsize=(10,5))\nax = sns.histplot(data=customers_data, x='fashion_news_frequency', color='blue')\nax.set_xlabel('Distribution of fashion news frequency status')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:31.820955Z","iopub.execute_input":"2022-03-03T20:17:31.821891Z","iopub.status.idle":"2022-03-03T20:17:33.764636Z","shell.execute_reply.started":"2022-03-03T20:17:31.821845Z","shell.execute_reply":"2022-03-03T20:17:33.763888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_data.loc[customers_data.fashion_news_frequency.isin(['NONE','None'])].shape[0]/customers_data.shape[0] * 100","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:33.76577Z","iopub.execute_input":"2022-03-03T20:17:33.766076Z","iopub.status.idle":"2022-03-03T20:17:33.981657Z","shell.execute_reply.started":"2022-03-03T20:17:33.766022Z","shell.execute_reply":"2022-03-03T20:17:33.980819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Almost 63.9 i.e. 64 % of the customers receive no communication regrading any latest fashion trends in H&M otherwise they dont agree to receive any notifications for the same","metadata":{}},{"cell_type":"markdown","source":"## Exploring Transaction Metadata","metadata":{}},{"cell_type":"code","source":"# List all columns present in article metadata\ntransactions_data.columns","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:33.98301Z","iopub.execute_input":"2022-03-03T20:17:33.983292Z","iopub.status.idle":"2022-03-03T20:17:33.990146Z","shell.execute_reply.started":"2022-03-03T20:17:33.983256Z","shell.execute_reply":"2022-03-03T20:17:33.989189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:33.997518Z","iopub.execute_input":"2022-03-03T20:17:33.998008Z","iopub.status.idle":"2022-03-03T20:17:34.012051Z","shell.execute_reply.started":"2022-03-03T20:17:33.997965Z","shell.execute_reply":"2022-03-03T20:17:34.011141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.float_format', '{:.4f}'.format)\ntransactions_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:34.013543Z","iopub.execute_input":"2022-03-03T20:17:34.013966Z","iopub.status.idle":"2022-03-03T20:17:41.02934Z","shell.execute_reply.started":"2022-03-03T20:17:34.013928Z","shell.execute_reply":"2022-03-03T20:17:41.028559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_style(\"darkgrid\")\nf, ax = plt.subplots(figsize=(10,5))\nax = sns.boxplot(data=transactions_data, x='price', color='blue')\nax.set_xlabel('Article price outliers')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:41.030368Z","iopub.execute_input":"2022-03-03T20:17:41.030563Z","iopub.status.idle":"2022-03-03T20:17:49.152457Z","shell.execute_reply.started":"2022-03-03T20:17:41.030539Z","shell.execute_reply":"2022-03-03T20:17:49.151664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Finding the best transaction date range with the highest transactions made","metadata":{}},{"cell_type":"code","source":"transactions_data.groupby(['t_dat']).agg({'price':'sum'}).sort_values('price', ascending=False)[:10]","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:49.153701Z","iopub.execute_input":"2022-03-03T20:17:49.153912Z","iopub.status.idle":"2022-03-03T20:17:52.046311Z","shell.execute_reply.started":"2022-03-03T20:17:49.153887Z","shell.execute_reply":"2022-03-03T20:17:52.045357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can clearly observe that the topmost sells happened during the time of season sale, clearance sale or christmas celebration kind of shopping and the trend could be observed across the data of 3 years ","metadata":{}},{"cell_type":"markdown","source":"Top 10 customers based on number of transcation that were made","metadata":{}},{"cell_type":"code","source":"transactions_data.groupby(['customer_id']).count().sort_values(by='price', ascending=False)['price'][:10]","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:17:52.047614Z","iopub.execute_input":"2022-03-03T20:17:52.048596Z","iopub.status.idle":"2022-03-03T20:18:09.190158Z","shell.execute_reply.started":"2022-03-03T20:17:52.048548Z","shell.execute_reply":"2022-03-03T20:18:09.189469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:18:09.191263Z","iopub.execute_input":"2022-03-03T20:18:09.191475Z","iopub.status.idle":"2022-03-03T20:18:09.41664Z","shell.execute_reply.started":"2022-03-03T20:18:09.191451Z","shell.execute_reply":"2022-03-03T20:18:09.415857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Merge transaction details with articles","metadata":{}},{"cell_type":"code","source":"articles_data_extract = articles_data[['article_id', 'prod_name', 'product_type_name', 'product_group_name', 'index_name']]","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:18:09.418206Z","iopub.execute_input":"2022-03-03T20:18:09.418516Z","iopub.status.idle":"2022-03-03T20:18:09.430391Z","shell.execute_reply.started":"2022-03-03T20:18:09.418473Z","shell.execute_reply":"2022-03-03T20:18:09.429815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_data_extract = transactions_data[['customer_id', 'article_id', 'price', 't_dat']].merge(articles_data_extract, on='article_id', how='left')\narticles_data_extract.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:18:09.431697Z","iopub.execute_input":"2022-03-03T20:18:09.432274Z","iopub.status.idle":"2022-03-03T20:18:19.640166Z","shell.execute_reply.started":"2022-03-03T20:18:09.432242Z","shell.execute_reply":"2022-03-03T20:18:19.639226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_data_extract.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:18:19.641344Z","iopub.execute_input":"2022-03-03T20:18:19.641587Z","iopub.status.idle":"2022-03-03T20:18:19.648308Z","shell.execute_reply.started":"2022-03-03T20:18:19.641538Z","shell.execute_reply":"2022-03-03T20:18:19.647471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Optimizing transaction dataframe having article info \narticles_data_extract = reduce_mem_usage(articles_data_extract)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:18:19.64946Z","iopub.execute_input":"2022-03-03T20:18:19.649678Z","iopub.status.idle":"2022-03-03T20:18:20.770402Z","shell.execute_reply.started":"2022-03-03T20:18:19.649652Z","shell.execute_reply":"2022-03-03T20:18:20.769558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sns.set_style(\"darkgrid\")\n# f, ax = plt.subplots(figsize=(20,12))\n# ax = sns.boxplot(data=articles_data_extract[['price','product_group_name']], x='price', y='product_group_name')\n# ax.set_xlabel('Product group level pricing outliers', fontsize=15)\n# ax.set_ylabel('Index names for product groups', fontsize=15)\n# ax.xaxis.set_tick_params(labelsize=15)\n# ax.yaxis.set_tick_params(labelsize=15)\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:18:20.771537Z","iopub.execute_input":"2022-03-03T20:18:20.771745Z","iopub.status.idle":"2022-03-03T20:18:20.775468Z","shell.execute_reply.started":"2022-03-03T20:18:20.77172Z","shell.execute_reply":"2022-03-03T20:18:20.774499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In the above boxplot, we see outliers for group name prices. Lower/Upper/Full body have a huge price variance. I guess it could be like some unique collections, relative to casual ones. Some high price articles even belong to accessories group.","metadata":{}},{"cell_type":"code","source":"sns.set_style(\"darkgrid\")\nf, ax = plt.subplots(figsize=(20,12))\n_ = articles_data_extract[articles_data_extract['product_group_name'] == 'Accessories']\nax = sns.boxplot(data=_, x='price', y='product_type_name')\nax.set_xlabel('Accessories group level pricing outliers', fontsize=15)\nax.set_ylabel('Index names for accessory item', fontsize=15)\nax.xaxis.set_tick_params(labelsize=15)\nax.yaxis.set_tick_params(labelsize=15)\ndel _\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:18:20.776406Z","iopub.execute_input":"2022-03-03T20:18:20.776581Z","iopub.status.idle":"2022-03-03T20:18:49.957599Z","shell.execute_reply.started":"2022-03-03T20:18:20.776558Z","shell.execute_reply":"2022-03-03T20:18:49.956833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets look at boxplot prices according to accessories product group and find the reasons of high prices inside this specific group. We can easily observe that the largest outliers can be found among bags, which is logical enough. In addition, scarves and other accessories have articles with prices highly contrasting to the rest of garments.","metadata":{}},{"cell_type":"code","source":"articles_index = articles_data_extract[['product_group_name', 'price']].groupby('product_group_name').mean()\nsns.set_style(\"darkgrid\")\nf, ax = plt.subplots(figsize=(15,8))\nax = sns.barplot(x=articles_index.price, y=articles_index.index, color='blue', alpha=0.8)\nax.set_xlabel('Price by product group')\nax.set_ylabel('Product group')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:18:49.959008Z","iopub.execute_input":"2022-03-03T20:18:49.959217Z","iopub.status.idle":"2022-03-03T20:18:54.71433Z","shell.execute_reply.started":"2022-03-03T20:18:49.95919Z","shell.execute_reply":"2022-03-03T20:18:54.713551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the above bar plot it is clearly visible that we are seeing the means are seen in below mentioned product group names. Top 10 product group names are:  \nShoes  \nGarment Full body  \nBags  \nGarment Lower body  \nUnderwear/nightwear  \nSwimwear  \nGarment and Shoe Care  \nInterior textile  \nAccessories  \nSocks & Tights","metadata":{}},{"cell_type":"code","source":"articles_data_extract['t_dat'] = pd.to_datetime(articles_data_extract['t_dat'])\n\nproduct_list = ['Shoes', 'Garment Full body', 'Bags', 'Garment Lower body', 'Underwear/nightwear','Swimwear','Garment and Shoe care','Interior textile','Accessories','Socks & Tights']\ncolors = ['cadetblue', 'orange', 'mediumspringgreen', 'tomato', 'lightseagreen', 'cadetblue', 'orange', 'mediumspringgreen', 'tomato', 'lightseagreen']\nk = 0\nf, ax = plt.subplots(5, 2, figsize=(20, 30))\nfor i in range(5):\n    for j in range(2):\n        try:\n            product = product_list[k]\n            articles_for_merge_product = articles_data_extract[articles_data_extract.product_group_name == product_list[k]]\n            series_mean = articles_for_merge_product[['t_dat', 'price']].groupby(pd.Grouper(key=\"t_dat\", freq='M')).mean().fillna(0)\n            series_std = articles_for_merge_product[['t_dat', 'price']].groupby(pd.Grouper(key=\"t_dat\", freq='M')).std().fillna(0)\n            ax[i, j].plot(series_mean, linewidth=4, color=colors[k])\n            ax[i, j].fill_between(series_mean.index, (series_mean.values-2*series_std.values).ravel(), \n                             (series_mean.values+2*series_std.values).ravel(), color=colors[k], alpha=.1)\n            ax[i, j].set_title(f'Mean {product_list[k]} price in time')\n            ax[i, j].set_xlabel('month')\n            ax[i, j].set_xlabel(f'{product_list[k]}')\n            k += 1\n        except IndexError:\n            ax[i, j].set_visible(False)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:24:39.890785Z","iopub.execute_input":"2022-03-03T20:24:39.891147Z","iopub.status.idle":"2022-03-03T20:25:33.048674Z","shell.execute_reply.started":"2022-03-03T20:24:39.891103Z","shell.execute_reply":"2022-03-03T20:25:33.048092Z"},"trusted":true},"execution_count":null,"outputs":[]}]}