{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Splitting the dataframes","metadata":{}},{"cell_type":"code","source":"transaction = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\ncustomer = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/customers.csv')\narticles = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')\n","metadata":{"execution":{"iopub.status.busy":"2022-03-10T12:15:40.014131Z","iopub.execute_input":"2022-03-10T12:15:40.014884Z","iopub.status.idle":"2022-03-10T12:16:15.875514Z","shell.execute_reply.started":"2022-03-10T12:15:40.014844Z","shell.execute_reply":"2022-03-10T12:16:15.874628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Checking the number of rows \ntransaction.loc[transaction['customer_id']=='00000dbacae5abe5e23885899a1fa44253a17956c6d1c3d25f88aa139fdfc657'].shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Join transaction and customer by customer id and get age and postal code","metadata":{}},{"cell_type":"code","source":"customer_detail_by_trans = pd.merge(transaction,customer,on='customer_id')","metadata":{"execution":{"iopub.status.busy":"2022-03-10T12:18:58.727402Z","iopub.execute_input":"2022-03-10T12:18:58.727919Z","iopub.status.idle":"2022-03-10T12:19:17.052823Z","shell.execute_reply.started":"2022-03-10T12:18:58.727877Z","shell.execute_reply":"2022-03-10T12:19:17.051793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let us now merge the table with article table","metadata":{}},{"cell_type":"code","source":"customer_detail_by_trans_by_article = pd.merge(customer_detail_by_trans,articles[['article_id','prod_name','product_type_name','product_group_name','graphical_appearance_name','colour_group_name','perceived_colour_value_name','perceived_colour_master_name','department_name','index_name','index_group_name','section_name','garment_group_name','detail_desc']],on='article_id')","metadata":{"execution":{"iopub.status.busy":"2022-03-10T11:10:44.527693Z","iopub.execute_input":"2022-03-10T11:10:44.528184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Table below has no identifiers but more of names for features","metadata":{}},{"cell_type":"code","source":"customer_detail_by_trans_by_article.head(5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generate word cloud on a dataframe column \"detail_desc\"","metadata":{}},{"cell_type":"code","source":"#Final word cloud after all the cleaning and pre-processing\nimport matplotlib.pyplot as plt\nfrom wordcloud import WordCloud, STOPWORDS\ncomment_words = ' '\nstopwords = set(STOPWORDS) ","metadata":{"execution":{"iopub.status.busy":"2022-03-10T11:17:58.398944Z","iopub.execute_input":"2022-03-10T11:17:58.399645Z","iopub.status.idle":"2022-03-10T11:17:58.444943Z","shell.execute_reply.started":"2022-03-10T11:17:58.399602Z","shell.execute_reply":"2022-03-10T11:17:58.444304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# iterate through the csv file \nfor val in articles['detail_desc']: \n\n   # typecaste each val to string \n   val = str(val) \n\n   # split the value \n   tokens = val.split() \n\n# Converts each token into lowercase \nfor i in range(len(tokens)): \n    tokens[i] = tokens[i].lower() \n\nfor words in tokens: \n    comment_words = comment_words + words + ' '\n\n\nwordcloud = WordCloud(width = 800, height = 800, \n            background_color ='white', \n            stopwords = stopwords, \n            min_font_size = 10).generate(comment_words) \n\n# plot the WordCloud image                        \nplt.figure(figsize = (8, 8), facecolor = None) \nplt.imshow(wordcloud) \nplt.axis(\"off\") \nplt.tight_layout(pad = 0) \n\nplt.show() ","metadata":{"execution":{"iopub.status.busy":"2022-03-10T11:18:00.134970Z","iopub.execute_input":"2022-03-10T11:18:00.135776Z","iopub.status.idle":"2022-03-10T11:18:01.454062Z","shell.execute_reply.started":"2022-03-10T11:18:00.135733Z","shell.execute_reply":"2022-03-10T11:18:01.453362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We could see from the above cell most people prefer soft and cotton items","metadata":{}},{"cell_type":"code","source":"customer['age'].hist(bins=10)","metadata":{"execution":{"iopub.status.busy":"2022-03-10T11:18:35.108090Z","iopub.execute_input":"2022-03-10T11:18:35.108391Z","iopub.status.idle":"2022-03-10T11:18:35.357842Z","shell.execute_reply.started":"2022-03-10T11:18:35.108356Z","shell.execute_reply":"2022-03-10T11:18:35.357164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let us first define what is expensive","metadata":{}},{"cell_type":"code","source":"customer_prod_price = pd.merge(articles[['article_id','product_group_name']],transaction[['article_id','price']],on='article_id').drop_duplicates()\ncustomer_prod_price = customer_prod_price[['product_group_name','price']]\ncustomer_prod_price.plot.scatter(x = 'product_group_name', y = 'price', s = 'price',figsize=(30, 15), c = 'red');","metadata":{"execution":{"iopub.status.busy":"2022-03-10T12:24:59.216521Z","iopub.execute_input":"2022-03-10T12:24:59.216860Z","iopub.status.idle":"2022-03-10T12:26:18.752935Z","shell.execute_reply.started":"2022-03-10T12:24:59.216824Z","shell.execute_reply":"2022-03-10T12:26:18.752202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"customer_prod_price['product_group_name'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-03-10T11:22:46.184244Z","iopub.execute_input":"2022-03-10T11:22:46.185066Z","iopub.status.idle":"2022-03-10T11:22:46.441073Z","shell.execute_reply.started":"2022-03-10T11:22:46.185024Z","shell.execute_reply":"2022-03-10T11:22:46.440267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# How fashion news influences the buying behaviour ","metadata":{}},{"cell_type":"code","source":"fashion_news_group = pd.merge(customer[['customer_id','fashion_news_frequency']],transaction[['customer_id','price']],on='customer_id').drop_duplicates()\nfashion_news_group = fashion_news_group[['fashion_news_frequency','price']]\nfashion_news_group['fashion_news_frequency'] = fashion_news_group['fashion_news_frequency'].replace(np.nan, 'None')\nfashion_news_group['fashion_news_frequency'] = fashion_news_group['fashion_news_frequency'].replace('NONE', 'None')\nfashion_news_group.plot.scatter(x = 'fashion_news_frequency', y = 'price', s = 'price',figsize=(15, 15), c = 'red');","metadata":{"execution":{"iopub.status.busy":"2022-03-10T13:01:29.376906Z","iopub.execute_input":"2022-03-10T13:01:29.377374Z","iopub.status.idle":"2022-03-10T13:06:28.215749Z","shell.execute_reply.started":"2022-03-10T13:01:29.377330Z","shell.execute_reply":"2022-03-10T13:06:28.214810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n# Plotting a bar graph of the number of stores in each city, for the first ten cities listed\n# in the column 'City'\nsales_channel  = transaction['sales_channel_id'].value_counts()\nsales_channel_count = sales_channel[:3,]\nplt.figure(figsize=(10,5))\nsns.barplot(sales_channel_count.index, sales_channel_count.values, alpha=0.8)\nplt.title('Sales Channel split')\nplt.ylabel('Number of Occurrences', fontsize=12)\nplt.xlabel('Channel id', fontsize=12)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-03-10T13:18:37.467488Z","iopub.execute_input":"2022-03-10T13:18:37.468313Z","iopub.status.idle":"2022-03-10T13:18:38.549667Z","shell.execute_reply.started":"2022-03-10T13:18:37.468242Z","shell.execute_reply":"2022-03-10T13:18:38.548942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}}]}