{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**What we'll be doing here:**\nThis is a heuristics based notebook. As we have already found out, popularity and repetition is king in this competition. We'll combine these two to create a good enough baseline.\n\n1. Recommend most bought items from last 4 weeks.\n1. Recommend popular items from last 2 weeks weighted down by time.","metadata":{}},{"cell_type":"markdown","source":"# Data Extraction","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom matplotlib.pyplot import figure\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-01T17:51:58.417332Z","iopub.execute_input":"2022-06-01T17:51:58.417660Z","iopub.status.idle":"2022-06-01T17:51:59.293383Z","shell.execute_reply.started":"2022-06-01T17:51:58.417586Z","shell.execute_reply":"2022-06-01T17:51:59.292679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"artc=pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')\nartc.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"artc.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"artc.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"artc.nunique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cust=pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/customers.csv')\ncust.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cust.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cust.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cust.nunique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def missing_percent_of_column(train_set):\n    nan_percent = 100*(train_set.isnull().sum()/len(train_set))\n    nan_percent = nan_percent[nan_percent>0].sort_values(ascending=False).round(1)\n    DataFrame = pd.DataFrame(nan_percent)\n    # Rename the columns\n    mis_percent_table = DataFrame.rename(columns = {0 : '% of Misiing Values'}) \n    # Sort the table by percentage of missing descending\n    mis_percent = mis_percent_table\n    return mis_percent","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_percent_of_column(train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_percent_of_column(cust)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_percent_of_column(artc)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"article_id\t\nproduct_code\t\nprod_name\t\nproduct_type_no\t\nproduct_type_name\t\nproduct_group_name\t\ngraphical_appearance_no\t\ngraphical_appearance_name\t\ncolour_group_code\t\ncolour_group_name\t...\t\ndepartment_name\t\nindex_code\t\nindex_name\t\nindex_group_no\t\nindex_group_name\t\nsection_no\t\nsection_name\t\ngarment_group_no\t\ngarment_group_name\t\ndetail_desc","metadata":{}},{"cell_type":"code","source":"artc.head(3)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* customer_id\n* FN\n* Active\n* club_member_status\n* fashion_news_frequency\n* age\n* postal_code","metadata":{}},{"cell_type":"markdown","source":"# Data Analysis","metadata":{}},{"cell_type":"markdown","source":"**Exploratory data analysis**\n\nLets first visualise data with the help of different graphs, this would help us to understand data the data.","metadata":{}},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(10,5))\nsns.histplot(x=\"age\", data=cust, bins=50, kde=True)\nax.set_xlabel('Distribution of the customers age')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(7,5))\nsns.countplot(data=cust, x=\"club_member_status\")\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(7,5))\nsns.countplot(data=cust, x=\"fashion_news_frequency\")\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(10,5))\nax = sns.boxplot(data=train, x='price')\nax.set_xlabel('Price outliers')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(15,5))\npgn = artc.product_group_name.value_counts().rename_axis('product_group_name').reset_index(name='counts')\nsns.barplot(data=pgn, x=\"counts\", y = 'product_group_name',palette=\"Blues_d\")\nax.set_xlabel('Count of article by product_group_name')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(10,5))\nsns.countplot(data=artc, y=\"index_name\",dodge=True)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(figsize=(10,6))\nsns.histplot(data=artc, y=\"index_group_name\",hue='index_name', multiple=\"stack\")\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data Cleaning**","metadata":{}},{"cell_type":"markdown","source":"**Feaure Engineering**","metadata":{}},{"cell_type":"code","source":"train.head(4)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"If you find this notebook useful or interesting, please, support with an upvote :)","metadata":{}}]}