{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-09T19:39:03.841771Z","iopub.execute_input":"2022-05-09T19:39:03.842552Z","iopub.status.idle":"2022-05-09T19:40:24.208635Z","shell.execute_reply.started":"2022-05-09T19:39:03.842423Z","shell.execute_reply":"2022-05-09T19:40:24.203839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Calson, Keaoleboga,Rethabile,Princess Presentation","metadata":{}},{"cell_type":"code","source":"# Libraries that we will use\n\nimport numpy as np\nimport pandas as pd\nimport os\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom wordcloud import WordCloud, STOPWORDS\nfrom datetime import datetime\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:06:15.816206Z","iopub.execute_input":"2022-05-09T21:06:15.816741Z","iopub.status.idle":"2022-05-09T21:06:15.821848Z","shell.execute_reply.started":"2022-05-09T21:06:15.816695Z","shell.execute_reply":"2022-05-09T21:06:15.821108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we want to read the competition data as well as have a glimpse of it.","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"os.listdir(\"/kaggle/input\")","metadata":{"execution":{"iopub.status.busy":"2022-05-09T20:31:01.901245Z","iopub.execute_input":"2022-05-09T20:31:01.902203Z","iopub.status.idle":"2022-05-09T20:31:01.909683Z","shell.execute_reply.started":"2022-05-09T20:31:01.902128Z","shell.execute_reply":"2022-05-09T20:31:01.908395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"files and folders: {os.listdir('/kaggle/input/h-and-m-personalized-fashion-recommendations/')}\")\nprint(\"Subfolders in images folder: \", len(list(os.listdir(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/images\"))))","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:40:40.348734Z","iopub.execute_input":"2022-05-09T21:40:40.350039Z","iopub.status.idle":"2022-05-09T21:40:40.375261Z","shell.execute_reply.started":"2022-05-09T21:40:40.349986Z","shell.execute_reply":"2022-05-09T21:40:40.374614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_folders = total_files = 0\nfolder_info = []\nimages_names = []\nfor base, dirs, files in tqdm(os.walk('/kaggle/input/h-and-m-personalized-fashion-recommendations/')):\n    for directories in dirs:\n        folder_info.append((directories, len(os.listdir(os.path.join(base, directories)))))\n        total_folders += 1\n    for _files in files:\n        total_files += 1\n        if len(_files.split(\".jpg\"))==2:\n            images_names.append(_files.split(\".jpg\")[0])","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:41:22.045862Z","iopub.execute_input":"2022-05-09T21:41:22.046229Z","iopub.status.idle":"2022-05-09T21:42:30.567882Z","shell.execute_reply.started":"2022-05-09T21:41:22.046193Z","shell.execute_reply":"2022-05-09T21:42:30.566978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df= pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\ncustomers_df = pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv\")\nsample_submission_df = pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\")\ntransactions_train_df = pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2022-05-09T20:31:11.730892Z","iopub.execute_input":"2022-05-09T20:31:11.731173Z","iopub.status.idle":"2022-05-09T20:32:21.652488Z","shell.execute_reply.started":"2022-05-09T20:31:11.731144Z","shell.execute_reply":"2022-05-09T20:32:21.651564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T20:32:21.654434Z","iopub.execute_input":"2022-05-09T20:32:21.654687Z","iopub.status.idle":"2022-05-09T20:32:21.68258Z","shell.execute_reply.started":"2022-05-09T20:32:21.654657Z","shell.execute_reply":"2022-05-09T20:32:21.681733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T20:32:21.689187Z","iopub.execute_input":"2022-05-09T20:32:21.689489Z","iopub.status.idle":"2022-05-09T20:32:21.704547Z","shell.execute_reply.started":"2022-05-09T20:32:21.689455Z","shell.execute_reply":"2022-05-09T20:32:21.703604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T20:32:21.706488Z","iopub.execute_input":"2022-05-09T20:32:21.707144Z","iopub.status.idle":"2022-05-09T20:32:21.723355Z","shell.execute_reply.started":"2022-05-09T20:32:21.707097Z","shell.execute_reply":"2022-05-09T20:32:21.722725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Oserve that the customers ID does not provide us with clear information as it is a binary number.","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"transactions_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T20:32:21.724473Z","iopub.execute_input":"2022-05-09T20:32:21.72472Z","iopub.status.idle":"2022-05-09T20:32:21.744166Z","shell.execute_reply.started":"2022-05-09T20:32:21.724691Z","shell.execute_reply":"2022-05-09T20:32:21.743227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We begin with some univariative analysis of the data tha twe have loaded\nThere are 3 main tables:\n\narticles - contains informations about each article (like product code, name, product group code, name ...)\ncustomers - contains informations about each customer (fidelity card membership, age, postal code)\ntransactions (train)\nTransactions have customer_id and article_id, which are foreign keys for the customer and articles tables. Beside this, transaction also contains sales_channel_id.\n\nTransaction train data has entries for the date of the transaction, the customer id, the article id, a price (per transaction) and a sales channel id.","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"Now we will contibue with creating functions to help us munipulate our data.\n","metadata":{}},{"cell_type":"code","source":"# This is a function to find missing data from the data that we have loaded.\n\ndef missing_data(data):\n    total = data.isnull().sum().sort_values(ascending = False)\n    percent = (data.isnull().sum()/data.isnull().count()*100).sort_values(ascending = False)\n    return pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])\n\n","metadata":{"execution":{"iopub.status.busy":"2022-05-09T20:42:34.490071Z","iopub.execute_input":"2022-05-09T20:42:34.491301Z","iopub.status.idle":"2022-05-09T20:42:34.500815Z","shell.execute_reply.started":"2022-05-09T20:42:34.491211Z","shell.execute_reply":"2022-05-09T20:42:34.499479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This function is to find any missing values in the data.\n\ndef unique_values(data):\n    total = data.count()\n    tt = pd.DataFrame(total)\n    tt.columns = ['Total']\n    uniques = []\n    for col in data.columns:\n        unique = data[col].nunique()\n        uniques.append(unique)\n    tt['Uniques'] = uniques\n    return tt","metadata":{"execution":{"iopub.status.busy":"2022-05-09T20:43:15.300044Z","iopub.execute_input":"2022-05-09T20:43:15.301178Z","iopub.status.idle":"2022-05-09T20:43:15.308729Z","shell.execute_reply.started":"2022-05-09T20:43:15.301123Z","shell.execute_reply":"2022-05-09T20:43:15.307643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T20:43:59.498239Z","iopub.execute_input":"2022-05-09T20:43:59.498629Z","iopub.status.idle":"2022-05-09T20:43:59.712414Z","shell.execute_reply.started":"2022-05-09T20:43:59.49855Z","shell.execute_reply":"2022-05-09T20:43:59.711355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(articles_df)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T20:45:11.617307Z","iopub.execute_input":"2022-05-09T20:45:11.618466Z","iopub.status.idle":"2022-05-09T20:45:12.137323Z","shell.execute_reply.started":"2022-05-09T20:45:11.618396Z","shell.execute_reply":"2022-05-09T20:45:12.135938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"In the article data, the only missing data is for the detailed description of the article (0.4% missing data).\n\n","metadata":{}},{"cell_type":"markdown","source":"\n\n\n","metadata":{}},{"cell_type":"code","source":"customers_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T20:49:06.654621Z","iopub.execute_input":"2022-05-09T20:49:06.654995Z","iopub.status.idle":"2022-05-09T20:49:07.321632Z","shell.execute_reply.started":"2022-05-09T20:49:06.654953Z","shell.execute_reply":"2022-05-09T20:49:07.320605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(customers_df)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T20:49:30.922066Z","iopub.execute_input":"2022-05-09T20:49:30.922415Z","iopub.status.idle":"2022-05-09T20:49:32.827399Z","shell.execute_reply.started":"2022-05-09T20:49:30.922382Z","shell.execute_reply":"2022-05-09T20:49:32.826251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Only customer id and postal code are completely filled. Age, fashion news frequency have arounfd 1% misssing data, FN has 65% missing and Active has 66% missing data.\n\n","metadata":{}},{"cell_type":"code","source":"sample_submission_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T20:51:10.743293Z","iopub.execute_input":"2022-05-09T20:51:10.743658Z","iopub.status.idle":"2022-05-09T20:51:11.072227Z","shell.execute_reply.started":"2022-05-09T20:51:10.74362Z","shell.execute_reply":"2022-05-09T20:51:11.071106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T20:53:16.163595Z","iopub.execute_input":"2022-05-09T20:53:16.163923Z","iopub.status.idle":"2022-05-09T20:53:16.176936Z","shell.execute_reply.started":"2022-05-09T20:53:16.163892Z","shell.execute_reply":"2022-05-09T20:53:16.175619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(transactions_train_df)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T20:53:43.853493Z","iopub.execute_input":"2022-05-09T20:53:43.853829Z","iopub.status.idle":"2022-05-09T20:54:05.215786Z","shell.execute_reply.started":"2022-05-09T20:53:43.853798Z","shell.execute_reply":"2022-05-09T20:54:05.214763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"No missing data from transactions train data source.","metadata":{}},{"cell_type":"code","source":"unique_values(articles_df)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T20:55:20.707852Z","iopub.execute_input":"2022-05-09T20:55:20.708165Z","iopub.status.idle":"2022-05-09T20:55:21.072236Z","shell.execute_reply.started":"2022-05-09T20:55:20.708135Z","shell.execute_reply":"2022-05-09T20:55:21.071337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We observe that features for which we expect to have the same number of unique value, like:\n\n1.product_type_no and product_type_name,\n2.departmant_no and department_name,\n3.section_no and section_name have different number of unique values, which might means that we might   have categories with same name. Others, like:\n  index_code and index_name,\n4.garment_group_no and garment_group_name have the same number of unique values.\n","metadata":{}},{"cell_type":"code","source":"unique_values(customers_df)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T20:58:06.150994Z","iopub.execute_input":"2022-05-09T20:58:06.152247Z","iopub.status.idle":"2022-05-09T20:58:08.548069Z","shell.execute_reply.started":"2022-05-09T20:58:06.152181Z","shell.execute_reply":"2022-05-09T20:58:08.54734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_values(transactions_train_df)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T20:58:47.680484Z","iopub.execute_input":"2022-05-09T20:58:47.681387Z","iopub.status.idle":"2022-05-09T20:59:08.508438Z","shell.execute_reply.started":"2022-05-09T20:58:47.681335Z","shell.execute_reply":"2022-05-09T20:59:08.507542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We observe that not all the customers in customer data appears as having transactions in transaction train data. As well, not all articles are represented in this data. It is interesting that the number of different prices is quite small, out of 31.7M transactions, and for 1.3M customers, buying 104K different articles. Same for the dates, there are only 734 different dates. Let's check some stats here.","metadata":{}},{"cell_type":"code","source":"print(f\"Percent of articles present in customer data: {round(104547/105542,3)*100}%\")\nprint(f\"Percent of articles present in transactions data: {round(1362281/1371980,3)*100}%\")","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:02:24.896691Z","iopub.execute_input":"2022-05-09T21:02:24.897562Z","iopub.status.idle":"2022-05-09T21:02:24.903587Z","shell.execute_reply.started":"2022-05-09T21:02:24.897498Z","shell.execute_reply":"2022-05-09T21:02:24.902886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualizations","metadata":{}},{"cell_type":"markdown","source":"## Articles Data","metadata":{}},{"cell_type":"code","source":"temp = articles_df.groupby([\"product_group_name\"])[\"product_type_name\"].nunique()\ndf = pd.DataFrame({'Product Group': temp.index,\n                   'Product Types': temp.values\n                  })\ndf = df.sort_values(['Product Types'], ascending=False)\nplt.figure(figsize = (8,6))\nplt.title('Number of Product Types per each Product Group')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Product Group', y=\"Product Types\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:06:22.139749Z","iopub.execute_input":"2022-05-09T21:06:22.140097Z","iopub.status.idle":"2022-05-09T21:06:22.624215Z","shell.execute_reply.started":"2022-05-09T21:06:22.140054Z","shell.execute_reply":"2022-05-09T21:06:22.622957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"stopwords = set(STOPWORDS)\n\ndef show_wordcloud(data, title = None):\n    wordcloud = WordCloud(\n        background_color='white',\n        stopwords=stopwords,\n        max_words=200,\n        max_font_size=40, \n        scale=5,\n        random_state=1\n    ).generate(str(data))\n\n    fig = plt.figure(1, figsize=(10,10))\n    plt.axis('off')\n    if title: \n        fig.suptitle(title, fontsize=14)\n        fig.subplots_adjust(top=2.3)\n\n    plt.imshow(wordcloud)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:12:26.961875Z","iopub.execute_input":"2022-05-09T21:12:26.962236Z","iopub.status.idle":"2022-05-09T21:12:26.970672Z","shell.execute_reply.started":"2022-05-09T21:12:26.962197Z","shell.execute_reply":"2022-05-09T21:12:26.969658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_wordcloud(articles_df[\"prod_name\"], \"Wordcloud from product name\")","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:12:43.432313Z","iopub.execute_input":"2022-05-09T21:12:43.432717Z","iopub.status.idle":"2022-05-09T21:12:44.092751Z","shell.execute_reply.started":"2022-05-09T21:12:43.432669Z","shell.execute_reply":"2022-05-09T21:12:44.091624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"product_group_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Product Group': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (8,6))\nplt.title('Number of Articles per each Product Group')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Product Group', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:14:38.44767Z","iopub.execute_input":"2022-05-09T21:14:38.44826Z","iopub.status.idle":"2022-05-09T21:14:38.776845Z","shell.execute_reply.started":"2022-05-09T21:14:38.448223Z","shell.execute_reply":"2022-05-09T21:14:38.775992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"product_type_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Product Type': temp.index,\n                   'Articles': temp.values\n                  })\ntotal_types = len(df['Product Type'].unique())\ndf = df.sort_values(['Articles'], ascending=False)[0:50]\nplt.figure(figsize = (16,6))\nplt.title(f'Number of Articles per each Product Type (top 50 from total: {total_types})')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Product Type', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:15:31.517002Z","iopub.execute_input":"2022-05-09T21:15:31.517363Z","iopub.status.idle":"2022-05-09T21:15:32.362286Z","shell.execute_reply.started":"2022-05-09T21:15:31.517323Z","shell.execute_reply":"2022-05-09T21:15:32.361365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"department_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Department Name': temp.index,\n                   'Articles': temp.values\n                  })\ntotal_depts = len(df['Department Name'].unique())\ndf = df.sort_values(['Articles'], ascending=False).head(50)\nplt.figure(figsize = (16,6))\nplt.title(f'Number of Articles per each Department (top 50 from total: {total_depts})')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Department Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:16:17.303818Z","iopub.execute_input":"2022-05-09T21:16:17.304164Z","iopub.status.idle":"2022-05-09T21:16:18.218041Z","shell.execute_reply.started":"2022-05-09T21:16:17.304125Z","shell.execute_reply":"2022-05-09T21:16:18.216669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"graphical_appearance_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Graphical Appearance Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False).head(50)\nplt.figure(figsize = (16,6))\nplt.title(f'Number of Articles per each Graphical Appearance Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Graphical Appearance Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:18:00.786206Z","iopub.execute_input":"2022-05-09T21:18:00.786583Z","iopub.status.idle":"2022-05-09T21:18:01.837093Z","shell.execute_reply.started":"2022-05-09T21:18:00.786548Z","shell.execute_reply":"2022-05-09T21:18:01.835865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"index_group_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Index Group Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (6,6))\nplt.title(f'Number of Articles per each Index Group Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Index Group Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:18:51.540338Z","iopub.execute_input":"2022-05-09T21:18:51.540997Z","iopub.status.idle":"2022-05-09T21:18:51.742453Z","shell.execute_reply.started":"2022-05-09T21:18:51.540933Z","shell.execute_reply":"2022-05-09T21:18:51.741715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"colour_group_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Colour Group Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (12,6))\nplt.title(f'Number of Articles per each Colour Group Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Colour Group Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:19:50.450089Z","iopub.execute_input":"2022-05-09T21:19:50.450432Z","iopub.status.idle":"2022-05-09T21:19:51.301941Z","shell.execute_reply.started":"2022-05-09T21:19:50.450399Z","shell.execute_reply":"2022-05-09T21:19:51.30073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"perceived_colour_master_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Perceived Colour Master Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (12,6))\nplt.title(f'Number of Articles per each Perceived Colour Master Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Perceived Colour Master Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:21:42.789961Z","iopub.execute_input":"2022-05-09T21:21:42.790301Z","iopub.status.idle":"2022-05-09T21:21:43.088584Z","shell.execute_reply.started":"2022-05-09T21:21:42.790266Z","shell.execute_reply":"2022-05-09T21:21:43.087492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"perceived_colour_value_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Perceived Colour Group Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (6,6))\nplt.title(f'Number of Articles per each Perceived Colour Group Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Perceived Colour Group Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:20:37.089964Z","iopub.execute_input":"2022-05-09T21:20:37.090368Z","iopub.status.idle":"2022-05-09T21:20:37.309285Z","shell.execute_reply.started":"2022-05-09T21:20:37.090325Z","shell.execute_reply":"2022-05-09T21:20:37.307986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"index_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Index Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (8,6))\nplt.title(f'Number of Articles per each Index Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Index Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:23:15.601818Z","iopub.execute_input":"2022-05-09T21:23:15.603055Z","iopub.status.idle":"2022-05-09T21:23:15.88822Z","shell.execute_reply.started":"2022-05-09T21:23:15.602979Z","shell.execute_reply":"2022-05-09T21:23:15.887505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"garment_group_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Garment Group Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (12,6))\nplt.title(f'Number of Articles per each Garment Group Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Garment Group Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:24:06.36277Z","iopub.execute_input":"2022-05-09T21:24:06.363326Z","iopub.status.idle":"2022-05-09T21:24:06.730192Z","shell.execute_reply.started":"2022-05-09T21:24:06.363289Z","shell.execute_reply":"2022-05-09T21:24:06.729307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"section_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Section Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (16,6))\nplt.title(f'Number of Articles per each Section Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Section Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:24:52.39017Z","iopub.execute_input":"2022-05-09T21:24:52.390788Z","iopub.status.idle":"2022-05-09T21:24:53.497357Z","shell.execute_reply.started":"2022-05-09T21:24:52.390741Z","shell.execute_reply":"2022-05-09T21:24:53.4963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"section_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Section Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (16,6))\nplt.title(f'Number of Articles per each Section Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Section Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:25:30.273705Z","iopub.execute_input":"2022-05-09T21:25:30.274062Z","iopub.status.idle":"2022-05-09T21:25:31.371573Z","shell.execute_reply.started":"2022-05-09T21:25:30.274026Z","shell.execute_reply":"2022-05-09T21:25:31.370757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Customers data","metadata":{}},{"cell_type":"code","source":"temp = customers_df.groupby([\"age\"])[\"customer_id\"].count()\ndf = pd.DataFrame({'Age': temp.index,\n                   'Customers': temp.values\n                  })\ndf = df.sort_values(['Age'], ascending=False)\nplt.figure(figsize = (16,6))\nplt.title(f'Number of Customers per each Age')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Age', y=\"Customers\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:27:35.821161Z","iopub.execute_input":"2022-05-09T21:27:35.821513Z","iopub.status.idle":"2022-05-09T21:27:37.186601Z","shell.execute_reply.started":"2022-05-09T21:27:35.821478Z","shell.execute_reply":"2022-05-09T21:27:37.185739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = customers_df.groupby([\"fashion_news_frequency\"])[\"customer_id\"].count()\ndf = pd.DataFrame({'Fashion News Frequency': temp.index,\n                   'Customers': temp.values\n                  })\ndf = df.sort_values(['Customers'], ascending=False)\nplt.figure(figsize = (6,6))\nplt.title(f'Number of Customers per each Fashion News Frequency')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Fashion News Frequency', y=\"Customers\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:28:58.068277Z","iopub.execute_input":"2022-05-09T21:28:58.068991Z","iopub.status.idle":"2022-05-09T21:28:58.628705Z","shell.execute_reply.started":"2022-05-09T21:28:58.068914Z","shell.execute_reply":"2022-05-09T21:28:58.627437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = customers_df.groupby([\"club_member_status\"])[\"customer_id\"].count()\ndf = pd.DataFrame({'Club Member Status': temp.index,\n                   'Customers': temp.values\n                  })\ndf = df.sort_values(['Customers'], ascending=False)\nplt.figure(figsize = (6,6))\nplt.title(f'Number of Customers per each Club Member Status')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Club Member Status', y=\"Customers\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:29:24.497232Z","iopub.execute_input":"2022-05-09T21:29:24.498192Z","iopub.status.idle":"2022-05-09T21:29:25.04499Z","shell.execute_reply.started":"2022-05-09T21:29:24.498151Z","shell.execute_reply":"2022-05-09T21:29:25.043737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Transactions data","metadata":{}},{"cell_type":"code","source":"df = transactions_train_df.sample(100_000)\nfig, ax = plt.subplots(1, 1, figsize=(14, 7))\nsns.kdeplot(np.log(df.loc[df[\"sales_channel_id\"]==1].price.value_counts()))\nsns.kdeplot(np.log(df.loc[df[\"sales_channel_id\"]==2].price.value_counts()))\nax.legend(labels=['Sales channel 1', 'Sales channel 1'])\nplt.title(\"Logaritmic distribution of price frequency in transactions, grouped per sales channel (100k sample)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:30:51.477507Z","iopub.execute_input":"2022-05-09T21:30:51.477922Z","iopub.status.idle":"2022-05-09T21:30:53.720861Z","shell.execute_reply.started":"2022-05-09T21:30:51.477884Z","shell.execute_reply":"2022-05-09T21:30:53.719939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = transactions_train_df.sample(100_000).groupby([\"t_dat\"])[\"article_id\"].count().reset_index()\ndf[\"t_dat\"] = df[\"t_dat\"].apply(lambda x: datetime.strptime(x, '%Y-%m-%d'))\ndf.columns = [\"Date\", \"Transactions\"]\nfig, ax = plt.subplots(1, 1, figsize=(16,6))\nplt.plot(df[\"Date\"], df[\"Transactions\"], color=\"Darkgreen\")\nplt.xlabel(\"Date\")\nplt.ylabel(\"Transactions\")\nplt.title(f\"Transactions per day (100k sample; to get the real volume, please consider that real transaction count is {round(transactions_train_df.shape[0]/10.e6,2)}M)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:31:35.159801Z","iopub.execute_input":"2022-05-09T21:31:35.160129Z","iopub.status.idle":"2022-05-09T21:31:37.554825Z","shell.execute_reply.started":"2022-05-09T21:31:35.160097Z","shell.execute_reply":"2022-05-09T21:31:37.553658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = transactions_train_df.groupby([\"t_dat\", \"sales_channel_id\"])[\"article_id\"].nunique().reset_index()\ndf[\"t_dat\"] = df[\"t_dat\"].apply(lambda x: datetime.strptime(x, '%Y-%m-%d'))\ndf.columns = [\"Date\", \"Sales Channel Id\", \"Unique Articles\"]\nfig, ax = plt.subplots(1, 1, figsize=(16,6))\ng1 = ax.plot(df.loc[df[\"Sales Channel Id\"]==1, \"Date\"], df.loc[df[\"Sales Channel Id\"]==1, \"Unique Articles\"], label=\"Sales Channel 1\", color=\"Blue\")\ng2 = ax.plot(df.loc[df[\"Sales Channel Id\"]==2, \"Date\"], df.loc[df[\"Sales Channel Id\"]==2, \"Unique Articles\"], label=\"Sales Channel 2\", color=\"Green\")\nplt.xlabel(\"Date\")\nplt.ylabel(\"Unique Articles / Day\")\nax.legend()\nplt.title(f\"Unique articles per day, grouped by Sales Channel\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:33:56.224476Z","iopub.execute_input":"2022-05-09T21:33:56.224849Z","iopub.status.idle":"2022-05-09T21:34:14.973423Z","shell.execute_reply.started":"2022-05-09T21:33:56.224812Z","shell.execute_reply":"2022-05-09T21:34:14.972708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Image Data ","metadata":{}},{"cell_type":"code","source":"image_name_df = pd.DataFrame(images_names, columns = [\"image_name\"])\nimage_name_df[\"article_id\"] = image_name_df[\"image_name\"].apply(lambda x: int(x[1:]))","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:42:30.570184Z","iopub.execute_input":"2022-05-09T21:42:30.570827Z","iopub.status.idle":"2022-05-09T21:42:30.683656Z","shell.execute_reply.started":"2022-05-09T21:42:30.570779Z","shell.execute_reply":"2022-05-09T21:42:30.682696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_name_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:43:07.808264Z","iopub.execute_input":"2022-05-09T21:43:07.808955Z","iopub.status.idle":"2022-05-09T21:43:07.825575Z","shell.execute_reply.started":"2022-05-09T21:43:07.808911Z","shell.execute_reply":"2022-05-09T21:43:07.824214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_article_df = articles_df[[\"article_id\", \"product_code\", \"product_group_name\", \"product_type_name\"]].merge(image_name_df, on=[\"article_id\"], how=\"left\")\nprint(image_article_df.shape)\nimage_article_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:43:55.575985Z","iopub.execute_input":"2022-05-09T21:43:55.576913Z","iopub.status.idle":"2022-05-09T21:43:55.673193Z","shell.execute_reply.started":"2022-05-09T21:43:55.576872Z","shell.execute_reply":"2022-05-09T21:43:55.672247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This gives us the products without displaying the images.","metadata":{}},{"cell_type":"code","source":"article_no_image_df = image_article_df.loc[image_article_df.image_name.isna()]\nprint(article_no_image_df.shape)\narticle_no_image_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:46:06.005042Z","iopub.execute_input":"2022-05-09T21:46:06.005597Z","iopub.status.idle":"2022-05-09T21:46:06.044841Z","shell.execute_reply.started":"2022-05-09T21:46:06.005542Z","shell.execute_reply":"2022-05-09T21:46:06.043867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This gives us a dataframe of products with missing images.","metadata":{}},{"cell_type":"code","source":"print(\"Product codes with some missing images: \", article_no_image_df.product_code.nunique())\nprint(\"Product groups with some missing images: \", list(article_no_image_df.product_group_name.unique()))","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:48:30.869764Z","iopub.execute_input":"2022-05-09T21:48:30.870098Z","iopub.status.idle":"2022-05-09T21:48:30.878521Z","shell.execute_reply.started":"2022-05-09T21:48:30.870065Z","shell.execute_reply":"2022-05-09T21:48:30.877426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Here we want to visualize a few images.","metadata":{}},{"cell_type":"code","source":"def plot_image_samples(image_article_df, product_group_name, cols=1, rows=-1):\n    image_path = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/\"\n    _df = image_article_df.loc[image_article_df.product_group_name==product_group_name]\n    article_ids = _df.article_id.values[0:cols*rows]\n    plt.figure(figsize=(2 + 3 * cols, 2 + 4 * rows))\n    for i in range(cols * rows):\n        article_id = (\"0\" + str(article_ids[i]))[-10:]\n        plt.subplot(rows, cols, i + 1)\n        plt.axis('off')\n        plt.title(f\"{product_group_name} {article_id[:3]}\\n{article_id}.jpg\")\n        image = Image.open(f\"{image_path}{article_id[:3]}/{article_id}.jpg\")\n        plt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T21:50:33.009788Z","iopub.execute_input":"2022-05-09T21:50:33.010132Z","iopub.status.idle":"2022-05-09T21:50:33.020111Z","shell.execute_reply.started":"2022-05-09T21:50:33.010098Z","shell.execute_reply":"2022-05-09T21:50:33.019254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets begin the process of visualizing the images by choosing som product group names.","metadata":{}},{"cell_type":"code","source":"print(image_article_df.product_group_name.unique())","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:10:23.373191Z","iopub.execute_input":"2022-05-09T22:10:23.373785Z","iopub.status.idle":"2022-05-09T22:10:23.3891Z","shell.execute_reply.started":"2022-05-09T22:10:23.373747Z","shell.execute_reply":"2022-05-09T22:10:23.388195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will represent images grouped on product group name.","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"plot_image_samples(image_article_df, \"Garment Lower body\", 4, 2)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:11:35.293218Z","iopub.execute_input":"2022-05-09T22:11:35.293841Z","iopub.status.idle":"2022-05-09T22:11:37.898696Z","shell.execute_reply.started":"2022-05-09T22:11:35.293784Z","shell.execute_reply":"2022-05-09T22:11:37.89781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(image_article_df, \"Stationery\", 4, 1)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:12:25.085495Z","iopub.execute_input":"2022-05-09T22:12:25.085844Z","iopub.status.idle":"2022-05-09T22:12:26.858704Z","shell.execute_reply.started":"2022-05-09T22:12:25.085811Z","shell.execute_reply":"2022-05-09T22:12:26.857782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(image_article_df, \"Fun\", 2, 1)\n","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:12:57.61281Z","iopub.execute_input":"2022-05-09T22:12:57.613138Z","iopub.status.idle":"2022-05-09T22:12:58.584158Z","shell.execute_reply.started":"2022-05-09T22:12:57.613106Z","shell.execute_reply":"2022-05-09T22:12:58.583328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(image_article_df, \"Accessories\", 4, 1)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:13:25.31093Z","iopub.execute_input":"2022-05-09T22:13:25.311449Z","iopub.status.idle":"2022-05-09T22:13:28.391135Z","shell.execute_reply.started":"2022-05-09T22:13:25.311413Z","shell.execute_reply":"2022-05-09T22:13:28.389946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(image_article_df, \"Swimwear\", 4, 2)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:13:57.63488Z","iopub.execute_input":"2022-05-09T22:13:57.635347Z","iopub.status.idle":"2022-05-09T22:14:00.918503Z","shell.execute_reply.started":"2022-05-09T22:13:57.635294Z","shell.execute_reply":"2022-05-09T22:14:00.917855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(image_article_df, \"Furniture\", 4, 2)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:16:04.718765Z","iopub.execute_input":"2022-05-09T22:16:04.719312Z","iopub.status.idle":"2022-05-09T22:16:08.042329Z","shell.execute_reply.started":"2022-05-09T22:16:04.719252Z","shell.execute_reply":"2022-05-09T22:16:08.041483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(image_article_df, \"Cosmetic\", 4, 1)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:16:34.177863Z","iopub.execute_input":"2022-05-09T22:16:34.17846Z","iopub.status.idle":"2022-05-09T22:16:36.398515Z","shell.execute_reply.started":"2022-05-09T22:16:34.178419Z","shell.execute_reply":"2022-05-09T22:16:36.397201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(image_article_df, \"Cosmetic\", 4, 1)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:16:52.134958Z","iopub.execute_input":"2022-05-09T22:16:52.135268Z","iopub.status.idle":"2022-05-09T22:16:54.330982Z","shell.execute_reply.started":"2022-05-09T22:16:52.135239Z","shell.execute_reply":"2022-05-09T22:16:54.330385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(image_article_df, \"Bags\", 4, 3)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:17:18.589151Z","iopub.execute_input":"2022-05-09T22:17:18.590277Z","iopub.status.idle":"2022-05-09T22:17:23.543946Z","shell.execute_reply.started":"2022-05-09T22:17:18.590189Z","shell.execute_reply":"2022-05-09T22:17:23.54305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"markdown","source":"Let's prepare a very basic initial submission.\n\nFor this initial submission, we apply the following simplified logic:\n\nif there are articles for a certain client, pick the most recent buys;\nif there are not articles for a certain client, just pick the most frequently buyed articles.","metadata":{}},{"cell_type":"code","source":"transactions_train_df = transactions_train_df.sort_values([\"customer_id\", \"t_dat\"], ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:20:10.040151Z","iopub.execute_input":"2022-05-09T22:20:10.040487Z","iopub.status.idle":"2022-05-09T22:20:41.085236Z","shell.execute_reply.started":"2022-05-09T22:20:10.040455Z","shell.execute_reply":"2022-05-09T22:20:41.08393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:20:44.736316Z","iopub.execute_input":"2022-05-09T22:20:44.736876Z","iopub.status.idle":"2022-05-09T22:20:44.750926Z","shell.execute_reply.started":"2022-05-09T22:20:44.736839Z","shell.execute_reply":"2022-05-09T22:20:44.749964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's capture first what are the most frequent recently bought articles.","metadata":{}},{"cell_type":"code","source":"last_date = transactions_train_df.t_dat.max()\nprint(last_date)\nprint(transactions_train_df.loc[transactions_train_df.t_dat==last_date].shape)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:21:53.853237Z","iopub.execute_input":"2022-05-09T22:21:53.853614Z","iopub.status.idle":"2022-05-09T22:22:03.803741Z","shell.execute_reply.started":"2022-05-09T22:21:53.853575Z","shell.execute_reply":"2022-05-09T22:22:03.802562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_frequent_articles = list(transactions_train_df.loc[transactions_train_df.t_dat==last_date].article_id.value_counts()[0:12].index)\nart_list = []\nfor art in most_frequent_articles:\n    art = \"0\"+str(art)\n    art_list.append(art)\nart_str = \" \".join(art_list)\nprint(\"Frequent articles bought recently: \", art_str)\n","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:33:48.532943Z","iopub.execute_input":"2022-05-09T22:33:48.533229Z","iopub.status.idle":"2022-05-09T22:33:53.662272Z","shell.execute_reply.started":"2022-05-09T22:33:48.533198Z","shell.execute_reply":"2022-05-09T22:33:53.660812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We see that frequent bought articles that were bought recently are \"Frequent articles bought recently:  0924243002 0751471001 0448509014 0918522001 0866731001 0714790020 0788575004 0915529005 0573085028 0918292001 0850917001 0928206001\"\n","metadata":{}},{"cell_type":"code","source":"agg_df = transactions_train_df.groupby([\"customer_id\"])[\"article_id\"].agg(lambda x: str(x.values[0:12])[1:-1]).reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:36:13.061014Z","iopub.execute_input":"2022-05-09T22:36:13.061432Z","iopub.status.idle":"2022-05-09T22:38:00.561988Z","shell.execute_reply.started":"2022-05-09T22:36:13.061395Z","shell.execute_reply":"2022-05-09T22:38:00.56101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def padding_articles(x):\n    if x:\n        xl = x.split()\n        x = []\n        for xi in xl:\n            x.append(\"0\"+xi)\n        dimm_x = len(x)\n        if dimm_x < 12:\n            x.extend(art_list[:12-dimm_x])\n        return(\" \".join(x))","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:38:34.574836Z","iopub.execute_input":"2022-05-09T22:38:34.575184Z","iopub.status.idle":"2022-05-09T22:38:34.582882Z","shell.execute_reply.started":"2022-05-09T22:38:34.575147Z","shell.execute_reply":"2022-05-09T22:38:34.581951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"agg_df[\"article_id\"] = agg_df[\"article_id\"].apply(lambda x: padding_articles(x))","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:38:53.265986Z","iopub.execute_input":"2022-05-09T22:38:53.267159Z","iopub.status.idle":"2022-05-09T22:38:57.755196Z","shell.execute_reply.started":"2022-05-09T22:38:53.267112Z","shell.execute_reply":"2022-05-09T22:38:57.754066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Aggregated transaction history: \", agg_df.customer_id.nunique())\nprint(\"Submission sample: \", sample_submission_df.customer_id.nunique())","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:39:20.423512Z","iopub.execute_input":"2022-05-09T22:39:20.423878Z","iopub.status.idle":"2022-05-09T22:39:23.074561Z","shell.execute_reply.started":"2022-05-09T22:39:20.423843Z","shell.execute_reply":"2022-05-09T22:39:23.073366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will replace the values in sample submission with the existent in aggregated transactions data and just let the default one otherwise.","metadata":{}},{"cell_type":"code","source":"print(sample_submission_df.shape)\nsample_submission_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:40:21.950298Z","iopub.execute_input":"2022-05-09T22:40:21.950643Z","iopub.status.idle":"2022-05-09T22:40:21.962887Z","shell.execute_reply.started":"2022-05-09T22:40:21.950606Z","shell.execute_reply":"2022-05-09T22:40:21.96212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For the customers with missing articles, we simply replace with most frequent buyed articles in most recent day(s).\n\n","metadata":{}},{"cell_type":"code","source":"submission_df = agg_df.merge(sample_submission_df[[\"customer_id\"]], how=\"right\")\nsubmission_df.columns = [\"customer_id\", \"prediction\"]\nprint(submission_df.shape)\nsubmission_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:41:20.350194Z","iopub.execute_input":"2022-05-09T22:41:20.350618Z","iopub.status.idle":"2022-05-09T22:41:22.825686Z","shell.execute_reply.started":"2022-05-09T22:41:20.350551Z","shell.execute_reply":"2022-05-09T22:41:22.824713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Rows with missing data in submission: \", submission_df.loc[submission_df.prediction.isna()].shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:41:45.074476Z","iopub.execute_input":"2022-05-09T22:41:45.074915Z","iopub.status.idle":"2022-05-09T22:41:45.454996Z","shell.execute_reply.started":"2022-05-09T22:41:45.074876Z","shell.execute_reply":"2022-05-09T22:41:45.454063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We replace the missing data with the most frequently bought articles, from recent days. We calculated it before.","metadata":{}},{"cell_type":"code","source":"submission_df.loc[submission_df.prediction.isna(), [\"prediction\"]] = art_str","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:42:46.237067Z","iopub.execute_input":"2022-05-09T22:42:46.237393Z","iopub.status.idle":"2022-05-09T22:42:46.409682Z","shell.execute_reply.started":"2022-05-09T22:42:46.237359Z","shell.execute_reply":"2022-05-09T22:42:46.408083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Rows with missing data in submission: \", submission_df.loc[submission_df.prediction.isna()].shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:43:11.281035Z","iopub.execute_input":"2022-05-09T22:43:11.281548Z","iopub.status.idle":"2022-05-09T22:43:11.451823Z","shell.execute_reply.started":"2022-05-09T22:43:11.281497Z","shell.execute_reply":"2022-05-09T22:43:11.450992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T22:43:36.94464Z","iopub.execute_input":"2022-05-09T22:43:36.945137Z","iopub.status.idle":"2022-05-09T22:43:50.782543Z","shell.execute_reply.started":"2022-05-09T22:43:36.9451Z","shell.execute_reply":"2022-05-09T22:43:50.781452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}