{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        c=1","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-01T20:51:53.398433Z","iopub.execute_input":"2024-09-01T20:51:53.399402Z","iopub.status.idle":"2024-09-01T20:52:09.454979Z","shell.execute_reply.started":"2024-09-01T20:51:53.399344Z","shell.execute_reply":"2024-09-01T20:52:09.453503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Introduction\n\nThe dataset contains 4 csv files and one folder with several subfolders, each with a different number of images.\n\nIn this Exploratory Data Analysis Notebook we will look to the data, will analyze the content of each csv file, check for missing data, understand the data distribution, see what are the relations between data in various files.\n\nWe will also explore the image data, understand how images are indexed in the csv files, if there are articles in the dataset without images. We will also explore image additional information, like image width and height.\n\nWe also investigate a very simple baseline model and create an initial submission.\n\n\n\n<img src=\"https://images.unsplash.com/photo-1578983662508-41895226ebfb?ixlib=rb-1.2.1&ixid=MnwxMjA3fDB8MHxwaG90by1wYWdlfHx8fGVufDB8fHx8&auto=format&fit=crop&w=1211&q=80\" width=600></img>\n","metadata":{}},{"cell_type":"markdown","source":"# Analysis preparation\n\nWe will include here the required packages for reading, parsing, filtering, processing, visualizing the data, both tabular and image.\n\n<img src=\"https://images.unsplash.com/photo-1607160199580-1b0c9b736b66?ixlib=rb-1.2.1&ixid=MnwxMjA3fDB8MHxwaG90by1wYWdlfHx8fGVufDB8fHx8&auto=format&fit=crop&w=2070&q=80\" width=600></img>\n","metadata":{}},{"cell_type":"markdown","source":"# Read and glimpse the data\n\n<img src=\"https://images.unsplash.com/photo-1532453288672-3a27e9be9efd?ixlib=rb-1.2.1&ixid=MnwxMjA3fDB8MHxwaG90by1wYWdlfHx8fGVufDB8fHx8&auto=format&fit=crop&w=764&q=80\" width=400></img>","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom wordcloud import WordCloud, STOPWORDS\nfrom datetime import datetime\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:52:09.456688Z","iopub.execute_input":"2024-09-01T20:52:09.457223Z","iopub.status.idle":"2024-09-01T20:52:10.156999Z","shell.execute_reply.started":"2024-09-01T20:52:09.457180Z","shell.execute_reply":"2024-09-01T20:52:10.155792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"files and folders: {os.listdir('/kaggle/input/h-and-m-personalized-fashion-recommendations/')}\")\nprint(\"Subfolders in images folder: \", len(list(os.listdir(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/images\"))))","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:52:10.159194Z","iopub.execute_input":"2024-09-01T20:52:10.159855Z","iopub.status.idle":"2024-09-01T20:52:10.167954Z","shell.execute_reply.started":"2024-09-01T20:52:10.159800Z","shell.execute_reply":"2024-09-01T20:52:10.166612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_folders = total_files = 0\nfolder_info = []\nimages_names = []\nfor base, dirs, files in tqdm(os.walk('/kaggle/input/h-and-m-personalized-fashion-recommendations/')):\n    for directories in dirs:\n        folder_info.append((directories, len(os.listdir(os.path.join(base, directories)))))\n        total_folders += 1\n    for _files in files:\n        total_files += 1\n        if len(_files.split(\".jpg\"))==2:\n            images_names.append(_files.split(\".jpg\")[0])","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:52:10.170994Z","iopub.execute_input":"2024-09-01T20:52:10.171434Z","iopub.status.idle":"2024-09-01T20:52:10.607570Z","shell.execute_reply.started":"2024-09-01T20:52:10.171394Z","shell.execute_reply":"2024-09-01T20:52:10.606472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Total number of folders: {total_folders}\\nTotal number of files: {total_files}\")\nfolder_info_df = pd.DataFrame(folder_info, columns=[\"folder\", \"files count\"])\nfolder_info_df.sort_values([\"files count\"], ascending=False).head()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:52:10.609089Z","iopub.execute_input":"2024-09-01T20:52:10.609561Z","iopub.status.idle":"2024-09-01T20:52:10.634499Z","shell.execute_reply.started":"2024-09-01T20:52:10.609521Z","shell.execute_reply":"2024-09-01T20:52:10.633378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"folder names: \", list(folder_info_df.folder.unique()))","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:52:10.636042Z","iopub.execute_input":"2024-09-01T20:52:10.636426Z","iopub.status.idle":"2024-09-01T20:52:10.643271Z","shell.execute_reply.started":"2024-09-01T20:52:10.636387Z","shell.execute_reply":"2024-09-01T20:52:10.641709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df = pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv\")\ncustomers_df = pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv\")\nsample_submission_df = pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:52:10.644869Z","iopub.execute_input":"2024-09-01T20:52:10.645400Z","iopub.status.idle":"2024-09-01T20:52:24.329208Z","shell.execute_reply.started":"2024-09-01T20:52:10.645358Z","shell.execute_reply":"2024-09-01T20:52:24.327994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_train_df = pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:52:24.330709Z","iopub.execute_input":"2024-09-01T20:52:24.331095Z","iopub.status.idle":"2024-09-01T20:53:20.969384Z","shell.execute_reply.started":"2024-09-01T20:52:24.331035Z","shell.execute_reply":"2024-09-01T20:53:20.968083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:20.971080Z","iopub.execute_input":"2024-09-01T20:53:20.972117Z","iopub.status.idle":"2024-09-01T20:53:21.000598Z","shell.execute_reply.started":"2024-09-01T20:53:20.972062Z","shell.execute_reply":"2024-09-01T20:53:20.999219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:21.006074Z","iopub.execute_input":"2024-09-01T20:53:21.006711Z","iopub.status.idle":"2024-09-01T20:53:21.028526Z","shell.execute_reply.started":"2024-09-01T20:53:21.006666Z","shell.execute_reply":"2024-09-01T20:53:21.027173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:21.029978Z","iopub.execute_input":"2024-09-01T20:53:21.030531Z","iopub.status.idle":"2024-09-01T20:53:21.045335Z","shell.execute_reply.started":"2024-09-01T20:53:21.030488Z","shell.execute_reply":"2024-09-01T20:53:21.043957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:21.047221Z","iopub.execute_input":"2024-09-01T20:53:21.048007Z","iopub.status.idle":"2024-09-01T20:53:21.065469Z","shell.execute_reply.started":"2024-09-01T20:53:21.047951Z","shell.execute_reply":"2024-09-01T20:53:21.064056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let's look closer to the data\n\n<img src=\"https://images.unsplash.com/photo-1569484221992-2a453658fff3?ixlib=rb-1.2.1&ixid=MnwxMjA3fDB8MHxwaG90by1wYWdlfHx8fGVufDB8fHx8&auto=format&fit=crop&w=1179&q=80\" width=600></img>\n\nThere are 3 main tables:\n- articles - contains informations about each article (like product code, name, product group code, name ...)    \n- customers - contains informations about each customer (fidelity card membership, age, postal code)\n- transactions (train)  \n\nTransactions have `customer_id` and `article_id`, which are foreign keys for the customer and articles tables.\nBeside this, transaction also contains `sales_channel_id`.","metadata":{}},{"cell_type":"code","source":"def missing_data(data):\n    total = data.isnull().sum().sort_values(ascending = False)\n    percent = (data.isnull().sum()/data.isnull().count()*100).sort_values(ascending = False)\n    return pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:21.072387Z","iopub.execute_input":"2024-09-01T20:53:21.072836Z","iopub.status.idle":"2024-09-01T20:53:21.079703Z","shell.execute_reply.started":"2024-09-01T20:53:21.072799Z","shell.execute_reply":"2024-09-01T20:53:21.078471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def unique_values(data):\n    total = data.count()\n    tt = pd.DataFrame(total)\n    tt.columns = ['Total']\n    uniques = []\n    for col in data.columns:\n        unique = data[col].nunique()\n        uniques.append(unique)\n    tt['Uniques'] = uniques\n    return tt","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:21.081212Z","iopub.execute_input":"2024-09-01T20:53:21.081580Z","iopub.status.idle":"2024-09-01T20:53:21.181555Z","shell.execute_reply.started":"2024-09-01T20:53:21.081542Z","shell.execute_reply":"2024-09-01T20:53:21.180172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:21.183096Z","iopub.execute_input":"2024-09-01T20:53:21.183479Z","iopub.status.idle":"2024-09-01T20:53:21.278885Z","shell.execute_reply.started":"2024-09-01T20:53:21.183444Z","shell.execute_reply":"2024-09-01T20:53:21.277611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(articles_df)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:21.280452Z","iopub.execute_input":"2024-09-01T20:53:21.280833Z","iopub.status.idle":"2024-09-01T20:53:21.512868Z","shell.execute_reply.started":"2024-09-01T20:53:21.280793Z","shell.execute_reply":"2024-09-01T20:53:21.511621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:21.514577Z","iopub.execute_input":"2024-09-01T20:53:21.515096Z","iopub.status.idle":"2024-09-01T20:53:21.825852Z","shell.execute_reply.started":"2024-09-01T20:53:21.515043Z","shell.execute_reply":"2024-09-01T20:53:21.824621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(customers_df)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:21.827312Z","iopub.execute_input":"2024-09-01T20:53:21.827808Z","iopub.status.idle":"2024-09-01T20:53:22.711984Z","shell.execute_reply.started":"2024-09-01T20:53:21.827756Z","shell.execute_reply":"2024-09-01T20:53:22.710742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:22.714164Z","iopub.execute_input":"2024-09-01T20:53:22.714649Z","iopub.status.idle":"2024-09-01T20:53:22.862003Z","shell.execute_reply.started":"2024-09-01T20:53:22.714595Z","shell.execute_reply":"2024-09-01T20:53:22.860881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_train_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:22.863343Z","iopub.execute_input":"2024-09-01T20:53:22.863697Z","iopub.status.idle":"2024-09-01T20:53:22.874401Z","shell.execute_reply.started":"2024-09-01T20:53:22.863659Z","shell.execute_reply":"2024-09-01T20:53:22.873306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(transactions_train_df)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:22.876060Z","iopub.execute_input":"2024-09-01T20:53:22.876577Z","iopub.status.idle":"2024-09-01T20:53:32.143271Z","shell.execute_reply.started":"2024-09-01T20:53:22.876510Z","shell.execute_reply":"2024-09-01T20:53:32.141970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Unique values","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:32.144904Z","iopub.execute_input":"2024-09-01T20:53:32.145421Z","iopub.status.idle":"2024-09-01T20:53:32.152381Z","shell.execute_reply.started":"2024-09-01T20:53:32.145368Z","shell.execute_reply":"2024-09-01T20:53:32.150922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_values(articles_df)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:32.154545Z","iopub.execute_input":"2024-09-01T20:53:32.155115Z","iopub.status.idle":"2024-09-01T20:53:32.384186Z","shell.execute_reply.started":"2024-09-01T20:53:32.155058Z","shell.execute_reply":"2024-09-01T20:53:32.382962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We observe that features for which we expect to have the same number of unique value, like:\n* product_type_no and product_type_name,  \n* departmant_no and department_name,  \n* section_no and section_name \nhave different number of unique values, which might means that we might have categories with same name.\nOthers, like:\n* index_code and index_name,\n* garment_group_no and garment_group_name\nhave the same number of unique values.","metadata":{}},{"cell_type":"code","source":"unique_values(customers_df)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:32.385329Z","iopub.execute_input":"2024-09-01T20:53:32.385686Z","iopub.status.idle":"2024-09-01T20:53:34.133858Z","shell.execute_reply.started":"2024-09-01T20:53:32.385649Z","shell.execute_reply":"2024-09-01T20:53:34.132612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_values(transactions_train_df)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:34.135855Z","iopub.execute_input":"2024-09-01T20:53:34.136242Z","iopub.status.idle":"2024-09-01T20:53:48.974208Z","shell.execute_reply.started":"2024-09-01T20:53:34.136205Z","shell.execute_reply":"2024-09-01T20:53:48.973051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We observe that not all the customers in customer data appears as having transactions in transaction train data. As well, not all articles are represented in this data. It is interesting that the number of different prices is quite small, out of 31.7M transactions, and for 1.3M customers, buying 104K different articles. Same for the dates, there are only 734 different dates. Let's check some stats here.","metadata":{}},{"cell_type":"code","source":"print(f\"Percent of articles present in transactions: {round(104547/105542,3)*100}%\")\nprint(f\"Percent of articles present in transactions: {round(1362281/1371980,3)*100}%\")","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:48.975858Z","iopub.execute_input":"2024-09-01T20:53:48.976228Z","iopub.status.idle":"2024-09-01T20:53:48.982148Z","shell.execute_reply.started":"2024-09-01T20:53:48.976191Z","shell.execute_reply":"2024-09-01T20:53:48.980865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Articles data","metadata":{}},{"cell_type":"code","source":"temp = articles_df.groupby([\"product_group_name\"])[\"product_type_name\"].nunique()\ndf = pd.DataFrame({'Product Group': temp.index,\n                   'Product Types': temp.values\n                  })\ndf = df.sort_values(['Product Types'], ascending=False)\nplt.figure(figsize = (8,6))\nplt.title('Number of Product Types per each Product Group')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Product Group', y=\"Product Types\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:48.983865Z","iopub.execute_input":"2024-09-01T20:53:48.984294Z","iopub.status.idle":"2024-09-01T20:53:49.384641Z","shell.execute_reply.started":"2024-09-01T20:53:48.984240Z","shell.execute_reply":"2024-09-01T20:53:49.383372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stopwords = set(STOPWORDS)\n\ndef show_wordcloud(data, title = None):\n    wordcloud = WordCloud(\n        background_color='white',\n        stopwords=stopwords,\n        max_words=200,\n        max_font_size=40, \n        scale=5,\n        random_state=1\n    ).generate(str(data))\n\n    fig = plt.figure(1, figsize=(10,10))\n    plt.axis('off')\n    if title: \n        fig.suptitle(title, fontsize=14)\n        fig.subplots_adjust(top=2.3)\n\n    plt.imshow(wordcloud)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:49.386598Z","iopub.execute_input":"2024-09-01T20:53:49.387082Z","iopub.status.idle":"2024-09-01T20:53:49.396730Z","shell.execute_reply.started":"2024-09-01T20:53:49.387017Z","shell.execute_reply":"2024-09-01T20:53:49.395288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_wordcloud(articles_df[\"prod_name\"], \"Wordcloud from product name\")","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:49.398389Z","iopub.execute_input":"2024-09-01T20:53:49.398804Z","iopub.status.idle":"2024-09-01T20:53:49.925844Z","shell.execute_reply.started":"2024-09-01T20:53:49.398755Z","shell.execute_reply":"2024-09-01T20:53:49.924523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"product_group_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Product Group': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (8,6))\nplt.title('Number of Articles per each Product Group')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Product Group', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:49.927773Z","iopub.execute_input":"2024-09-01T20:53:49.928230Z","iopub.status.idle":"2024-09-01T20:53:50.357282Z","shell.execute_reply.started":"2024-09-01T20:53:49.928183Z","shell.execute_reply":"2024-09-01T20:53:50.355909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"product_type_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Product Type': temp.index,\n                   'Articles': temp.values\n                  })\ntotal_types = len(df['Product Type'].unique())\ndf = df.sort_values(['Articles'], ascending=False)[0:50]\nplt.figure(figsize = (16,6))\nplt.title(f'Number of Articles per each Product Type (top 50 from total: {total_types})')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Product Type', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:50.359056Z","iopub.execute_input":"2024-09-01T20:53:50.359536Z","iopub.status.idle":"2024-09-01T20:53:51.072053Z","shell.execute_reply.started":"2024-09-01T20:53:50.359485Z","shell.execute_reply":"2024-09-01T20:53:51.070671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"department_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Department Name': temp.index,\n                   'Articles': temp.values\n                  })\ntotal_depts = len(df['Department Name'].unique())\ndf = df.sort_values(['Articles'], ascending=False).head(50)\nplt.figure(figsize = (16,6))\nplt.title(f'Number of Articles per each Department (top 50 from total: {total_depts})')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Department Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:51.073986Z","iopub.execute_input":"2024-09-01T20:53:51.074438Z","iopub.status.idle":"2024-09-01T20:53:51.799937Z","shell.execute_reply.started":"2024-09-01T20:53:51.074389Z","shell.execute_reply":"2024-09-01T20:53:51.798710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"graphical_appearance_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Graphical Appearance Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False).head(50)\nplt.figure(figsize = (16,6))\nplt.title(f'Number of Articles per each Graphical Appearance Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Graphical Appearance Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:51.801470Z","iopub.execute_input":"2024-09-01T20:53:51.801857Z","iopub.status.idle":"2024-09-01T20:53:52.342808Z","shell.execute_reply.started":"2024-09-01T20:53:51.801819Z","shell.execute_reply":"2024-09-01T20:53:52.341551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"index_group_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Index Group Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (6,6))\nplt.title(f'Number of Articles per each Index Group Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Index Group Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:52.344393Z","iopub.execute_input":"2024-09-01T20:53:52.344863Z","iopub.status.idle":"2024-09-01T20:53:52.603530Z","shell.execute_reply.started":"2024-09-01T20:53:52.344810Z","shell.execute_reply":"2024-09-01T20:53:52.602003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"colour_group_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Colour Group Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (12,6))\nplt.title(f'Number of Articles per each Colour Group Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Colour Group Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:52.605585Z","iopub.execute_input":"2024-09-01T20:53:52.606004Z","iopub.status.idle":"2024-09-01T20:53:53.404373Z","shell.execute_reply.started":"2024-09-01T20:53:52.605955Z","shell.execute_reply":"2024-09-01T20:53:53.402906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"perceived_colour_value_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Perceived Colour Group Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (6,6))\nplt.title(f'Number of Articles per each Perceived Colour Group Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Perceived Colour Group Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:53.405981Z","iopub.execute_input":"2024-09-01T20:53:53.406642Z","iopub.status.idle":"2024-09-01T20:53:53.738128Z","shell.execute_reply.started":"2024-09-01T20:53:53.406590Z","shell.execute_reply":"2024-09-01T20:53:53.736970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"perceived_colour_master_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Perceived Colour Master Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (12,6))\nplt.title(f'Number of Articles per each Perceived Colour Master Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Perceived Colour Master Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:53.739555Z","iopub.execute_input":"2024-09-01T20:53:53.739897Z","iopub.status.idle":"2024-09-01T20:53:54.151119Z","shell.execute_reply.started":"2024-09-01T20:53:53.739861Z","shell.execute_reply":"2024-09-01T20:53:54.149765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"index_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Index Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (8,6))\nplt.title(f'Number of Articles per each Index Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Index Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:54.153077Z","iopub.execute_input":"2024-09-01T20:53:54.153859Z","iopub.status.idle":"2024-09-01T20:53:54.496283Z","shell.execute_reply.started":"2024-09-01T20:53:54.153805Z","shell.execute_reply":"2024-09-01T20:53:54.495092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"garment_group_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Garment Group Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (12,6))\nplt.title(f'Number of Articles per each Garment Group Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Garment Group Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:54.497877Z","iopub.execute_input":"2024-09-01T20:53:54.498302Z","iopub.status.idle":"2024-09-01T20:53:54.951115Z","shell.execute_reply.started":"2024-09-01T20:53:54.498260Z","shell.execute_reply":"2024-09-01T20:53:54.949756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"section_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Section Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (16,6))\nplt.title(f'Number of Articles per each Section Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Section Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:54.963472Z","iopub.execute_input":"2024-09-01T20:53:54.963890Z","iopub.status.idle":"2024-09-01T20:53:55.796487Z","shell.execute_reply.started":"2024-09-01T20:53:54.963851Z","shell.execute_reply":"2024-09-01T20:53:55.795185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"section_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Section Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (16,6))\nplt.title(f'Number of Articles per each Section Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Section Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:55.798089Z","iopub.execute_input":"2024-09-01T20:53:55.798542Z","iopub.status.idle":"2024-09-01T20:53:56.629283Z","shell.execute_reply.started":"2024-09-01T20:53:55.798496Z","shell.execute_reply":"2024-09-01T20:53:56.628108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_wordcloud(articles_df[\"detail_desc\"], \"Wordcloud from detailed description of articles\")","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:56.631113Z","iopub.execute_input":"2024-09-01T20:53:56.632123Z","iopub.status.idle":"2024-09-01T20:53:57.204755Z","shell.execute_reply.started":"2024-09-01T20:53:56.632068Z","shell.execute_reply":"2024-09-01T20:53:57.203438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Customers data","metadata":{}},{"cell_type":"code","source":"temp = customers_df.groupby([\"age\"])[\"customer_id\"].count()\ndf = pd.DataFrame({'Age': temp.index,\n                   'Customers': temp.values\n                  })\ndf = df.sort_values(['Age'], ascending=False)\nplt.figure(figsize = (16,6))\nplt.title(f'Number of Customers per each Age')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Age', y=\"Customers\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:57.206367Z","iopub.execute_input":"2024-09-01T20:53:57.206844Z","iopub.status.idle":"2024-09-01T20:53:58.370706Z","shell.execute_reply.started":"2024-09-01T20:53:57.206792Z","shell.execute_reply":"2024-09-01T20:53:58.369541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = customers_df.groupby([\"fashion_news_frequency\"])[\"customer_id\"].count()\ndf = pd.DataFrame({'Fashion News Frequency': temp.index,\n                   'Customers': temp.values\n                  })\ndf = df.sort_values(['Customers'], ascending=False)\nplt.figure(figsize = (6,6))\nplt.title(f'Number of Customers per each Fashion News Frequency')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Fashion News Frequency', y=\"Customers\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:58.372431Z","iopub.execute_input":"2024-09-01T20:53:58.372835Z","iopub.status.idle":"2024-09-01T20:53:58.766098Z","shell.execute_reply.started":"2024-09-01T20:53:58.372796Z","shell.execute_reply":"2024-09-01T20:53:58.764882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = customers_df.groupby([\"club_member_status\"])[\"customer_id\"].count()\ndf = pd.DataFrame({'Club Member Status': temp.index,\n                   'Customers': temp.values\n                  })\ndf = df.sort_values(['Customers'], ascending=False)\nplt.figure(figsize = (6,6))\nplt.title(f'Number of Customers per each Club Member Status')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Club Member Status', y=\"Customers\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:58.768048Z","iopub.execute_input":"2024-09-01T20:53:58.768562Z","iopub.status.idle":"2024-09-01T20:53:59.261659Z","shell.execute_reply.started":"2024-09-01T20:53:58.768489Z","shell.execute_reply":"2024-09-01T20:53:59.260266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transactions data","metadata":{}},{"cell_type":"code","source":"df = transactions_train_df.sample(100_000)\nfig, ax = plt.subplots(1, 1, figsize=(14, 7))\nsns.kdeplot(np.log(df.loc[df[\"sales_channel_id\"]==1].price.value_counts()))\nsns.kdeplot(np.log(df.loc[df[\"sales_channel_id\"]==2].price.value_counts()))\nax.legend(labels=['Sales channel 1', 'Sales channel 1'])\nplt.title(\"Logaritmic distribution of price frequency in transactions, grouped per sales channel (100k sample)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:53:59.263419Z","iopub.execute_input":"2024-09-01T20:53:59.264643Z","iopub.status.idle":"2024-09-01T20:54:01.681423Z","shell.execute_reply.started":"2024-09-01T20:53:59.264580Z","shell.execute_reply":"2024-09-01T20:54:01.680124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = transactions_train_df.sample(100_000).groupby([\"t_dat\"])[\"article_id\"].count().reset_index()\ndf[\"t_dat\"] = df[\"t_dat\"].apply(lambda x: datetime.strptime(x, '%Y-%m-%d'))\ndf.columns = [\"Date\", \"Transactions\"]\nfig, ax = plt.subplots(1, 1, figsize=(16,6))\nplt.plot(df[\"Date\"], df[\"Transactions\"], color=\"Darkgreen\")\nplt.xlabel(\"Date\")\nplt.ylabel(\"Transactions\")\nplt.title(f\"Transactions per day (100k sample; to get the real volume, please consider that real transaction count is {round(transactions_train_df.shape[0]/10.e6,2)}M)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:01.683601Z","iopub.execute_input":"2024-09-01T20:54:01.684008Z","iopub.status.idle":"2024-09-01T20:54:04.141228Z","shell.execute_reply.started":"2024-09-01T20:54:01.683966Z","shell.execute_reply":"2024-09-01T20:54:04.140058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = transactions_train_df.sample(100_000).groupby([\"t_dat\", \"sales_channel_id\"])[\"article_id\"].count().reset_index()\ndf[\"t_dat\"] = df[\"t_dat\"].apply(lambda x: datetime.strptime(x, '%Y-%m-%d'))\ndf.columns = [\"Date\", \"Sales Channel Id\", \"Transactions\"]\nfig, ax = plt.subplots(1, 1, figsize=(16,6))\ng1 = ax.plot(df.loc[df[\"Sales Channel Id\"]==1, \"Date\"], df.loc[df[\"Sales Channel Id\"]==1, \"Transactions\"], label=\"Sales Channel 1\", color=\"Darkblue\")\ng2 = ax.plot(df.loc[df[\"Sales Channel Id\"]==2, \"Date\"], df.loc[df[\"Sales Channel Id\"]==2, \"Transactions\"], label=\"Sales Channel 2\", color=\"Magenta\")\nplt.xlabel(\"Date\")\nplt.ylabel(\"Transactions\")\nax.legend()\nplt.title(f\"Transactions per day, grouped by Sales Channel (100k sample; to get the real volume, please consider that real transaction count is {round(transactions_train_df.shape[0]/10.e6,2)}M)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:04.142670Z","iopub.execute_input":"2024-09-01T20:54:04.143194Z","iopub.status.idle":"2024-09-01T20:54:06.617619Z","shell.execute_reply.started":"2024-09-01T20:54:04.143144Z","shell.execute_reply":"2024-09-01T20:54:06.616121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = transactions_train_df.groupby([\"t_dat\", \"sales_channel_id\"])[\"article_id\"].nunique().reset_index()\ndf[\"t_dat\"] = df[\"t_dat\"].apply(lambda x: datetime.strptime(x, '%Y-%m-%d'))\ndf.columns = [\"Date\", \"Sales Channel Id\", \"Unique Articles\"]\nfig, ax = plt.subplots(1, 1, figsize=(16,6))\ng1 = ax.plot(df.loc[df[\"Sales Channel Id\"]==1, \"Date\"], df.loc[df[\"Sales Channel Id\"]==1, \"Unique Articles\"], label=\"Sales Channel 1\", color=\"Blue\")\ng2 = ax.plot(df.loc[df[\"Sales Channel Id\"]==2, \"Date\"], df.loc[df[\"Sales Channel Id\"]==2, \"Unique Articles\"], label=\"Sales Channel 2\", color=\"Green\")\nplt.xlabel(\"Date\")\nplt.ylabel(\"Unique Articles / Day\")\nax.legend()\nplt.title(f\"Unique articles per day, grouped by Sales Channel\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:06.619121Z","iopub.execute_input":"2024-09-01T20:54:06.619619Z","iopub.status.idle":"2024-09-01T20:54:14.612368Z","shell.execute_reply.started":"2024-09-01T20:54:06.619577Z","shell.execute_reply":"2024-09-01T20:54:14.611085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image data\n\n<img src=\"https://images.unsplash.com/photo-1575729312527-1bdecaae271e?ixlib=rb-1.2.1&ixid=MnwxMjA3fDB8MHxwaG90by1wYWdlfHx8fGVufDB8fHx8&auto=format&fit=crop&w=687&q=80\" width=400></img>\n\nThere are 105542 articles and 105100 different images. Let's check first which articles does not have corresponding images.\n\nThe `article_id` corresponds to digits from 2nd to the last of the image name. \nThe digits from 2nd to 7th of image name  correspond to product code (`product_code`). ","metadata":{}},{"cell_type":"code","source":"image_name_df = pd.DataFrame(images_names, columns = [\"image_name\"])\nimage_name_df[\"article_id\"] = image_name_df[\"image_name\"].apply(lambda x: int(x[1:]))","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:14.613677Z","iopub.execute_input":"2024-09-01T20:54:14.614050Z","iopub.status.idle":"2024-09-01T20:54:14.690116Z","shell.execute_reply.started":"2024-09-01T20:54:14.613998Z","shell.execute_reply":"2024-09-01T20:54:14.688890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_name_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:14.692160Z","iopub.execute_input":"2024-09-01T20:54:14.692565Z","iopub.status.idle":"2024-09-01T20:54:14.703815Z","shell.execute_reply.started":"2024-09-01T20:54:14.692518Z","shell.execute_reply":"2024-09-01T20:54:14.702197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_article_df = articles_df[[\"article_id\", \"product_code\", \"product_group_name\", \"product_type_name\"]].merge(image_name_df, on=[\"article_id\"], how=\"left\")\nprint(image_article_df.shape)\nimage_article_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:14.705262Z","iopub.execute_input":"2024-09-01T20:54:14.705700Z","iopub.status.idle":"2024-09-01T20:54:14.760528Z","shell.execute_reply.started":"2024-09-01T20:54:14.705649Z","shell.execute_reply":"2024-09-01T20:54:14.759001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"article_no_image_df = image_article_df.loc[image_article_df.image_name.isna()]\nprint(article_no_image_df.shape)\narticle_no_image_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:14.763420Z","iopub.execute_input":"2024-09-01T20:54:14.763945Z","iopub.status.idle":"2024-09-01T20:54:14.791051Z","shell.execute_reply.started":"2024-09-01T20:54:14.763888Z","shell.execute_reply":"2024-09-01T20:54:14.789491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Product codes with some missing images: \", article_no_image_df.product_code.nunique())\nprint(\"Product groups with some missing images: \", list(article_no_image_df.product_group_name.unique()))","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:14.792649Z","iopub.execute_input":"2024-09-01T20:54:14.793039Z","iopub.status.idle":"2024-09-01T20:54:14.800617Z","shell.execute_reply.started":"2024-09-01T20:54:14.792980Z","shell.execute_reply":"2024-09-01T20:54:14.799487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's visualize few images.","metadata":{}},{"cell_type":"code","source":"def plot_image_samples(image_article_df, product_group_name, cols=1, rows=-1):\n    image_path = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/\"\n    _df = image_article_df.loc[image_article_df.product_group_name==product_group_name]\n    article_ids = _df.article_id.values[0:cols*rows]\n    plt.figure(figsize=(2 + 3 * cols, 2 + 4 * rows))\n    for i in range(cols * rows):\n        article_id = (\"0\" + str(article_ids[i]))[-10:]\n        plt.subplot(rows, cols, i + 1)\n        plt.axis('off')\n        plt.title(f\"{product_group_name} {article_id[:3]}\\n{article_id}.jpg\")\n        image = Image.open(f\"{image_path}{article_id[:3]}/{article_id}.jpg\")\n        plt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:14.801878Z","iopub.execute_input":"2024-09-01T20:54:14.802319Z","iopub.status.idle":"2024-09-01T20:54:14.811915Z","shell.execute_reply.started":"2024-09-01T20:54:14.802278Z","shell.execute_reply":"2024-09-01T20:54:14.810631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(image_article_df.product_group_name.unique())","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:14.813450Z","iopub.execute_input":"2024-09-01T20:54:14.813818Z","iopub.status.idle":"2024-09-01T20:54:14.836624Z","shell.execute_reply.started":"2024-09-01T20:54:14.813782Z","shell.execute_reply":"2024-09-01T20:54:14.835301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We will represent images grouped on product group name.","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:14.838392Z","iopub.execute_input":"2024-09-01T20:54:14.838831Z","iopub.status.idle":"2024-09-01T20:54:14.844609Z","shell.execute_reply.started":"2024-09-01T20:54:14.838782Z","shell.execute_reply":"2024-09-01T20:54:14.843411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(image_article_df, \"Garment Lower body\", 4, 2)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:14.846162Z","iopub.execute_input":"2024-09-01T20:54:14.846572Z","iopub.status.idle":"2024-09-01T20:54:18.376846Z","shell.execute_reply.started":"2024-09-01T20:54:14.846534Z","shell.execute_reply":"2024-09-01T20:54:18.375570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(image_article_df, \"Stationery\", 4, 1)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:18.378483Z","iopub.execute_input":"2024-09-01T20:54:18.378887Z","iopub.status.idle":"2024-09-01T20:54:20.778754Z","shell.execute_reply.started":"2024-09-01T20:54:18.378848Z","shell.execute_reply":"2024-09-01T20:54:20.777477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(image_article_df, \"Fun\", 2, 1)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:20.780107Z","iopub.execute_input":"2024-09-01T20:54:20.780525Z","iopub.status.idle":"2024-09-01T20:54:22.003563Z","shell.execute_reply.started":"2024-09-01T20:54:20.780485Z","shell.execute_reply":"2024-09-01T20:54:22.002407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(image_article_df, \"Fun\", 2, 1)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:22.004883Z","iopub.execute_input":"2024-09-01T20:54:22.005285Z","iopub.status.idle":"2024-09-01T20:54:23.207979Z","shell.execute_reply.started":"2024-09-01T20:54:22.005237Z","shell.execute_reply":"2024-09-01T20:54:23.206860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(image_article_df, \"Accessories\", 4, 1)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:23.209428Z","iopub.execute_input":"2024-09-01T20:54:23.209805Z","iopub.status.idle":"2024-09-01T20:54:26.154050Z","shell.execute_reply.started":"2024-09-01T20:54:23.209768Z","shell.execute_reply":"2024-09-01T20:54:26.150851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(image_article_df, \"Swimwear\", 4, 2)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:26.155944Z","iopub.execute_input":"2024-09-01T20:54:26.156745Z","iopub.status.idle":"2024-09-01T20:54:30.513822Z","shell.execute_reply.started":"2024-09-01T20:54:26.156689Z","shell.execute_reply":"2024-09-01T20:54:30.512389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(image_article_df, \"Furniture\", 4, 2)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:30.515265Z","iopub.execute_input":"2024-09-01T20:54:30.515732Z","iopub.status.idle":"2024-09-01T20:54:35.107323Z","shell.execute_reply.started":"2024-09-01T20:54:30.515691Z","shell.execute_reply":"2024-09-01T20:54:35.106127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(image_article_df, \"Cosmetic\", 4, 1)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:35.108738Z","iopub.execute_input":"2024-09-01T20:54:35.109786Z","iopub.status.idle":"2024-09-01T20:54:37.840689Z","shell.execute_reply.started":"2024-09-01T20:54:35.109727Z","shell.execute_reply":"2024-09-01T20:54:37.839566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_image_samples(image_article_df, \"Bags\", 4, 3)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:37.842006Z","iopub.execute_input":"2024-09-01T20:54:37.842455Z","iopub.status.idle":"2024-09-01T20:54:44.338280Z","shell.execute_reply.started":"2024-09-01T20:54:37.842414Z","shell.execute_reply":"2024-09-01T20:54:44.337085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Initial submission\n\n\n<img src=\"https://images.unsplash.com/photo-1533120164489-96c6ca1f43eb?ixlib=rb-1.2.1&ixid=MnwxMjA3fDB8MHxwaG90by1wYWdlfHx8fGVufDB8fHx8&auto=format&fit=crop&w=764&q=80\" width=400></img>\n\nLet's prepare a very basic initial submission.\n\nFor this initial submission, we apply the following simplified logic:\n- if there are articles for a certain client, pick the most recent buys;  \n- if there are not articles for a certain client, just pick the most frequently buyed articles.","metadata":{}},{"cell_type":"code","source":"transactions_train_df = transactions_train_df.sort_values([\"customer_id\", \"t_dat\"], ascending=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:54:44.340408Z","iopub.execute_input":"2024-09-01T20:54:44.340801Z","iopub.status.idle":"2024-09-01T20:55:06.910455Z","shell.execute_reply.started":"2024-09-01T20:54:44.340755Z","shell.execute_reply":"2024-09-01T20:55:06.909128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:55:06.912053Z","iopub.execute_input":"2024-09-01T20:55:06.912508Z","iopub.status.idle":"2024-09-01T20:55:06.926183Z","shell.execute_reply.started":"2024-09-01T20:55:06.912462Z","shell.execute_reply":"2024-09-01T20:55:06.924924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_date = transactions_train_df.t_dat.max()\nprint(last_date)\nprint(transactions_train_df.loc[transactions_train_df.t_dat==last_date].shape)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:55:06.928091Z","iopub.execute_input":"2024-09-01T20:55:06.928499Z","iopub.status.idle":"2024-09-01T20:55:11.806564Z","shell.execute_reply.started":"2024-09-01T20:55:06.928438Z","shell.execute_reply":"2024-09-01T20:55:11.805454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_frequent_articles = list(transactions_train_df.loc[transactions_train_df.t_dat==last_date].article_id.value_counts()[0:12].index)\nart_list = []\nfor art in most_frequent_articles:\n    art = \"0\"+str(art)\n    art_list.append(art)\nart_str = \" \".join(art_list)\nprint(\"Frequent articles bought recently: \", art_str)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:55:11.807881Z","iopub.execute_input":"2024-09-01T20:55:11.808246Z","iopub.status.idle":"2024-09-01T20:55:14.377222Z","shell.execute_reply.started":"2024-09-01T20:55:11.808203Z","shell.execute_reply":"2024-09-01T20:55:14.375996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"agg_df = transactions_train_df.groupby([\"customer_id\"])[\"article_id\"].agg(lambda x: str(x.values[0:12])[1:-1]).reset_index()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:55:14.379502Z","iopub.execute_input":"2024-09-01T20:55:14.380034Z","iopub.status.idle":"2024-09-01T20:57:22.042637Z","shell.execute_reply.started":"2024-09-01T20:55:14.379964Z","shell.execute_reply":"2024-09-01T20:57:22.041250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def padding_articles(x):\n    if x:\n        xl = x.split()\n        x = []\n        for xi in xl:\n            x.append(\"0\"+xi)\n        dimm_x = len(x)\n        if dimm_x < 12:\n            x.extend(art_list[:12-dimm_x])\n        return(\" \".join(x))","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:57:22.044161Z","iopub.execute_input":"2024-09-01T20:57:22.044632Z","iopub.status.idle":"2024-09-01T20:57:22.051809Z","shell.execute_reply.started":"2024-09-01T20:57:22.044592Z","shell.execute_reply":"2024-09-01T20:57:22.050158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"agg_df[\"article_id\"] = agg_df[\"article_id\"].apply(lambda x: padding_articles(x))","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:57:22.053658Z","iopub.execute_input":"2024-09-01T20:57:22.054215Z","iopub.status.idle":"2024-09-01T20:57:25.516539Z","shell.execute_reply.started":"2024-09-01T20:57:22.054161Z","shell.execute_reply":"2024-09-01T20:57:25.515362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Aggregated transaction history: \", agg_df.customer_id.nunique())\nprint(\"Submission sample: \", sample_submission_df.customer_id.nunique())","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:57:25.517895Z","iopub.execute_input":"2024-09-01T20:57:25.518385Z","iopub.status.idle":"2024-09-01T20:57:27.280457Z","shell.execute_reply.started":"2024-09-01T20:57:25.518335Z","shell.execute_reply":"2024-09-01T20:57:27.278958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(sample_submission_df.shape)\nsample_submission_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:57:27.281745Z","iopub.execute_input":"2024-09-01T20:57:27.282152Z","iopub.status.idle":"2024-09-01T20:57:27.295198Z","shell.execute_reply.started":"2024-09-01T20:57:27.282115Z","shell.execute_reply":"2024-09-01T20:57:27.293766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = agg_df.merge(sample_submission_df[[\"customer_id\"]], how=\"right\")\nsubmission_df.columns = [\"customer_id\", \"prediction\"]\nprint(submission_df.shape)\nsubmission_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:57:27.296929Z","iopub.execute_input":"2024-09-01T20:57:27.297434Z","iopub.status.idle":"2024-09-01T20:57:28.508251Z","shell.execute_reply.started":"2024-09-01T20:57:27.297390Z","shell.execute_reply":"2024-09-01T20:57:28.506996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Rows with missing data in submission: \", submission_df.loc[submission_df.prediction.isna()].shape[0])","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:57:28.509408Z","iopub.execute_input":"2024-09-01T20:57:28.509740Z","iopub.status.idle":"2024-09-01T20:57:28.610377Z","shell.execute_reply.started":"2024-09-01T20:57:28.509707Z","shell.execute_reply":"2024-09-01T20:57:28.608920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.loc[submission_df.prediction.isna(), [\"prediction\"]] = art_str","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:57:28.612508Z","iopub.execute_input":"2024-09-01T20:57:28.613214Z","iopub.status.idle":"2024-09-01T20:57:28.712733Z","shell.execute_reply.started":"2024-09-01T20:57:28.613155Z","shell.execute_reply":"2024-09-01T20:57:28.711349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Rows with missing data in submission: \", submission_df.loc[submission_df.prediction.isna()].shape[0])","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:57:28.714284Z","iopub.execute_input":"2024-09-01T20:57:28.714693Z","iopub.status.idle":"2024-09-01T20:57:28.811784Z","shell.execute_reply.started":"2024-09-01T20:57:28.714653Z","shell.execute_reply":"2024-09-01T20:57:28.810438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-09-01T20:57:28.813480Z","iopub.execute_input":"2024-09-01T20:57:28.813915Z","iopub.status.idle":"2024-09-01T20:57:37.241436Z","shell.execute_reply.started":"2024-09-01T20:57:28.813863Z","shell.execute_reply":"2024-09-01T20:57:37.239990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}