{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport plotly.express as px\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nfrom IPython.core.display import display, HTML\nimport ipywidgets as widgets\nfrom IPython.display import display,clear_output\nfrom ipywidgets import Output\nfrom ipywidgets import TwoByTwoLayout\n# Utils widgets\nfrom ipywidgets import Button, Layout, jslink, IntText, IntSlider, Box, VBox\n\nfrom scipy.stats import pearsonr\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go\nimport plotly.io as pio\npio.templates\nfrom PIL import Image\nfrom IPython.display import Image as img\nfrom plotly.offline import plot, iplot, init_notebook_mode\nimport random\npd.options.mode.chained_assignment = None  # default='warn'\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-10T10:44:31.040250Z","iopub.execute_input":"2022-02-10T10:44:31.041105Z","iopub.status.idle":"2022-02-10T10:44:33.063636Z","shell.execute_reply.started":"2022-02-10T10:44:31.041063Z","shell.execute_reply":"2022-02-10T10:44:33.063066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the notebook [👚 Data exploration of H&M product recommendation](https://www.kaggle.com/tianmin/data-exploration-of-h-m-product-recommendation), I have found something interesting - there is a clear drop of number of articles sold in **sales_channel_id 1** during the period of **2020 April**, the time covid hit countries globally. So I started to guess the sales_channel_id could be a good indicator for clustering the users since either segment could approach H&M products in very different behavior. \n\nI will keep exploring this feature and post further finding here.\n\n","metadata":{}},{"cell_type":"code","source":"transactions = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")\ntransaction_sales = transactions.groupby(['t_dat','sales_channel_id']).nunique().reset_index()\nfig = px.line(transaction_sales, x='t_dat', y='customer_id', color='sales_channel_id',title=\"Nr of articles purchased per sales channel\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-10T10:46:39.639117Z","iopub.execute_input":"2022-02-10T10:46:39.639935Z","iopub.status.idle":"2022-02-10T10:48:02.109841Z","shell.execute_reply.started":"2022-02-10T10:46:39.639898Z","shell.execute_reply":"2022-02-10T10:48:02.109048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}