{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nfrom glob import glob\nfrom tqdm import tqdm\nimport lightgbm as lgbm\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold, KFold\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\nimport random\nfrom PIL import Image\nimport math\nimport itertools\nfrom plotly.graph_objects import treemap\nimport plotly\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nimport plotly.io as pio\nimport plotly.subplots as sp\nimport gc\nfrom collections import Counter\n\ndef show_clear_plt():\n    plt.tight_layout()\n    plt.show()\n    plt.clf()\n\n\nsns.set(font_scale=1.5)\nsns.set_style(style='darkgrid')\nplt.rcParams['figure.figsize'] = (10, 6)\nplt.rcParams['legend.facecolor'] = 'white'\n\ndef reduce_memory_usage(df, columns, verbose=True):\n    numerics = [\"int8\", \"int16\", \"int32\", \"int64\", \"float16\", \"float32\", \"float64\"]\n    start_mem = df.memory_usage().sum() / 1024 ** 2\n    for col in columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == \"int\":\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if (\n                        c_min > np.finfo(np.float16).min\n                        and c_max < np.finfo(np.float16).max\n                ):\n                    df[col] = df[col].astype(np.float16)\n                elif (\n                        c_min > np.finfo(np.float32).min\n                        and c_max < np.finfo(np.float32).max\n                ):\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n    end_mem = df.memory_usage().sum() / 1024 ** 2\n    if verbose:\n        print(\n            \"Mem. usage decreased to {:.2f} Mb ({:.1f}% reduction)\".format(\n                end_mem, 100 * (start_mem - end_mem) / start_mem\n            )\n        )\n    return df","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-17T01:43:57.604364Z","iopub.execute_input":"2022-02-17T01:43:57.605029Z","iopub.status.idle":"2022-02-17T01:44:00.338828Z","shell.execute_reply.started":"2022-02-17T01:43:57.604867Z","shell.execute_reply":"2022-02-17T01:44:00.337931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The\naim\nof\nthis\nnotebook is to\nprovide\nsome\nquick\nEDA of the H&M competition data\nby\nproducing\na\nnumber\nof\ngraphs / outputs\nwith limited comments. \n\nThis should provide an impression of the data and some examples and understanding of \nthe article (product) categorisations and transactions.\n\nIf\nyou\nhappen\nto\nbe\nlooking\nat\nthis\nnotebook and spot\nany\nmistakes, errors, misconceptions\netc, or any\nother\nproblem\nplease\nfeel\nfree\nto\npost in comments\nsection.\n","metadata":{}},{"cell_type":"markdown","source":"# Setup","metadata":{}},{"cell_type":"code","source":"class CONFIG:\n    KAGGLE = os.path.exists('../input/h-and-m-personalized-fashion-recommendations/')\n\n    if KAGGLE:\n        PATH = '../input/h-and-m-personalized-fashion-recommendations/'\n        print('running on Kaggle')\n    else:\n        PATH = 'NA'\n        print('not running on Kaggle')\n\n    DEBUG = False\n    DEBUG_PC = 0.1\n\n    print(f'debugging / reduce data rows is {DEBUG}')\n\n    EXAMPLE_LIMIT = 10","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-17T01:44:00.340279Z","iopub.execute_input":"2022-02-17T01:44:00.340481Z","iopub.status.idle":"2022-02-17T01:44:00.347231Z","shell.execute_reply.started":"2022-02-17T01:44:00.340457Z","shell.execute_reply":"2022-02-17T01:44:00.346319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In this notebook am not analysing the submission file","metadata":{}},{"cell_type":"code","source":"customers = pd.read_csv(CONFIG.PATH + 'customers.csv')\narticles = pd.read_csv(CONFIG.PATH + 'articles.csv')\ntransactions = pd.read_csv(CONFIG.PATH + 'transactions_train.csv',\n                           parse_dates=['t_dat'])\n\n#reduce memory usage\ncustomers = reduce_memory_usage(customers, customers.columns)\narticles = reduce_memory_usage(articles, articles.columns)\ntransactions = reduce_memory_usage(transactions, transactions.columns)\n\nprint('dataframe shapes, customes / articles / transactions')\nprint(customers.shape, articles.shape, transactions.shape)","metadata":{"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"collapsed":false,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-17T01:44:00.348692Z","iopub.execute_input":"2022-02-17T01:44:00.348995Z","iopub.status.idle":"2022-02-17T01:45:13.572792Z","shell.execute_reply.started":"2022-02-17T01:44:00.348938Z","shell.execute_reply":"2022-02-17T01:45:13.571164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data subsampling","metadata":{}},{"cell_type":"markdown","source":"(this reduces data size if DEBUG selected in CONFIG)","metadata":{}},{"cell_type":"code","source":"if CONFIG.DEBUG:\n    random.seed(42)\n\n    # subsample the articles\n    unique_articles = articles['article_id'].unique().tolist()\n    sample_articles = random.sample(unique_articles, int(CONFIG.DEBUG_PC * len(unique_articles)))\n    print(f'number of sample articles {len(sample_articles)}')\n\n    print(articles.shape)\n    articles = articles[articles['article_id'].isin(sample_articles)].reset_index(drop=True)\n    print(articles.shape)\n\n    # subsample the customers\n    unique_customers = customers['customer_id'].unique().tolist()\n    sample_customers = random.sample(unique_customers, int(CONFIG.DEBUG_PC * len(unique_customers)))\n    print(f'number of sample customers {len(sample_customers)}')\n\n    print(customers.shape)\n    customers = customers[customers['customer_id'].isin(sample_customers)].reset_index(drop=True)\n    print(customers.shape)\n\n    print(f'original train transactions shape {transactions.shape}')\n    transactions = transactions[(transactions['customer_id'].isin(sample_customers)) &\n                                (transactions['article_id'].isin(sample_articles))].reset_index(drop=True)\n\n    print(f'reduced train transactions shape {transactions.shape}')\n    \nelse:\n    print('running with all train data')","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-17T01:45:13.574122Z","iopub.execute_input":"2022-02-17T01:45:13.574341Z","iopub.status.idle":"2022-02-17T01:45:13.58503Z","shell.execute_reply.started":"2022-02-17T01:45:13.574314Z","shell.execute_reply":"2022-02-17T01:45:13.584199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explore Articles","metadata":{}},{"cell_type":"code","source":"print('Article data columns')\nprint(articles.columns.tolist())","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-17T01:45:13.587233Z","iopub.execute_input":"2022-02-17T01:45:13.587458Z","iopub.status.idle":"2022-02-17T01:45:13.602289Z","shell.execute_reply.started":"2022-02-17T01:45:13.587431Z","shell.execute_reply":"2022-02-17T01:45:13.601296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'there are {articles.shape[0]} rows in the articles data')\nprint(' ')\nfor c in articles.columns:\n    print(f'for {c} there are {articles[c].nunique()} unique entries')","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-17T01:45:13.603782Z","iopub.execute_input":"2022-02-17T01:45:13.604392Z","iopub.status.idle":"2022-02-17T01:45:13.76311Z","shell.execute_reply.started":"2022-02-17T01:45:13.604343Z","shell.execute_reply":"2022-02-17T01:45:13.762122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Numbers of unique entries by column","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(figsize=(10, 9))\nplt.barh(y=articles.nunique().index,\n         width=articles.nunique().values,\n         color='Green')\nplt.title('Count of uniques for article categories')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-17T01:45:13.764866Z","iopub.execute_input":"2022-02-17T01:45:13.765121Z","iopub.status.idle":"2022-02-17T01:45:14.523792Z","shell.execute_reply.started":"2022-02-17T01:45:13.765088Z","shell.execute_reply":"2022-02-17T01:45:14.522941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(figsize=(12, 9))\nplt.barh(y=articles.nunique().index[articles.nunique()<500],\n         width=articles.nunique().values[articles.nunique()<500],\n         color='Green')\nplt.title('Count of uniques for article categories (lower count categories)')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-17T01:45:14.525164Z","iopub.execute_input":"2022-02-17T01:45:14.525431Z","iopub.status.idle":"2022-02-17T01:45:15.524451Z","shell.execute_reply.started":"2022-02-17T01:45:14.525396Z","shell.execute_reply":"2022-02-17T01:45:15.523572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Top descriptions in each category (sorted by the number of articles)","metadata":{}},{"cell_type":"code","source":"#categories more related to department / type of product\ncount_columns = [\n    'product_type_name',\n    'product_group_name',\n    'department_name',\n    'index_name',\n    'index_group_name',\n    'section_name',\n    'garment_group_name',\n]\nfig, axes = plt.subplots(ncols=2,\n                         nrows=len(count_columns),\n                         figsize=(15, len(count_columns) * 5),\n                        \n                         sharey='row',)\n                      #  sharex='col')\n\nfor count, cc in enumerate(count_columns):\n    vc = articles[cc].value_counts() / len(articles) * 100\n    vc = vc.sort_values(ascending=False)\n    vc = vc[:CONFIG.EXAMPLE_LIMIT]\n\n    axes[count, 0].barh(width=vc.values,\n                 y=vc.index,\n                 color='Green',\n                 )\n\n    axes[count, 1].barh(width=vc.values.cumsum(),\n                 y=vc.index,\n                 color='Green',\n                 )\n\n    axes[count, 0].set_xlim(0, 50)\n    axes[count, 1].set_xlim(0, 100)\n    axes[count, 0].set_title(f'% of total {cc}')\n    axes[count, 1].set_title(f'cumulative % {cc}')\n    \nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-17T01:45:15.525644Z","iopub.execute_input":"2022-02-17T01:45:15.525957Z","iopub.status.idle":"2022-02-17T01:45:18.333553Z","shell.execute_reply.started":"2022-02-17T01:45:15.52592Z","shell.execute_reply":"2022-02-17T01:45:18.332724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Categories related to colour / design","metadata":{}},{"cell_type":"code","source":"#columns related to colour or pattern/design\ncount_columns = [\n    'graphical_appearance_name',\n    'colour_group_name',\n    'perceived_colour_value_name',\n    'perceived_colour_master_name',\n]\n\nfig, axes = plt.subplots(ncols=2,\n                         nrows=len(count_columns),\n                         figsize=(15, len(count_columns) * 5),\n                         sharey='row',\n                        )\n\nfor count, cc in enumerate(count_columns):\n    vc = articles[cc].value_counts() / len(articles)  * 100\n    vc = vc[:CONFIG.EXAMPLE_LIMIT]\n\n    axes[count, 0].barh(width=vc.values,\n                 y=vc.index,\n                 color='Green',\n                 )\n\n    axes[count, 1].barh(width=vc.values.cumsum(),\n                 y=vc.index,\n                 color='Green',\n                 )\n\n    axes[count, 0].set_xlim(0, 50)\n    axes[count, 1].set_xlim(0, 100)\n    axes[count, 0].set_title(f'% of total {cc}')\n    axes[count, 1].set_title(f'cumulative % {cc}')\n    \nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-17T01:45:34.246534Z","iopub.execute_input":"2022-02-17T01:45:34.247943Z","iopub.status.idle":"2022-02-17T01:45:35.85761Z","shell.execute_reply.started":"2022-02-17T01:45:34.247887Z","shell.execute_reply":"2022-02-17T01:45:35.856564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Overlaps between some of the categories","metadata":{}},{"cell_type":"code","source":"#heatmaps to examine the correlation between categories\ncombo_columns = [\n    ['product_type_name',\n     'product_group_name', ],\n    ['section_name',\n     'garment_group_name', ],\n    ['product_type_name',\n     'graphical_appearance_name', ],\n    ['product_type_name',\n     'department_name', ],\n    ['product_type_name',\n     'index_name', ],\n    ['colour_group_name',\n     'perceived_colour_value_name', ],\n    ['colour_group_name',\n     'graphical_appearance_name', ],\n    ['colour_group_name',\n     'product_group_name', ],\n\n]\n\nfor cc in combo_columns:\n    gp = articles.groupby(cc)['article_id'].count().unstack(cc[1]) / len(articles) * 100\n\n    #sort by most common entries in each category\n    gp = gp.loc[gp.sum(axis=1).sort_values(ascending=False).index.tolist(),\n                gp.sum(axis=0).sort_values(ascending=False).index.tolist()\n    ]\n    #select examples (most common)\n    gp = gp.iloc[:CONFIG.EXAMPLE_LIMIT, :CONFIG.EXAMPLE_LIMIT]\n\n    fig, axes = plt.subplots(figsize=(15, max(6, int(len(gp) / 1))))\n    sns.heatmap(gp,\n                annot=True,\n                fmt=\".1f\",\n                linewidths=1,\n                cmap='Greens')\n    plt.yticks(rotation=0)\n    plt.title(f'percentage of data by columns {cc}')\n    show_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T23:57:33.605392Z","iopub.execute_input":"2022-02-16T23:57:33.605603Z","iopub.status.idle":"2022-02-16T23:57:39.226783Z","shell.execute_reply.started":"2022-02-16T23:57:33.605577Z","shell.execute_reply":"2022-02-16T23:57:39.225981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explore Customers","metadata":{}},{"cell_type":"code","source":"print('customers data shape, columns, data types')\nprint(customers.shape)\nprint(customers.columns.tolist())\nprint(customers.dtypes)","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T23:57:39.228087Z","iopub.execute_input":"2022-02-16T23:57:39.22869Z","iopub.status.idle":"2022-02-16T23:57:39.236393Z","shell.execute_reply.started":"2022-02-16T23:57:39.228647Z","shell.execute_reply":"2022-02-16T23:57:39.235617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rename column to make it easier to understand for EDA\ncustomers = customers.rename(columns={'FN': 'fashion_news'})\n\nprint(f'there are {customers.shape[0]} rows in the customers data')\nprint(' ')\nfor c in customers.columns:\n    print(f'for {c} there are {customers[c].nunique()} unique entries and {customers[c].isna().sum()} NAN')","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T23:57:39.237538Z","iopub.execute_input":"2022-02-16T23:57:39.237762Z","iopub.status.idle":"2022-02-16T23:57:41.581522Z","shell.execute_reply.started":"2022-02-16T23:57:39.237735Z","shell.execute_reply":"2022-02-16T23:57:41.580609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Group the ages into brackets","metadata":{}},{"cell_type":"code","source":"# fill the NAN and tidy up the entries to make the data clearer to understand\ncustomers['fashion_news'] = np.where(customers['fashion_news'].isna(), 'No', 'Yes')\ncustomers['Active'] = np.where(customers['Active'].isna(), 'No', 'Yes')\ncustomers['club_member_status'] = customers['club_member_status'].fillna('No data')\ncustomers['fashion_news_frequency'] = customers['fashion_news_frequency'].fillna('No data')\n\n\n#group the age into brackets of 10 years for convenience in analysis\ncustomers['age'] = customers['age'].fillna(value=-1)\ncustomers['age_decade'] = (customers['age'] // 10) * 10\ncustomers['age_decade'] = [f'{int(x)}-{int(x)+10}' for x in customers['age_decade']]\ncustomers['age_decade'].value_counts()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-16T23:57:41.582651Z","iopub.execute_input":"2022-02-16T23:57:41.582882Z","iopub.status.idle":"2022-02-16T23:57:43.698807Z","shell.execute_reply.started":"2022-02-16T23:57:41.582854Z","shell.execute_reply":"2022-02-16T23:57:43.697991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show distribution by age bracket\nsns.countplot(x=customers['age_decade'], color='Green',\n              order=sorted(customers['age_decade'].unique()))\nplt.title('Age Distribution of Customers, (-10-0) = No Data')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T23:57:43.699877Z","iopub.execute_input":"2022-02-16T23:57:43.700119Z","iopub.status.idle":"2022-02-16T23:57:45.218467Z","shell.execute_reply.started":"2022-02-16T23:57:43.700091Z","shell.execute_reply":"2022-02-16T23:57:45.217613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=customers['age_decade'], color='Green',\n              order=sorted(customers['age_decade'].unique()),\n              hue=customers['Active'],\n              palette='tab10',\n              )\nplt.title('Age Distribution of Customers and Activity, (-10-0) = No Age Data')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T23:57:45.220117Z","iopub.execute_input":"2022-02-16T23:57:45.220413Z","iopub.status.idle":"2022-02-16T23:57:48.134962Z","shell.execute_reply.started":"2022-02-16T23:57:45.220372Z","shell.execute_reply":"2022-02-16T23:57:48.133893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=customers['age_decade'], color='Green',\n              order=sorted(customers['age_decade'].unique()),\n              hue=customers['fashion_news_frequency'],\n              palette='tab10',\n              )\nplt.title('Age Distribution of Customers and fashion_news_frequency, (-10-0) = No Age Data')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T23:57:48.136501Z","iopub.execute_input":"2022-02-16T23:57:48.137084Z","iopub.status.idle":"2022-02-16T23:57:51.681727Z","shell.execute_reply.started":"2022-02-16T23:57:48.137048Z","shell.execute_reply":"2022-02-16T23:57:51.680786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=customers['age_decade'], color='Green',\n              order=sorted(customers['age_decade'].unique()),\n              hue=customers['club_member_status'],\n              palette='tab10',\n              )\nplt.title('Age Distribution of Customers and club_member_status, (-10-0) = No Age Data')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T23:57:51.683278Z","iopub.execute_input":"2022-02-16T23:57:51.683581Z","iopub.status.idle":"2022-02-16T23:57:54.882457Z","shell.execute_reply.started":"2022-02-16T23:57:51.683541Z","shell.execute_reply":"2022-02-16T23:57:54.881752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers['transaction_count'] = customers['customer_id'].map(transactions['customer_id'].value_counts())\nage_transactions = customers.groupby(['age_decade'])['transaction_count'].agg(['count', 'sum'])\nage_transactions['transaction_per_cust'] = age_transactions['sum'] / age_transactions['count']\n\nfig,axes=plt.subplots()\nax2=axes.twinx()\naxes.bar(x=age_transactions.index,\n         height=age_transactions['sum'],\n         color='Green')\n\n\nax2.plot(age_transactions.index,\n         age_transactions['transaction_per_cust'],\n          linewidth=5,\n         color='Red'\n        )\n\nax2.set_ylim(0,)\nplt.title('# of customers (green) vs transactions per customer (red)')\naxes.set_ylabel('# customers')\nax2.set_ylabel('transactions per customer')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T23:57:54.883493Z","iopub.execute_input":"2022-02-16T23:57:54.883699Z","iopub.status.idle":"2022-02-16T23:58:05.216996Z","shell.execute_reply.started":"2022-02-16T23:57:54.883674Z","shell.execute_reply":"2022-02-16T23:58:05.216416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combo_columns = [\n    'fashion_news',\n    'Active',\n    'club_member_status',\n    'fashion_news_frequency',\n]\n\nfor cc in itertools.combinations(combo_columns, 2):\n    cc = list(cc)\n    gp = customers.groupby(cc)['customer_id'].count().unstack(cc[1]) / len(customers) * 100\n\n    #sort by most common entries in each category\n    gp = gp.loc[gp.sum(axis=1).sort_values(ascending=False).index.tolist(),\n                gp.sum(axis=0).sort_values(ascending=False).index.tolist()\n    ]\n\n    gp = gp.iloc[:CONFIG.EXAMPLE_LIMIT, :CONFIG.EXAMPLE_LIMIT]\n\n    fig, axes = plt.subplots(figsize=(12, max(6, int(len(gp) / 2))))\n    sns.heatmap(gp,\n                annot=True,\n                fmt=\".1f\",\n                linewidths=1,\n                cmap='Greens')\n    plt.yticks(rotation=0)\n    plt.title(f'percentage of data by columns {cc}')\n    show_clear_plt()\n    ","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T23:58:05.217869Z","iopub.execute_input":"2022-02-16T23:58:05.218223Z","iopub.status.idle":"2022-02-16T23:58:09.420372Z","shell.execute_reply.started":"2022-02-16T23:58:05.218195Z","shell.execute_reply":"2022-02-16T23:58:09.419604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Analyse postcodes - do they tell us anything useful?","metadata":{}},{"cell_type":"code","source":"postal_codes = customers.groupby(['postal_code'])['customer_id'].count()\nprint('customers per postcode')\nprint(postal_codes.shape)\npostal_codes.sort_values(ascending=False).head(10)","metadata":{"collapsed":false,"pycharm":{"name":"#%% \n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T23:58:09.421615Z","iopub.execute_input":"2022-02-16T23:58:09.422087Z","iopub.status.idle":"2022-02-16T23:58:11.075651Z","shell.execute_reply.started":"2022-02-16T23:58:09.422043Z","shell.execute_reply":"2022-02-16T23:58:11.074872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# most postcodes have only 1 customer, so this may need more work to see if there could\n# be anything useful\nsns.histplot(postal_codes.values[postal_codes<300], discrete=True)\nplt.title('Customers per postcode, for values < 300 (clipped outlier)')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T23:58:11.076874Z","iopub.execute_input":"2022-02-16T23:58:11.077128Z","iopub.status.idle":"2022-02-16T23:58:12.337738Z","shell.execute_reply.started":"2022-02-16T23:58:11.0771Z","shell.execute_reply":"2022-02-16T23:58:12.336959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Do customers have >1 postal code?","metadata":{}},{"cell_type":"code","source":"max_codes = customers.groupby(['customer_id'])['postal_code'].nunique().max()\nprint(f'max postal codes per customer is {max_codes}')","metadata":{"execution":{"iopub.status.busy":"2022-02-16T23:58:12.338985Z","iopub.execute_input":"2022-02-16T23:58:12.339206Z","iopub.status.idle":"2022-02-16T23:58:15.418482Z","shell.execute_reply.started":"2022-02-16T23:58:12.339179Z","shell.execute_reply":"2022-02-16T23:58:15.417346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explore Transactions","metadata":{}},{"cell_type":"code","source":"print('transactions data shape, columns, data types')\nprint(transactions.shape)\nprint(transactions.columns.tolist())\nprint(transactions.dtypes)","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T23:58:15.419776Z","iopub.execute_input":"2022-02-16T23:58:15.420084Z","iopub.status.idle":"2022-02-16T23:58:15.427493Z","shell.execute_reply.started":"2022-02-16T23:58:15.42005Z","shell.execute_reply":"2022-02-16T23:58:15.426598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'there are {transactions.shape[0]} rows in the transactions data')\nprint(' ')\nfor c in transactions.columns:\n    print(f'for {c} there are {transactions[c].nunique()} unique entries and {transactions[c].isna().sum()} NAN')","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T23:58:15.428486Z","iopub.execute_input":"2022-02-16T23:58:15.428691Z","iopub.status.idle":"2022-02-16T23:58:28.370284Z","shell.execute_reply.started":"2022-02-16T23:58:15.428665Z","shell.execute_reply":"2022-02-16T23:58:28.369247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Some data processing","metadata":{}},{"cell_type":"code","source":"# make naming easier to remember for EDA purposes\nchannel_dict = {\n    1: 'store',\n    2: 'online',\n}\ntransactions['sales_channel_name'] = transactions['sales_channel_id'].map(channel_dict)\n\n# add weeks, days, etc\ntransactions['quarter'] = transactions['t_dat'].dt.quarter\ntransactions['month'] = transactions['t_dat'].dt.month\ntransactions['week'] = transactions['t_dat'].dt.isocalendar().week\ntransactions['weekday'] = transactions['t_dat'].dt.weekday\ntransactions['day_name'] = transactions['t_dat'].dt.day_name()\n\n# cyclic encode - for the week - for demo later\ndef cyclic_encode(df, column):\n    df[f'{column}_sin'] = np.sin(2 * np.pi * df[column] / df[column].max())\n    df[f'{column}_cos'] = np.cos(2 * np.pi * df[column] / df[column].max())\n    return df\n\nencode_cols = [\n    'week',\n]\n\nfor ec in encode_cols:\n    transactions = cyclic_encode(transactions, ec)\n\n# display example\n\ndaily_transactions = transactions.groupby(['t_dat'])[['week_sin', 'week_cos']].mean()\n\nfor c in daily_transactions.columns:\n    sns.lineplot(x=daily_transactions[c].resample('w').mean().index,\n                 y=daily_transactions[c].resample('w').mean().values,\n                 linewidth=4)\nplt.title('Week cyclic encoding')\nplt.legend(daily_transactions.columns.tolist())\nplt.ylabel('Cyclic encoding')\nshow_clear_plt()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T00:12:19.24864Z","iopub.execute_input":"2022-02-17T00:12:19.249246Z","iopub.status.idle":"2022-02-17T00:13:05.102108Z","shell.execute_reply.started":"2022-02-17T00:12:19.249195Z","shell.execute_reply":"2022-02-17T00:13:05.100734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Pareto - Customers / Products","metadata":{}},{"cell_type":"markdown","source":"20% of most frequent customers are around 65-70% of transactions\n\n20% of higest selling articles are around 80% of transactions","metadata":{}},{"cell_type":"code","source":"cust_pareto = transactions.groupby(['customer_id'])['customer_id'].count()\nprod_pareto = transactions.groupby(['article_id'])['article_id'].count()\n\nfig,axes=plt.subplots(figsize=(14,7), ncols=2)\n\ncust_pareto = cust_pareto.sort_values(ascending=False)\naxes[0].scatter(y=cust_pareto.cumsum() / cust_pareto.sum() * 100,\n           x=np.ones(cust_pareto.shape).cumsum() / len(cust_pareto) * 100,\n               color='Red')\naxes[0].set_xlabel('% of customers')\naxes[0].set_ylabel('% of transactions')\naxes[0].set_title('Customer Pareto')\n\n\nprod_pareto = prod_pareto.sort_values(ascending=False)\naxes[1].scatter(y=prod_pareto.cumsum() / prod_pareto.sum() * 100,\n           x=np.ones(prod_pareto.shape).cumsum() / len(prod_pareto) * 100,\n               color='Red')\naxes[1].set_xlabel('% of Article')\naxes[1].set_ylabel('% of transactions')\naxes[1].set_title('Article Pareto')\n\nshow_clear_plt()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T00:07:43.377221Z","iopub.execute_input":"2022-02-17T00:07:43.377497Z","iopub.status.idle":"2022-02-17T00:08:03.990269Z","shell.execute_reply.started":"2022-02-17T00:07:43.377469Z","shell.execute_reply":"2022-02-17T00:08:03.989435Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can make 12 predictions per customer, what is the cumulative percentage of the top 12 articles?\n\nIt appears to be around 1% of total transactions (though as we are predicting at a customer level, think this is not exactly equivalent to filling in the submission file with a top 12)","metadata":{}},{"cell_type":"code","source":"plt.barh(width=(prod_pareto.cumsum() / prod_pareto.sum())[:12] * 100,\n        y=[str(x) for x in prod_pareto.index[:12]],\n        color='Green')\n\nplt.title('Cumulative % of top 12 articles')\nshow_clear_plt()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T00:10:38.325429Z","iopub.execute_input":"2022-02-17T00:10:38.326348Z","iopub.status.idle":"2022-02-17T00:10:38.66779Z","shell.execute_reply.started":"2022-02-17T00:10:38.326285Z","shell.execute_reply":"2022-02-17T00:10:38.66698Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Sum of transaction distributions by week, month, day (name)","metadata":{}},{"cell_type":"code","source":"group_cols = [\n    't_dat',  # day\n    'week',\n    'month',\n    'day_name'\n]\n\nfor g_col in group_cols:\n    cust_t_count = transactions.groupby(['sales_channel_name',\n                                         g_col])['customer_id'].count().unstack('sales_channel_name').sort_index().fillna(\n        value=0)\n\n    for c in cust_t_count.columns:\n        sns.kdeplot(cust_t_count[c],\n                    linewidth=3)\n    plt.title(f'Distribution # of sum of transactions per {g_col} by channel')\n    plt.legend(cust_t_count.columns.tolist())\n    plt.xlabel(f'{g_col} transaction count')\n    show_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-17T00:13:09.132568Z","iopub.execute_input":"2022-02-17T00:13:09.133169Z","iopub.status.idle":"2022-02-17T00:13:44.134733Z","shell.execute_reply.started":"2022-02-17T00:13:09.133116Z","shell.execute_reply":"2022-02-17T00:13:44.133876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Online vs Store by Weekday (name)\n\nWe can see that Sunday/Monday are strongest for online % share, while Friday/Saturday are strongest for store share\n\nOnline absolute numbers are more steady over the week","metadata":{}},{"cell_type":"code","source":"# store vs online by weekday\ndaily_transactions = transactions.groupby(['sales_channel_name',\n                                           'day_name'])['customer_id'].count().unstack(\n    'sales_channel_name').sort_index().fillna(value=0)\norder = ['Monday', 'Tuesday', 'Wednesday', 'Thursday', 'Friday', 'Saturday',\n         'Sunday']\n\ndaily_transactions.loc[order].plot(kind='barh', stacked=True)\nplt.title('Total Transactions by Weekday')\nshow_clear_plt()\n\ndaily_transactions = daily_transactions / daily_transactions.sum(axis=1).values.reshape(-1, 1)\ndaily_transactions.loc[order].plot(kind='barh', stacked=True, width=0.8)\nplt.title('% of Weekday Total Transactions')\nshow_clear_plt()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T00:13:44.136607Z","iopub.execute_input":"2022-02-17T00:13:44.137346Z","iopub.status.idle":"2022-02-17T00:13:55.23363Z","shell.execute_reply.started":"2022-02-17T00:13:44.1373Z","shell.execute_reply":"2022-02-17T00:13:55.232939Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Transaction trend over time","metadata":{}},{"cell_type":"code","source":"#weekly transactions by channel\ndaily_transactions = transactions.groupby(['sales_channel_name',\n                                           't_dat'])['customer_id'].count().unstack(\n    'sales_channel_name').sort_index().fillna(value=0)\n\nfor c in daily_transactions.columns:\n    sns.lineplot(x=daily_transactions[c].resample('w').sum().index,\n                 y=daily_transactions[c].resample('w').sum().values,\n                 linewidth=4)\nplt.title('Weekly transaction volumes over time by channel')\nplt.legend(cust_t_count.columns.tolist())\nplt.ylabel('Weekly transaction count')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-17T00:13:55.234816Z","iopub.execute_input":"2022-02-17T00:13:55.235064Z","iopub.status.idle":"2022-02-17T00:14:04.051184Z","shell.execute_reply.started":"2022-02-17T00:13:55.23504Z","shell.execute_reply":"2022-02-17T00:14:04.050346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for c in daily_transactions.columns:\n    sns.lineplot(x=daily_transactions[c].resample('M').sum().index,\n                 y=daily_transactions[c].resample('M').sum().values,\n                 linewidth=4)\nplt.title('Monthly transaction volumes over time by channel')\nplt.legend(cust_t_count.columns.tolist())\nplt.ylabel('Monthly transaction count')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-17T00:14:04.053442Z","iopub.execute_input":"2022-02-17T00:14:04.05389Z","iopub.status.idle":"2022-02-17T00:14:04.65323Z","shell.execute_reply.started":"2022-02-17T00:14:04.053833Z","shell.execute_reply":"2022-02-17T00:14:04.652427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Transactions per Customer","metadata":{}},{"cell_type":"code","source":"cust_t_count = transactions['customer_id'].value_counts()\nsns.histplot(cust_t_count, discrete=True)\nplt.title('Transactions per Customer in Transaction Data, x-axis clipped')\n# ignoring outliers\nplt.xlim(0, 40)\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-17T00:14:04.654614Z","iopub.execute_input":"2022-02-17T00:14:04.654816Z","iopub.status.idle":"2022-02-17T00:14:17.856118Z","shell.execute_reply.started":"2022-02-17T00:14:04.654791Z","shell.execute_reply":"2022-02-17T00:14:17.855298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Order sizes (assume 1 order = all items purchased by a single customer in 1 day)","metadata":{}},{"cell_type":"code","source":"cust_day_group = transactions.groupby(['customer_id', 't_dat'], as_index=False)['article_id'].count()\nprint(cust_day_group.shape, transactions.shape)\nsns.histplot(cust_day_group['article_id'], discrete=True,\n            shrink=0.9)\nplt.title('Order Sizes (# articles by customer & day), x-axis clipped')\n# ignoring outlier\nplt.xlim(0, 15)\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-17T00:15:57.230649Z","iopub.execute_input":"2022-02-17T00:15:57.230986Z","iopub.status.idle":"2022-02-17T00:16:29.901634Z","shell.execute_reply.started":"2022-02-17T00:15:57.230948Z","shell.execute_reply":"2022-02-17T00:16:29.900921Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Mix of store / online at a customer level","metadata":{}},{"cell_type":"code","source":"# split by channel, by customer\n# relatively few customers have a mix of online and store shopping\ncust_t_count = transactions.groupby(['sales_channel_name',\n                                     'customer_id', ])['customer_id'].count().unstack(\n    'sales_channel_name').sort_index().fillna(value=0)\n\ncust_t_count['pc_online'] = cust_t_count['online'] / cust_t_count[['online', 'store']].sum(axis=1)\nsns.histplot(cust_t_count['pc_online'], \n             binwidth=0.1,\n            shrink=0.9)\nplt.title('Distribution of customer % purchase online')\nplt.xlabel('Distribution of customer % purchase online')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-17T00:14:44.619772Z","iopub.execute_input":"2022-02-17T00:14:44.62008Z","iopub.status.idle":"2022-02-17T00:15:07.867878Z","shell.execute_reply.started":"2022-02-17T00:14:44.620049Z","shell.execute_reply":"2022-02-17T00:15:07.866889Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if we look at customers with >2 items there is more of a mix but still 100% online or store in many cases\nsns.histplot(cust_t_count['pc_online'][cust_t_count[['online', 'store']].sum(axis=1)>2],\n            binwidth=0.1,\n            shrink=0.9)\nplt.title('Distribution of customer % purchase online for customers with >2 items')\nplt.xlabel('Distribution of customer % purchase online for customers with >2 items')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-17T00:15:07.869183Z","iopub.execute_input":"2022-02-17T00:15:07.869375Z","iopub.status.idle":"2022-02-17T00:15:09.217967Z","shell.execute_reply.started":"2022-02-17T00:15:07.869351Z","shell.execute_reply":"2022-02-17T00:15:09.216961Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Popular Articles - transactions split by Channel","metadata":{}},{"cell_type":"code","source":"# split by channel, by article\n# is there a difference in most popular articles?\n\narticle_dict = dict(zip(articles['article_id'],\n                        articles['prod_name']))\narticle_t_count = transactions.groupby(['sales_channel_name',\n                                        'article_id', ])['article_id'].count().unstack(\n    'sales_channel_name').sort_index().fillna(value=0)\n\narticle_t_count['total_transactions'] = article_t_count[['online', 'store']].sum(axis=1)\narticle_t_count = article_t_count.sort_values('total_transactions', axis=0, ascending=False)\n\narticle_t_count = article_t_count.iloc[:CONFIG.EXAMPLE_LIMIT]\narticle_t_count.index = [f'{x} - {article_dict[x]}' for x in article_t_count.index]\n\nfig, axes = plt.subplots(figsize=(12, 7))\narticle_t_count[['online', 'store']].plot(kind='barh', stacked=True,\n                                          width=0.8,\n                                          ax=axes)\nplt.title('Popular articles - Online / Store transactions')\nplt.xlabel('transacton counts')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-17T00:15:09.219177Z","iopub.execute_input":"2022-02-17T00:15:09.219401Z","iopub.status.idle":"2022-02-17T00:15:14.483573Z","shell.execute_reply.started":"2022-02-17T00:15:09.219371Z","shell.execute_reply":"2022-02-17T00:15:14.48277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Popular Items - mix by Age Group","metadata":{}},{"cell_type":"code","source":"age_dict = dict(zip(\n    customers['customer_id'],\n    customers['age_decade']\n))\ntransactions['age_decade'] = transactions['customer_id'].map(age_dict)\nprint(f'# rows missing customer age data = {transactions[\"age_decade\"].isna().sum()}')\n\narticle_t_count = transactions.groupby(['age_decade',\n                                        'article_id', ])['article_id'].count().unstack(\n    'age_decade').sort_index().fillna(value=0)\narticle_t_count['total_transactions'] = article_t_count.sum(axis=1)\narticle_t_count = article_t_count.sort_values('total_transactions', axis=0, ascending=False)\narticle_t_count = article_t_count.drop('total_transactions', axis=1)\n\narticle_t_count = article_t_count.iloc[:CONFIG.EXAMPLE_LIMIT]\narticle_t_count.index = [f'{x} - {article_dict[x]}' for x in article_t_count.index]\n\nfig, axes = plt.subplots(figsize=(12, 7))\narticle_t_count.plot(kind='barh', stacked=True,\n                                          width=0.8,\n                                          ax=axes)\nplt.title('Popular articles - By Customer Age Bracket')\nplt.xlabel('transaction counts')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-17T00:15:14.485736Z","iopub.execute_input":"2022-02-17T00:15:14.48598Z","iopub.status.idle":"2022-02-17T00:15:37.063968Z","shell.execute_reply.started":"2022-02-17T00:15:14.485949Z","shell.execute_reply":"2022-02-17T00:15:37.063034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#there is some variation in the age profiles of top sellers\narticle_t_count = article_t_count / article_t_count.sum(axis=1).values.reshape(-1, 1)\nfig, axes = plt.subplots(figsize=(12, 7))\narticle_t_count.plot(kind='barh', stacked=True,\n                                          width=0.8,\n                                          ax=axes)\nplt.title('Popular articles - By Customer Age Bracket')\nplt.xlabel('transaction % mix')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-17T00:15:37.065127Z","iopub.execute_input":"2022-02-17T00:15:37.06533Z","iopub.status.idle":"2022-02-17T00:15:38.051849Z","shell.execute_reply.started":"2022-02-17T00:15:37.065305Z","shell.execute_reply":"2022-02-17T00:15:38.051019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"What does it look like if we take a top 5 for 3 different age brackets, rather than a top n overall?","metadata":{}},{"cell_type":"code","source":"brackets = ['20-30','40-50','60-70']\ntop_items = []\narticle_t_count = transactions.groupby(['age_decade',\n                                        'article_id', ])['article_id'].count().unstack(\n    'age_decade').sort_index().fillna(value=0)\nfor b in brackets:\n    temp = article_t_count.loc[article_t_count.sort_values(b, ascending=False).index[:5].tolist()]\n\n    temp.index = [f'{x} - {article_dict[x]}' for x in temp.index]\n\n    fig, axes = plt.subplots(figsize=(12, 7))\n    temp.plot(kind='barh', stacked=True,\n                                              width=0.8,\n                                              ax=axes)\n    plt.title(f'Most Popular articles - For Customer Age Bracket {b}')\n    plt.xlabel('transaction counts')\n    show_clear_plt()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T00:22:46.911412Z","iopub.execute_input":"2022-02-17T00:22:46.911727Z","iopub.status.idle":"2022-02-17T00:22:56.683349Z","shell.execute_reply.started":"2022-02-17T00:22:46.911696Z","shell.execute_reply":"2022-02-17T00:22:56.682393Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Do customers buy the same thing more than once (on different days)?","metadata":{}},{"cell_type":"code","source":"# count of transactions per article, for articles seen in transaction data\narticle_freq = transactions['article_id'].value_counts().sort_values(ascending=False)\n\n# do customers buy the same thing more than once (on different days)?\nprod_customers = transactions.groupby(['customer_id',\n                                       'article_id', 't_dat'], as_index=False)['t_dat'].count().groupby(['customer_id',\n                                                                                                         'article_id', ],\n                                                                                                        as_index=False)[\n    't_dat'].count()\nprod_customers_ = prod_customers[prod_customers['t_dat'] > 1.0].sort_values('t_dat', ascending=False).reset_index(\n    drop=True)\nprint(f'{len(prod_customers_)} instances found')\nprod_customers_.head(10)","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-16T12:20:38.152775Z","iopub.execute_input":"2022-02-16T12:20:38.153009Z","iopub.status.idle":"2022-02-16T12:21:51.696446Z","shell.execute_reply.started":"2022-02-16T12:20:38.152981Z","shell.execute_reply":"2022-02-16T12:21:51.695468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Double check the data - yes, this customer purchased the same item on different dates\n\nSome of these examples look a bit odd at a quick glance - may need further investigation","metadata":{}},{"cell_type":"code","source":"print(article_dict[prod_customers_.loc[0, 'article_id']])\ntransactions[(transactions['customer_id'] == prod_customers_.loc[0, 'customer_id']) &\n             (transactions['article_id'] == prod_customers_.loc[0, 'article_id'])]","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-16T12:21:51.702911Z","iopub.execute_input":"2022-02-16T12:21:51.703231Z","iopub.status.idle":"2022-02-16T12:21:56.87043Z","shell.execute_reply.started":"2022-02-16T12:21:51.703197Z","shell.execute_reply":"2022-02-16T12:21:56.869666Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It is definitely possible for a customer to buy the same item on different dates","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(figsize=(15, 7))\nsns.histplot(prod_customers['t_dat'], discrete=True)\nplt.title('# of instances when customer when customer purchased same item on multiple days (1 = no repeat purchase)')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-16T12:21:56.871672Z","iopub.execute_input":"2022-02-16T12:21:56.871891Z","iopub.status.idle":"2022-02-16T12:22:17.075776Z","shell.execute_reply.started":"2022-02-16T12:21:56.871865Z","shell.execute_reply":"2022-02-16T12:22:17.074851Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Sales trends for popular items - It is clear that some items are popular only during specific periods\n\nEither seasonality, or introduced / removed from range in some cases?\n\n% of sales online vs store shows some significant differences by article","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=CONFIG.EXAMPLE_LIMIT,\n                         ncols=1,\n                         figsize=(12, 6*CONFIG.EXAMPLE_LIMIT),\n                         sharex=True)\n\nfor count, a in enumerate(article_freq.index[:CONFIG.EXAMPLE_LIMIT]):\n    cust_t_count = transactions[transactions['article_id'] == a].groupby(['sales_channel_name',\n                                                                          't_dat'])['customer_id'].count().unstack(\n        'sales_channel_name').fillna(value=0).sort_index()\n\n\n    for c in cust_t_count.columns:\n        temp = cust_t_count[c].resample('w').sum()\n        sns.lineplot(x=temp.index,\n                     y=temp.values,\n                     linewidth=5,\n                     ax=axes[count])\n    axes[count].set_title(f'{article_dict[a]} art_ID {a} Weekly transaction volumes over time by channel')\n    axes[count].legend(cust_t_count.columns.tolist())\n    axes[count].set_ylabel('Weekly transaction count')\n    axes[count].set_xlabel('Date')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-16T12:22:17.077272Z","iopub.execute_input":"2022-02-16T12:22:17.077535Z","iopub.status.idle":"2022-02-16T12:22:21.855303Z","shell.execute_reply.started":"2022-02-16T12:22:17.077501Z","shell.execute_reply":"2022-02-16T12:22:21.854479Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del prod_customers, prod_customers_\ngc.collect()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T12:22:21.856697Z","iopub.execute_input":"2022-02-16T12:22:21.857651Z","iopub.status.idle":"2022-02-16T12:22:22.17132Z","shell.execute_reply.started":"2022-02-16T12:22:21.857591Z","shell.execute_reply":"2022-02-16T12:22:22.170488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transaction Locations","metadata":{}},{"cell_type":"markdown","source":"What is the impact of Location on store/online?\n\nThere is one postcode with a huge number of transactions, primarily Store (not online)\n\nIs this some form of NAN? Needs exploration.","metadata":{}},{"cell_type":"code","source":"transactions['postal_code'] = transactions['customer_id'].map(dict(zip(customers['customer_id'],\n                                                                       customers['postal_code'])))\n\ntransactions_locations = transactions.groupby(['postal_code','sales_channel_name'])['customer_id'].count().unstack('sales_channel_name').fillna(value=0)\ntransactions_locations = transactions_locations.loc[transactions_locations.sum(axis=1).sort_values(ascending=False).index.tolist()].iloc[:CONFIG.EXAMPLE_LIMIT,:]\nfig, axes = plt.subplots(figsize=(20, 7))\ntransactions_locations.plot(kind='barh', stacked=True,\n                                          width=0.8,\n                                          ax=axes)\nplt.title('Common Postcodes - Online / Store transactions')\nplt.xlabel('transacton counts')\nshow_clear_plt()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T12:22:22.172573Z","iopub.execute_input":"2022-02-16T12:22:22.172896Z","iopub.status.idle":"2022-02-16T12:22:57.173898Z","shell.execute_reply.started":"2022-02-16T12:22:22.172862Z","shell.execute_reply":"2022-02-16T12:22:57.173324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(figsize=(20, 7))\ntransactions_locations = transactions_locations / transactions_locations.sum(axis=1).values.reshape(-1,1)\ntransactions_locations.plot(kind='barh', stacked=True,\n                                          width=0.8,\n                                          ax=axes)\nplt.title('Common Postcodes - % Online / Store transactions')\nplt.xlabel('transacton counts')\nshow_clear_plt()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T12:22:57.174902Z","iopub.execute_input":"2022-02-16T12:22:57.175501Z","iopub.status.idle":"2022-02-16T12:22:57.701322Z","shell.execute_reply.started":"2022-02-16T12:22:57.175451Z","shell.execute_reply":"2022-02-16T12:22:57.700467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transaction Price","metadata":{}},{"cell_type":"code","source":"sns.histplot(transactions['price'], \n             binwidth=0.01)\nplt.title('Distribution of Price - All Transactions')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T12:22:57.702754Z","iopub.execute_input":"2022-02-16T12:22:57.703804Z","iopub.status.idle":"2022-02-16T12:23:29.939149Z","shell.execute_reply.started":"2022-02-16T12:22:57.70375Z","shell.execute_reply":"2022-02-16T12:23:29.93826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(transactions['price'][transactions['sales_channel_name'] == 'online'], binwidth=0.01,\n        color='Blue')\nsns.histplot(transactions['price'][transactions['sales_channel_name'] == 'store'], binwidth=0.01,\n             color='Orange')\nplt.legend(['online', 'store'])\nplt.title('Distribution of Price by channel')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T12:23:29.94065Z","iopub.execute_input":"2022-02-16T12:23:29.940873Z","iopub.status.idle":"2022-02-16T12:24:09.760076Z","shell.execute_reply.started":"2022-02-16T12:23:29.940848Z","shell.execute_reply":"2022-02-16T12:24:09.759453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Not surprisingly, the articles with the highest mean prices are generally not high volume items","metadata":{}},{"cell_type":"code","source":"#not surprisingly, the articles with the highest mean prices are generally not\n#high volume items\narticles_prices_volumes = transactions.groupby(['article_id'])['price'].agg(['mean', 'count',\n                                                                             'max', 'min', 'std'])\nplt.scatter(x=articles_prices_volumes['count'],\n            y=articles_prices_volumes['mean'],\n            color='Red',\n            s=2)\nplt.title('Article transaction count vs mean price')\nplt.xlabel('article transaction count')\nplt.ylabel('article price mean')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-16T12:24:09.761495Z","iopub.execute_input":"2022-02-16T12:24:09.761927Z","iopub.status.idle":"2022-02-16T12:24:13.247008Z","shell.execute_reply.started":"2022-02-16T12:24:09.761895Z","shell.execute_reply":"2022-02-16T12:24:13.246313Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Inspect most expensive products","metadata":{}},{"cell_type":"code","source":"articles_prices_volumes['name'] = articles_prices_volumes.index.map(article_dict)\narticles_prices_volumes.sort_values('mean', ascending=False).head(10)","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-16T12:24:13.248262Z","iopub.execute_input":"2022-02-16T12:24:13.248927Z","iopub.status.idle":"2022-02-16T12:24:13.358433Z","shell.execute_reply.started":"2022-02-16T12:24:13.248875Z","shell.execute_reply":"2022-02-16T12:24:13.357468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Inspect cheapest products","metadata":{}},{"cell_type":"code","source":"articles_prices_volumes.sort_values('mean', ascending=True).head(10)","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-16T12:43:02.277719Z","iopub.execute_input":"2022-02-16T12:43:02.278053Z","iopub.status.idle":"2022-02-16T12:43:02.323833Z","shell.execute_reply.started":"2022-02-16T12:43:02.278023Z","shell.execute_reply":"2022-02-16T12:43:02.322971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Inspect largest standard deviations in pricing","metadata":{}},{"cell_type":"code","source":"articles_prices_volumes.sort_values('std', ascending=False).head(10)","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-16T12:24:13.404346Z","iopub.execute_input":"2022-02-16T12:24:13.405049Z","iopub.status.idle":"2022-02-16T12:24:13.444569Z","shell.execute_reply.started":"2022-02-16T12:24:13.404998Z","shell.execute_reply":"2022-02-16T12:24:13.443887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looking at article transaction price / article mean transaction price, there are some outliers far to the right","metadata":{}},{"cell_type":"code","source":"transactions['article_mean_price'] = transactions['article_id'].map(articles_prices_volumes['mean'])\ntransactions['article_transaction_price_pc_mean'] = transactions['price'] / transactions['article_mean_price']\nsns.histplot(transactions['article_transaction_price_pc_mean'],\n            binwidth=0.05)\nplt.title('Distribution of transactions - price as a % of article mean transaction price')\nplt.xlim(0, 4)\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-16T12:24:13.445585Z","iopub.execute_input":"2022-02-16T12:24:13.44638Z","iopub.status.idle":"2022-02-16T12:24:47.936932Z","shell.execute_reply.started":"2022-02-16T12:24:13.446344Z","shell.execute_reply":"2022-02-16T12:24:47.936126Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Look at article transaction price / article mean transaction price by channel","metadata":{}},{"cell_type":"code","source":"sns.histplot(transactions['article_transaction_price_pc_mean'][transactions['sales_channel_name'] == 'online'], binwidth=0.05,\n        color='Blue')\nsns.histplot(transactions['article_transaction_price_pc_mean'][transactions['sales_channel_name'] == 'store'], binwidth=0.05,\n             color='Orange')\nplt.legend(['online', 'store'])\nplt.title('Distribution of Transaction Price vs Article Mean Transaction Price by channel')\nshow_clear_plt()","metadata":{"_kg_hide-input":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Look at sales vs price trends for some popular items","metadata":{}},{"cell_type":"code","source":"for a in article_freq.index[:CONFIG.EXAMPLE_LIMIT]:\n    cust_t_count = transactions[transactions['article_id'] == a].groupby([\n        't_dat'])['price'].agg(['mean', 'count'])\n    # print(cust_t_count)\n    fig, axes = plt.subplots(figsize=(12, 6))\n    ax2 = axes.twinx()\n\n    temp = cust_t_count['mean'].resample('w').mean()\n    sns.lineplot(x=temp.index,\n                 y=temp.values,\n                 linewidth=2,\n                 ax=ax2,\n                 color='Red')\n\n    temp = cust_t_count['count'].resample('w').sum()\n    sns.lineplot(x=temp.index,\n                 y=temp.values,\n                 linewidth=4,\n                 ax=axes,\n                 color='Black')\n\n    axes.set_ylabel('Weekly sales')\n    ax2.set_ylim(0, )\n    ax2.set_ylabel('Weekly mean Price')\n    plt.title(f'{article_dict[a]} Weekly price (red) vs volume (black)')\n    plt.legend(cust_t_count.columns.tolist())\n    plt.xlabel('Date')\n    show_clear_plt()       ","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-16T12:24:47.938173Z","iopub.execute_input":"2022-02-16T12:24:47.938394Z","iopub.status.idle":"2022-02-16T12:25:06.153219Z","shell.execute_reply.started":"2022-02-16T12:24:47.938368Z","shell.execute_reply":"2022-02-16T12:25:06.152506Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transactions vs Article Features","metadata":{}},{"cell_type":"code","source":"article_sales = transactions.groupby('article_id')['t_dat'].count()\narticles['sales'] = articles['article_id'].map(article_sales).fillna(value=0)\nprint(f'percent of articles with no transactions {sum(articles[\"sales\"] == 0.0) / len(articles)}')","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-16T12:25:06.155146Z","iopub.execute_input":"2022-02-16T12:25:06.155504Z","iopub.status.idle":"2022-02-16T12:25:07.479905Z","shell.execute_reply.started":"2022-02-16T12:25:06.155451Z","shell.execute_reply":"2022-02-16T12:25:07.478887Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Contributions to total transactions by category","metadata":{}},{"cell_type":"code","source":"count_columns = [\n    'product_type_name',\n    'product_group_name',\n    'department_name',\n    'index_name',\n    'index_group_name',\n    'section_name',\n    'garment_group_name',\n]\n\nfig, axes = plt.subplots(nrows=len(count_columns),\n                         figsize=(12, 6 * len(count_columns)),\n                      )\n\nfor count, cc in enumerate(count_columns):\n    vc = articles.groupby([cc])['sales'].sum() / articles['sales'].sum() * 100\n    vc = vc.sort_values(ascending=False)[:CONFIG.EXAMPLE_LIMIT]\n    \n    axes[count].barh(width=vc.values,\n             y=vc.index,\n             color='Green')\n    axes[count].set_title(f'percentage transaction counts {cc}')\n    axes[count].set_xlabel('percent of total transactions')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-16T12:44:58.114806Z","iopub.execute_input":"2022-02-16T12:44:58.115627Z","iopub.status.idle":"2022-02-16T12:45:00.607685Z","shell.execute_reply.started":"2022-02-16T12:44:58.115582Z","shell.execute_reply":"2022-02-16T12:45:00.606587Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Contributions to total transactions - categories more related to the colour / design","metadata":{}},{"cell_type":"code","source":"count_columns = [\n    'graphical_appearance_name',\n    'colour_group_name',\n    'perceived_colour_value_name',\n    'perceived_colour_master_name',\n]\nfig, axes = plt.subplots(nrows=len(count_columns),\n                         figsize=(12, 6 * len(count_columns)),\n                        )\n\nfor count, cc in enumerate(count_columns):\n    vc = articles.groupby([cc])['sales'].sum() / articles['sales'].sum() * 100\n    vc = vc.sort_values(ascending=False)[:CONFIG.EXAMPLE_LIMIT]\n\n    axes[count].barh(width=vc.values,\n                     y=vc.index,\n                     color='Green')\n    axes[count].set_title(f'percentage transaction counts {cc}')\n    axes[count].set_xlabel('percent of total transactions')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T12:45:35.548192Z","iopub.execute_input":"2022-02-16T12:45:35.548718Z","iopub.status.idle":"2022-02-16T12:45:36.827316Z","shell.execute_reply.started":"2022-02-16T12:45:35.548678Z","shell.execute_reply":"2022-02-16T12:45:36.826497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Seasonality by article category (examples - not all categories)\n\nWe can see types of clothing and even some colours (maybe white, pink, green) could be seasonal\n\nThere may also be some longer term trends, in colour for example Blue looks less popular in 2020 while Green looks more popular in 2020. It's not clear without further analysis if this is an overall trend or maybe driven by some specific top-selling products.","metadata":{}},{"cell_type":"code","source":"seasonality_example_columns = [\n    'product_group_name',\n    'perceived_colour_master_name',\n    'garment_group_name',\n]\n\nfor cc in seasonality_example_columns:\n    vc = articles.groupby([cc])['sales'].sum() / articles['sales'].sum()\n    vc = vc.sort_values(ascending=False)[:CONFIG.EXAMPLE_LIMIT]\n\n    transactions['article_cat'] = transactions['article_id'].map(dict(zip(articles['article_id'],\n                                                                          articles[cc])))\n\n    for v in vc.index.tolist():\n        cust_t_count = transactions[transactions['article_cat'] == v].groupby(['t_dat'])[\n            'customer_id'].count().sort_index()\n\n        fig, axes = plt.subplots(figsize=(12, 6))\n\n        temp = cust_t_count.resample('w').sum()\n        sns.lineplot(x=temp.index,\n                     y=temp.values,\n                     linewidth=5,\n                     color='Black')\n        plt.title(f'Category {cc} - {v} Weekly transaction volumes')\n        plt.ylabel('Weekly transaction count')\n        plt.xlabel('Date')\n        show_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-16T12:25:10.641424Z","iopub.execute_input":"2022-02-16T12:25:10.642081Z","iopub.status.idle":"2022-02-16T12:28:54.200897Z","shell.execute_reply.started":"2022-02-16T12:25:10.642029Z","shell.execute_reply":"2022-02-16T12:28:54.199978Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Displaying seasonality using cyclic feature with 2 examples","metadata":{}},{"cell_type":"code","source":"transactions['article_cat'] = transactions['article_id'].map(dict(zip(articles['article_id'],\n                                                                          articles['product_type_name'])))\n# first example\nexample = 'Sweater'\ncust_t_count = transactions[transactions['article_cat'] == example].groupby(['week',\n                                                                             'week_sin',\n                                                                             'week_cos'], as_index=False)[\n            'customer_id'].count().sort_index()\nplt.scatter(x=cust_t_count['week_sin'] * cust_t_count['customer_id'],\n            y=cust_t_count['week_cos'] * cust_t_count['customer_id'],\n            color='Red',\n            s=100)\n\n# second example\nexample = 'Dress'\ncust_t_count = transactions[transactions['article_cat'] == example].groupby(['week',\n                                                                             'week_sin',\n                                                                             'week_cos'], as_index=False)[\n            'customer_id'].count().sort_index()\n\nplt.scatter(x=cust_t_count['week_sin'] * cust_t_count['customer_id'],\n            y=cust_t_count['week_cos'] * cust_t_count['customer_id'],\n            color='Blue',\n            s=100)\n\n# we can see a pattern indicative of summer-winter cycle\nplt.legend(['Sweater', 'Dress'])\nplt.title('Sweater / Dress sales around year (as cycle)')\nplt.xlabel('week of year - sin')\nplt.ylabel('week of year - cos')\nshow_clear_plt()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T12:46:42.26238Z","iopub.execute_input":"2022-02-16T12:46:42.262714Z","iopub.status.idle":"2022-02-16T12:46:58.669737Z","shell.execute_reply.started":"2022-02-16T12:46:42.262682Z","shell.execute_reply":"2022-02-16T12:46:58.668736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Customer purchases by Ladies / Mens wear etc","metadata":{}},{"cell_type":"markdown","source":"What can we deduce about customers from the purchases - example, using index_group_name\n\nAre customers (for example) more likely to purchase from Ladieswear in the future, if most past purchases have been from that department?","metadata":{}},{"cell_type":"code","source":"transactions['index_group_name'] = transactions['article_id'].map(dict(zip(articles['article_id'],\n                                                                           articles['index_group_name'])))\n\norder = ['Ladieswear', 'Divided', 'Sport', 'Baby/Children',  'Menswear', ]\n\ncustomer_split = transactions.groupby(['customer_id', 'index_group_name'])['article_id'].count().unstack('index_group_name').fillna(value=0)[order]\ncustomer_split.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T12:48:58.401152Z","iopub.execute_input":"2022-02-16T12:48:58.401784Z","iopub.status.idle":"2022-02-16T12:49:21.697106Z","shell.execute_reply.started":"2022-02-16T12:48:58.401731Z","shell.execute_reply":"2022-02-16T12:49:21.696154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A simple analysis is just to look at correlations","metadata":{}},{"cell_type":"code","source":"corr = customer_split.corr()\nfig,axes=plt.subplots(figsize=(13,7))\nsns.heatmap(corr,\n           annot=True,\n           fmt=\".2f\",\n            cmap='seismic_r',\n            vmin=0, \n            vmax=1,\n           linewidth=1)\nplt.title('Customer Purchase Correlations')\nshow_clear_plt()","metadata":{"execution":{"iopub.status.busy":"2022-02-16T12:50:04.8165Z","iopub.execute_input":"2022-02-16T12:50:04.816829Z","iopub.status.idle":"2022-02-16T12:50:05.40496Z","shell.execute_reply.started":"2022-02-16T12:50:04.816797Z","shell.execute_reply":"2022-02-16T12:50:05.403999Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Look at customers with >1 purchase and <41 purchases","metadata":{}},{"cell_type":"code","source":"corr = customer_split[(customer_split.sum(axis=1)>1) & \n                     (customer_split.sum(axis=1)<41)].corr()\nfig,axes=plt.subplots(figsize=(13,7))\nsns.heatmap(corr,\n           annot=True,\n           fmt=\".2f\",\n            cmap='seismic_r',\n            vmin=-0.2, \n            vmax=1,\n           linewidth=1)\nplt.title('Customer Purchase Correlations, for customers buying 2-40 items')\nshow_clear_plt()","metadata":{"execution":{"iopub.status.busy":"2022-02-16T12:49:40.933762Z","iopub.execute_input":"2022-02-16T12:49:40.934038Z","iopub.status.idle":"2022-02-16T12:49:41.540924Z","shell.execute_reply.started":"2022-02-16T12:49:40.934009Z","shell.execute_reply":"2022-02-16T12:49:41.540337Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check correlation if we reduce to a Y/N (0 or 1)","metadata":{}},{"cell_type":"code","source":"customer_split_ = customer_split.copy()\ncustomer_split_[:] = np.where(customer_split_[:]==0, 0, 1)\ncorr = customer_split_.corr()\nfig,axes=plt.subplots(figsize=(13,7))\nsns.heatmap(corr,\n           annot=True,\n           fmt=\".2f\",\n            cmap='seismic_r',\n            vmin=-0.2, \n            vmax=1,\n           linewidth=1)\nplt.title('Customer Purchase Correlations, binary Y/N')\nshow_clear_plt()","metadata":{"execution":{"iopub.status.busy":"2022-02-16T12:48:10.61724Z","iopub.execute_input":"2022-02-16T12:48:10.617731Z","iopub.status.idle":"2022-02-16T12:48:11.160915Z","shell.execute_reply.started":"2022-02-16T12:48:10.61769Z","shell.execute_reply":"2022-02-16T12:48:11.160154Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check correlation if we reduce to a Y/N (0 or 1) and filter to customers buying 2-40 items","metadata":{}},{"cell_type":"code","source":"corr = customer_split_[(customer_split.sum(axis=1)>1) & \n                     (customer_split.sum(axis=1)<41)].corr()\nfig,axes=plt.subplots(figsize=(13,7))\nsns.heatmap(corr,\n           annot=True,\n           fmt=\".2f\",\n            cmap='seismic_r',\n            vmin=-0.2, \n            vmax=1,\n           linewidth=1)\nplt.title('Customer Purchase Correlations, binary Y/N, for customers buying 2-40 items')\nshow_clear_plt()","metadata":{"execution":{"iopub.status.busy":"2022-02-16T12:50:15.864252Z","iopub.execute_input":"2022-02-16T12:50:15.8646Z","iopub.status.idle":"2022-02-16T12:50:17.099607Z","shell.execute_reply.started":"2022-02-16T12:50:15.864565Z","shell.execute_reply":"2022-02-16T12:50:17.098602Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Items Commonly Bought by the same Customer","metadata":{}},{"cell_type":"markdown","source":"Review some examples (based on a subset of the data) of items most commonly found together within a customer's purchase history","metadata":{}},{"cell_type":"code","source":"#to avoid out of memory, need to filter to most common items (pending finding a more efficient approach)\ndel customers\ngc.collect()\ntransactions['article_total'] = transactions['article_id'].map(transactions['article_id'].value_counts())\ntransactions['customer_total'] = transactions['customer_id'].map(transactions['customer_id'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-02-16T12:29:35.464972Z","iopub.execute_input":"2022-02-16T12:29:35.46594Z","iopub.status.idle":"2022-02-16T12:29:54.745212Z","shell.execute_reply.started":"2022-02-16T12:29:35.465888Z","shell.execute_reply":"2022-02-16T12:29:54.744324Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions = reduce_memory_usage(transactions, transactions.columns)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T12:29:54.746431Z","iopub.execute_input":"2022-02-16T12:29:54.746787Z","iopub.status.idle":"2022-02-16T12:29:58.445451Z","shell.execute_reply.started":"2022-02-16T12:29:54.746757Z","shell.execute_reply":"2022-02-16T12:29:58.444763Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To avoid out of memory, need to filter to most common items (pending finding a more efficient approach)","metadata":{}},{"cell_type":"code","source":"FILTER = 5000\nCUST_FILTER = 3\npc_rows = sum((transactions['article_total']>FILTER) & \n             (transactions['customer_total'] > CUST_FILTER)) / len(transactions)\n\nprint(f'% of rows to include based on product volume filter - {pc_rows}')","metadata":{"execution":{"iopub.status.busy":"2022-02-16T12:29:58.446551Z","iopub.execute_input":"2022-02-16T12:29:58.446881Z","iopub.status.idle":"2022-02-16T12:30:03.151183Z","shell.execute_reply.started":"2022-02-16T12:29:58.446854Z","shell.execute_reply":"2022-02-16T12:30:03.150178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cust_hist = transactions[(transactions['article_total']>FILTER) & \n                        (transactions['customer_total'] > CUST_FILTER)].groupby(['customer_id'])['article_id'].apply(list)\nprint(len(cust_hist))\n\nsorted_lists = [sorted(list(set(x))) for x in cust_hist.values]\n\nfor count, l in enumerate(tqdm(sorted_lists)):\n    sorted_lists[count] = [f'{a}_{b}' for a, b in itertools.combinations(l, 2) if a != b]\n    \nall_items = [item for sublist in sorted_lists for item in sublist] \nprint(len(all_items))\ncounter = Counter(all_items)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T12:30:03.152599Z","iopub.execute_input":"2022-02-16T12:30:03.152932Z","iopub.status.idle":"2022-02-16T12:30:46.739Z","shell.execute_reply.started":"2022-02-16T12:30:03.152887Z","shell.execute_reply":"2022-02-16T12:30:46.738251Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets look at what these products are (descr + image)\n\nThe results look quite logical, however some of these maybe make more sense as joint purchases (same time) rather than as predictive for future purchase.","metadata":{}},{"cell_type":"markdown","source":"Copied the image path code from the notebook below + some edits\n\nhttps://www.kaggle.com/gpreda/h-m-eda-and-prediction","metadata":{}},{"cell_type":"code","source":"image_path = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/\"\nfor e in range(CONFIG.EXAMPLE_LIMIT):\n    sel_articles = list(counter.most_common()[e][0].split('_'))   \n    \n    fig, axes = plt.subplots(nrows=1, ncols=2, figsize=(8, 5))\n\n    for i, a in enumerate(sel_articles):\n        article_id = (\"0\" + str(sel_articles[i]))[-10:]\n        axes.ravel()[i].axis('off')\n\n        try:\n            image = Image.open(f\"{image_path}{article_id[:3]}/{article_id}.jpg\")\n            axes.ravel()[i].imshow(image)\n        except:\n            print(f'image not found for article {a}')\n        axes.ravel()[i].set_title(str(a), fontsize=18)\n        \n    plt.suptitle(sel_articles, fontsize=24)\n\n    show_clear_plt()\n    \ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-02-16T12:31:01.172394Z","iopub.execute_input":"2022-02-16T12:31:01.172975Z","iopub.status.idle":"2022-02-16T12:31:09.604066Z","shell.execute_reply.started":"2022-02-16T12:31:01.172922Z","shell.execute_reply":"2022-02-16T12:31:09.603119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Treemaps of Transactions / Articles","metadata":{}},{"cell_type":"code","source":"# treemaps of contribution to total transactions\nwidth = 800\n\nfig = px.treemap(articles[(articles['sales'] > 0)],\n                 path=[px.Constant(\"Total\"),\n                       'index_group_name',\n                       'product_type_name',\n                       'colour_group_name', ], values='sales',\n                 labels='sales',\n                color='index_group_name',                \n                 color_discrete_sequence=px.colors.qualitative.Pastel2,)\n\nfig.update_layout(title=dict(text=f'<b>Contribution to Transactions by index group <br>name, product type name, color group',\n                             font=dict(\n                                 family=\"Arial\",\n                                 size=24,\n                                 color='#000000'\n                             )),\n                  margin=dict(l=20, r=20, t=100, b=20),\n                  height=800,\n                  width=width,\n                  font=dict(\n                      family=\"Arial Black\",\n                      size=16,\n                      color='#000000'\n                  )\n                  )\n\nfig.update_traces(marker_line_width=1)\nfig.update_traces(marker_line_color='grey')\nfig.show()","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T12:31:09.605533Z","iopub.execute_input":"2022-02-16T12:31:09.605783Z","iopub.status.idle":"2022-02-16T12:31:14.327157Z","shell.execute_reply.started":"2022-02-16T12:31:09.605752Z","shell.execute_reply":"2022-02-16T12:31:14.326477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that trousers are steadily popular during whole year, while T-shirt and Dress for example is more popular in Q2/Q3 (presumably summer)\n\nSweater and jacket are more popular in Q4/Q1 (presumably winter)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T00:34:11.569724Z","iopub.execute_input":"2022-02-17T00:34:11.570127Z","iopub.status.idle":"2022-02-17T00:34:11.576383Z","shell.execute_reply.started":"2022-02-17T00:34:11.570091Z","shell.execute_reply":"2022-02-17T00:34:11.575369Z"}}},{"cell_type":"code","source":"# treemaps of contribution to total transactions by quarter and product type\ntransactions['product_type_name'] = transactions['article_id'].map(dict(zip(articles['article_id'],\n                                                                           articles['product_type_name'])))\n\nsummary = transactions.groupby(['product_type_name',\n                       'quarter'], as_index=False)['customer_id'].count()\n\nfig = px.treemap(summary,\n                 path=[px.Constant(\"Total\"),\n                       'product_type_name',\n                       'quarter',], values='customer_id',\n                 color='quarter',\n                 color_continuous_scale='twilight',\n                 labels='customer_id')\n\nfig.update_layout(title=dict(text=f'<b>Contribution to Transactions <br>by Product Type / Quarter',\n                             font=dict(\n                                 family=\"Arial\",\n                                size=24,\n                                 color='#000000'\n                             )),\n                  margin=dict(l=20, r=20, t=100, b=20),\n                  height=800,\n                  width=width,\n                  font=dict(\n                      family=\"Arial Black\",\n                      size=16,\n                      color='#000000'\n                  )\n                  )\n\nfig.update_traces(marker_line_width=1)\nfig.update_traces(marker_line_color='grey')\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T12:31:14.328197Z","iopub.execute_input":"2022-02-16T12:31:14.328572Z","iopub.status.idle":"2022-02-16T12:31:37.137808Z","shell.execute_reply.started":"2022-02-16T12:31:14.328541Z","shell.execute_reply":"2022-02-16T12:31:37.136821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There seem to be some seasonal colour trends, with lighter colours more popular in Q2/Q3.\n\nRed seems to be particularly popular in Q4 (festive?)","metadata":{}},{"cell_type":"code","source":"# treemaps of contribution to total transactions by quarter and product type\ntransactions['perceived_colour_master_name'] = transactions['article_id'].map(dict(zip(articles['article_id'],\n                                                                           articles['perceived_colour_master_name'])))\n\nsummary = transactions.groupby(['perceived_colour_master_name',\n                       'quarter'], as_index=False)['customer_id'].count()\n\nfig = px.treemap(summary,\n                 path=[px.Constant(\"Total\"),\n                       'perceived_colour_master_name',\n                       'quarter',], values='customer_id',\n                 color='quarter',\n                 color_continuous_scale='twilight',\n                 labels='customer_id')\n\nfig.update_layout(title=dict(text=f'<b>Contribution to Transactions <br>by Product Perceived Colour / Quarter',\n                             font=dict(\n                                 family=\"Arial\",\n                                 size=24,\n                                 color='#000000'\n                             )),\n                  margin=dict(l=20, r=20, t=100, b=20),\n                  height=800,\n                  width=width,\n                  font=dict(\n                      family=\"Arial Black\",\n                      size=16,\n                      color='#000000'\n                  )\n                  )\n\nfig.update_traces(marker_line_width=1)\nfig.update_traces(marker_line_color='grey')\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T12:31:37.139353Z","iopub.execute_input":"2022-02-16T12:31:37.140195Z","iopub.status.idle":"2022-02-16T12:31:59.135347Z","shell.execute_reply.started":"2022-02-16T12:31:37.140157Z","shell.execute_reply":"2022-02-16T12:31:59.134489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It is hard to pick out obvious age trends below.\n\n30s-40s looks under-represented in Menswear, and Baby/Children is more popular with customers in range 30-50.","metadata":{}},{"cell_type":"code","source":"transactions['index_group_name'] = transactions['article_id'].map(dict(zip(articles['article_id'],\n                                                                           articles['index_group_name'])))\n\nsummary = transactions.groupby(['index_group_name',\n                       'age_decade'], as_index=False)['customer_id'].count()\n\nfig = px.treemap(summary,\n                 path=[px.Constant(\"Total\"),\n                       'index_group_name',\n                       'age_decade',], values='customer_id',\n                 color='age_decade',\n                 color_discrete_sequence=px.colors.qualitative.Pastel2,\n                # color_continuous_scale='Blues',\n                 labels='customer_id')\n\nfig.update_layout(title=dict(text=f'<b>Contribution to Transactions <br>by Customer Age Bracket',\n                             font=dict(\n                                 family=\"Arial\",\n                                size=24,\n                                 color='#000000'\n                             )),\n                  margin=dict(l=20, r=20, t=100, b=20),\n                  height=800,\n                  width=width,\n                  font=dict(\n                      family=\"Arial Black\",\n                      size=16,\n                      color='#000000'\n                  )\n                  )\n\nfig.update_traces(marker_line_width=1)\nfig.update_traces(marker_line_color='grey')\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T12:31:59.137204Z","iopub.execute_input":"2022-02-16T12:31:59.137853Z","iopub.status.idle":"2022-02-16T12:32:14.555099Z","shell.execute_reply.started":"2022-02-16T12:31:59.137808Z","shell.execute_reply":"2022-02-16T12:32:14.554069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most product types see the majority of sales online.\n\nSocks is an exception with store purchases >50% (impulse purchase?)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T00:37:03.487357Z","iopub.execute_input":"2022-02-17T00:37:03.488342Z","iopub.status.idle":"2022-02-17T00:37:03.493251Z","shell.execute_reply.started":"2022-02-17T00:37:03.4883Z","shell.execute_reply":"2022-02-17T00:37:03.492154Z"}}},{"cell_type":"code","source":"summary = transactions.groupby(['product_type_name',\n                       'sales_channel_name'], as_index=False)['customer_id'].count()\n\nfig = px.treemap(summary,\n                 path=[px.Constant(\"Total\"),\n                       'product_type_name',\n                       'sales_channel_name',], values='customer_id',\n                 color='sales_channel_name',\n                 color_discrete_sequence=px.colors.qualitative.Pastel2,\n                 labels='customer_id')\n\nfig.update_layout(title=dict(text=f'<b>Contribution to Transactions <br>by Product Type / Channel Name',\n                             font=dict(\n                                 family=\"Arial\",\n                                 size=30,\n                                 color='#000000'\n                             )),\n                  margin=dict(l=20, r=20, t=100, b=20),\n                  height=800,\n                  width=width,\n                  font=dict(\n                      family=\"Arial Black\",\n                      size=16,\n                      color='#000000'\n                  )\n                  )\n\nfig.update_traces(marker_line_width=1)\nfig.update_traces(marker_line_color='grey')\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T12:32:14.556784Z","iopub.execute_input":"2022-02-16T12:32:14.557032Z","iopub.status.idle":"2022-02-16T12:32:25.813028Z","shell.execute_reply.started":"2022-02-16T12:32:14.557003Z","shell.execute_reply":"2022-02-16T12:32:25.81217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Age group 30-40 is the 2nd largest demographic for Online but 4th largest for Store - this age group seems relatively less inclined to shop in person","metadata":{}},{"cell_type":"code","source":"summary = transactions.groupby(['age_decade',\n                       'sales_channel_name'], as_index=False)['customer_id'].count()\n\nfig = px.treemap(summary,\n                 path=[px.Constant(\"Total\"),\n                       'sales_channel_name',\n                       'age_decade',], values='customer_id',\n                 color='sales_channel_name',\n                 color_discrete_sequence=px.colors.qualitative.Pastel2,\n                 labels='customer_id')\n\nfig.update_layout(title=dict(text=f'<b>Contribution to Transactions <br>by Product Type / Channel Name',\n                             font=dict(\n                                 family=\"Arial\",\n                                 size=30,\n                                 color='#000000'\n                             )),\n                  margin=dict(l=20, r=20, t=100, b=20),\n                  height=800,\n                  width=width,\n                  font=dict(\n                      family=\"Arial Black\",\n                      size=16,\n                      color='#000000'\n                  )\n                  )\n\nfig.update_traces(marker_line_width=1)\nfig.update_traces(marker_line_color='grey')\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T12:32:25.814469Z","iopub.execute_input":"2022-02-16T12:32:25.814696Z","iopub.status.idle":"2022-02-16T12:32:39.168692Z","shell.execute_reply.started":"2022-02-16T12:32:25.814669Z","shell.execute_reply":"2022-02-16T12:32:39.167683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image Sampling for various Categories","metadata":{}},{"cell_type":"markdown","source":"Copied\nthe\ncode\nfrom the notebook\nbelow + some\nedits\n\nhttps://www.kaggle.com/gpreda/h-m-eda-and-prediction\n","metadata":{}},{"cell_type":"code","source":"def plot_image_samples(image_article_df, col_name, cols=3, max_examples=10, max_images=10):\n    # extract list of top unique entries\n    unique_entries = image_article_df[col_name].value_counts().sort_values(ascending=False).index[\n                     :max_examples].tolist()\n\n    image_path = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/\"\n\n    for count, u in enumerate(unique_entries):\n        _df = image_article_df.loc[image_article_df[col_name] == u].sample(frac=1.0, random_state=count)\n\n        sel_articles = _df['article_id'].unique().tolist()[:max_images]\n        nr = math.ceil(len(sel_articles) / cols)\n\n        fig, axes = plt.subplots(nrows=nr, ncols=cols, figsize=(8 * cols, 4 + 3 * nr))\n\n        for i, a in enumerate(sel_articles):\n            article_id = (\"0\" + str(sel_articles[i]))[-10:]\n\n            axes.ravel()[i].axis('off')\n\n            try:\n                image = Image.open(f\"{image_path}{article_id[:3]}/{article_id}.jpg\")\n                axes.ravel()[i].imshow(image)\n            except:\n                print(f'image not found for article {a}')\n            axes.ravel()[i].set_title(u + str(a), fontsize=34)\n        plt.suptitle(u, fontsize=50)\n\n        show_clear_plt()\n","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-02-16T12:32:39.170068Z","iopub.execute_input":"2022-02-16T12:32:39.170302Z","iopub.status.idle":"2022-02-16T12:32:39.181792Z","shell.execute_reply.started":"2022-02-16T12:32:39.170274Z","shell.execute_reply":"2022-02-16T12:32:39.180687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The code cycles through categories and provides random examples of some of the categorisations using the images to provide a better understanding of how the terminology relates to products.","metadata":{}},{"cell_type":"code","source":"count_columns = [\n    'product_type_name',\n    'product_group_name',\n    'graphical_appearance_name',\n    'colour_group_name',\n    'perceived_colour_value_name',\n    'perceived_colour_master_name',\n    'department_name',\n    'index_name',\n    'index_group_name',\n    'section_name',\n    'garment_group_name',\n]\n\nfor cc in count_columns:\n    plot_image_samples(articles, cc, cols=5, max_examples=5, max_images=5)","metadata":{"collapsed":false,"pycharm":{"name":"#%%\n"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-16T12:32:39.183007Z","iopub.execute_input":"2022-02-16T12:32:39.183295Z","iopub.status.idle":"2022-02-16T12:34:33.587359Z","shell.execute_reply.started":"2022-02-16T12:32:39.183263Z","shell.execute_reply":"2022-02-16T12:34:33.586688Z"},"trusted":true},"execution_count":null,"outputs":[]}]}