{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![](https://profashion.ru/upload/iblock/bf6/f721319i3l8n3d5tn3tt0fvs207a200m/h_m820.jpg)","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport cv2\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-13T15:29:34.125906Z","iopub.execute_input":"2022-02-13T15:29:34.126247Z","iopub.status.idle":"2022-02-13T15:29:34.133447Z","shell.execute_reply.started":"2022-02-13T15:29:34.126213Z","shell.execute_reply":"2022-02-13T15:29:34.132336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_FOLDER = '/kaggle/input/h-and-m-personalized-fashion-recommendations'","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-13T15:29:34.135234Z","iopub.execute_input":"2022-02-13T15:29:34.135819Z","iopub.status.idle":"2022-02-13T15:29:34.152657Z","shell.execute_reply.started":"2022-02-13T15:29:34.135774Z","shell.execute_reply":"2022-02-13T15:29:34.151688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\narticles_df = pd.read_csv(f'{DATA_FOLDER}/articles.csv', dtype={'article_id': str})\ncustomers_df = pd.read_csv(f\"{DATA_FOLDER}/customers.csv\")\nsample_submission_df = pd.read_csv(f\"{DATA_FOLDER}/sample_submission.csv\", dtype={'article_id': str})","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-13T15:29:34.154128Z","iopub.execute_input":"2022-02-13T15:29:34.154613Z","iopub.status.idle":"2022-02-13T15:29:46.913520Z","shell.execute_reply.started":"2022-02-13T15:29:34.154567Z","shell.execute_reply":"2022-02-13T15:29:46.912443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_train_df = pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\", dtype={'article_id': str})","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-13T15:29:46.915453Z","iopub.execute_input":"2022-02-13T15:29:46.915986Z","iopub.status.idle":"2022-02-13T15:30:50.829069Z","shell.execute_reply.started":"2022-02-13T15:29:46.915945Z","shell.execute_reply":"2022-02-13T15:30:50.828156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df.head().T","metadata":{"execution":{"iopub.status.busy":"2022-02-13T15:30:50.831673Z","iopub.execute_input":"2022-02-13T15:30:50.832525Z","iopub.status.idle":"2022-02-13T15:30:50.852991Z","shell.execute_reply.started":"2022-02-13T15:30:50.832477Z","shell.execute_reply":"2022-02-13T15:30:50.851885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-02-13T15:30:50.854664Z","iopub.execute_input":"2022-02-13T15:30:50.854971Z","iopub.status.idle":"2022-02-13T15:30:50.870413Z","shell.execute_reply.started":"2022-02-13T15:30:50.854935Z","shell.execute_reply":"2022-02-13T15:30:50.869458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Amount of unique products in dataset\n\n#### The most interesting things that for each `product_code` several `article_id`s can be exist. So, the `product_code` here represents UPC (Universal Product Code) and `article_id` is a SKU (Stock keeping unit) which is assigned to a product with a different properties of the same product. Let's define how many unique products our dataset contains","metadata":{}},{"cell_type":"code","source":"articles_df.article_id.shape[0]==articles_df.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-02-13T15:30:50.872045Z","iopub.execute_input":"2022-02-13T15:30:50.872565Z","iopub.status.idle":"2022-02-13T15:30:50.887063Z","shell.execute_reply.started":"2022-02-13T15:30:50.872519Z","shell.execute_reply":"2022-02-13T15:30:50.886246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df.product_code.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-02-13T15:30:50.888429Z","iopub.execute_input":"2022-02-13T15:30:50.888997Z","iopub.status.idle":"2022-02-13T15:30:50.906731Z","shell.execute_reply.started":"2022-02-13T15:30:50.888942Z","shell.execute_reply":"2022-02-13T15:30:50.905420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### So that, dataset consist of 47224 unique products","metadata":{}},{"cell_type":"markdown","source":"# Articles EDA","metadata":{}},{"cell_type":"code","source":"temp = articles_df.groupby('garment_group_name')['garment_group_name'].count()\ndf = pd.DataFrame({'garment_group_name': temp.index,\n                   'amount': temp.values\n                  })\ndf = df.sort_values(['amount'], ascending=False)\nplt.figure(figsize = (20,8))\nplt.title('Garment Group distribution')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'garment_group_name', y=\"amount\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-13T15:30:50.908244Z","iopub.execute_input":"2022-02-13T15:30:50.909563Z","iopub.status.idle":"2022-02-13T15:30:51.319640Z","shell.execute_reply.started":"2022-02-13T15:30:50.909514Z","shell.execute_reply":"2022-02-13T15:30:51.318642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### There are 5 main categories that appears in dataset and they have following distribution","metadata":{}},{"cell_type":"code","source":"temp = articles_df.groupby('index_group_name')['index_group_name'].count()\ndf = pd.DataFrame({'index_group_name': temp.index,\n                   'amount': temp.values\n                  })\ndf = df.sort_values(['amount'], ascending=False)\nplt.figure(figsize = (15,6))\nplt.title('Main categories distribution')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'index_group_name', y=\"amount\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-13T15:30:51.321138Z","iopub.execute_input":"2022-02-13T15:30:51.321436Z","iopub.status.idle":"2022-02-13T15:30:51.581347Z","shell.execute_reply.started":"2022-02-13T15:30:51.321400Z","shell.execute_reply":"2022-02-13T15:30:51.579863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Each of this main categories have a leaf categories and they have the following distribution","metadata":{}},{"cell_type":"code","source":"temp = articles_df.groupby('section_name')['section_name'].count()\ndf = pd.DataFrame({'section_name': temp.index,\n                   'amount': temp.values\n                  })\ndf = df.sort_values(['amount'], ascending=False)\nplt.figure(figsize = (20,8))\nplt.title('distribution of sections')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'section_name', y=\"amount\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-13T15:30:51.582984Z","iopub.execute_input":"2022-02-13T15:30:51.583284Z","iopub.status.idle":"2022-02-13T15:30:52.753662Z","shell.execute_reply.started":"2022-02-13T15:30:51.583245Z","shell.execute_reply":"2022-02-13T15:30:52.752570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### So, let's have a look on a distribution of leaf categories separetley grouped by main categories","metadata":{}},{"cell_type":"code","source":"_temp = articles_df.groupby('index_group_name')['section_name']\nfor group_name, _df in _temp:\n    temp = _df.value_counts()\n    df = pd.DataFrame({group_name: temp.index,\n                       'amount': temp.values\n                      })\n    df = df.sort_values(['amount'], ascending=False)\n    plt.figure(figsize = (20,8))\n    plt.title(f'Distribution of sections inside {group_name} category')\n    sns.set_color_codes(\"pastel\")\n    s = sns.barplot(x = group_name, y=\"amount\", data=df)\n    s.set_xticklabels(s.get_xticklabels(),rotation=90)\n    locs, labels = plt.xticks()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-13T15:30:52.755622Z","iopub.execute_input":"2022-02-13T15:30:52.759720Z","iopub.status.idle":"2022-02-13T15:30:54.367211Z","shell.execute_reply.started":"2022-02-13T15:30:52.759630Z","shell.execute_reply":"2022-02-13T15:30:54.366351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Clients EDA","metadata":{}},{"cell_type":"code","source":"customers_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-13T15:30:54.368665Z","iopub.execute_input":"2022-02-13T15:30:54.369031Z","iopub.status.idle":"2022-02-13T15:30:54.387252Z","shell.execute_reply.started":"2022-02-13T15:30:54.368993Z","shell.execute_reply":"2022-02-13T15:30:54.386150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-02-13T15:30:54.391691Z","iopub.execute_input":"2022-02-13T15:30:54.392062Z","iopub.status.idle":"2022-02-13T15:30:54.400784Z","shell.execute_reply.started":"2022-02-13T15:30:54.392024Z","shell.execute_reply":"2022-02-13T15:30:54.399813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-02-13T15:30:54.403096Z","iopub.execute_input":"2022-02-13T15:30:54.405627Z","iopub.status.idle":"2022-02-13T15:30:55.053674Z","shell.execute_reply.started":"2022-02-13T15:30:54.405579Z","shell.execute_reply":"2022-02-13T15:30:55.052426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_df.Active.unique()","metadata":{"execution":{"iopub.status.busy":"2022-02-13T15:30:55.055218Z","iopub.execute_input":"2022-02-13T15:30:55.055514Z","iopub.status.idle":"2022-02-13T15:30:55.078657Z","shell.execute_reply.started":"2022-02-13T15:30:55.055479Z","shell.execute_reply":"2022-02-13T15:30:55.077509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_df.FN.unique()","metadata":{"execution":{"iopub.status.busy":"2022-02-13T15:30:55.080399Z","iopub.execute_input":"2022-02-13T15:30:55.081002Z","iopub.status.idle":"2022-02-13T15:30:55.107762Z","shell.execute_reply.started":"2022-02-13T15:30:55.080948Z","shell.execute_reply":"2022-02-13T15:30:55.105524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### so all `np.nan` values can be replaced with 0 value for example","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,8))\nsns.histplot(customers_df.fashion_news_frequency)","metadata":{"execution":{"iopub.status.busy":"2022-02-13T15:30:55.109638Z","iopub.execute_input":"2022-02-13T15:30:55.110038Z","iopub.status.idle":"2022-02-13T15:30:58.001283Z","shell.execute_reply.started":"2022-02-13T15:30:55.109988Z","shell.execute_reply":"2022-02-13T15:30:57.999947Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### So, we can see that `None`, `np.nan`, `NONE` can be grouped into one common label `None`","metadata":{}},{"cell_type":"code","source":"customers_df.fashion_news_frequency[customers_df.fashion_news_frequency.isin(['NONE',np.nan])]=\"None\"","metadata":{"execution":{"iopub.status.busy":"2022-02-13T15:30:58.162644Z","iopub.execute_input":"2022-02-13T15:30:58.163399Z","iopub.status.idle":"2022-02-13T15:30:58.303487Z","shell.execute_reply.started":"2022-02-13T15:30:58.163338Z","shell.execute_reply":"2022-02-13T15:30:58.302492Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,8))\nplt.yscale(\"log\")\nplt.title(\"Distribution of newsletter subscription\")\nsns.histplot(customers_df.fashion_news_frequency)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-13T15:30:58.304627Z","iopub.execute_input":"2022-02-13T15:30:58.304900Z","iopub.status.idle":"2022-02-13T15:31:01.768979Z","shell.execute_reply.started":"2022-02-13T15:30:58.304869Z","shell.execute_reply":"2022-02-13T15:31:01.768089Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,8))\nsns.histplot(x='club_member_status', data=customers_df)\nplt.title(\"Distribution of statuses\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-13T15:31:01.770645Z","iopub.execute_input":"2022-02-13T15:31:01.770966Z","iopub.status.idle":"2022-02-13T15:31:03.801744Z","shell.execute_reply.started":"2022-02-13T15:31:01.770919Z","shell.execute_reply":"2022-02-13T15:31:03.800695Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,8))\ntemp = customers_df.groupby('age')['customer_id'].count()\ndf = pd.DataFrame({\n                    \"age\": temp.index,\n                    \"count\": temp.values\n                   })\ns = sns.barplot(x='age', y='count', data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.title(\"Customers' age distribution\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-13T15:31:03.803266Z","iopub.execute_input":"2022-02-13T15:31:03.803568Z","iopub.status.idle":"2022-02-13T15:31:07.422342Z","shell.execute_reply.started":"2022-02-13T15:31:03.803532Z","shell.execute_reply":"2022-02-13T15:31:07.421035Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### As we can see from the Figure above, age has a multimodal distribution","metadata":{}},{"cell_type":"markdown","source":"# MORE INSIGHTS WILL BE RELEASED SOON","metadata":{"_kg_hide-input":true}}]}