{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# if u like it up vote it😁","metadata":{}},{"cell_type":"markdown","source":"# Importing libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom wordcloud import WordCloud, STOPWORDS\nfrom datetime import datetime\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:10:23.71444Z","iopub.execute_input":"2022-04-09T10:10:23.714704Z","iopub.status.idle":"2022-04-09T10:10:24.746712Z","shell.execute_reply.started":"2022-04-09T10:10:23.714678Z","shell.execute_reply":"2022-04-09T10:10:24.745764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Counrting total no. of files and folders :**","metadata":{}},{"cell_type":"code","source":"total_folders = total_files = 0\nfolder_info = []\nimages_names = []\nfor base, dirs, files in tqdm(os.walk('/kaggle/input/h-and-m-personalized-fashion-recommendations/')):\n    for directories in dirs:\n        folder_info.append((directories, len(os.listdir(os.path.join(base, directories)))))\n        total_folders += 1\n    for _files in files:\n        total_files += 1\n        if len(_files.split(\".jpg\"))==2:\n            images_names.append(_files.split(\".jpg\")[0])","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:10:26.043518Z","iopub.execute_input":"2022-04-09T10:10:26.04427Z","iopub.status.idle":"2022-04-09T10:10:52.47653Z","shell.execute_reply.started":"2022-04-09T10:10:26.044233Z","shell.execute_reply":"2022-04-09T10:10:52.475668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Total number of folders: {total_folders}\\nTotal number of files: {total_files}\")\nfolder_info_df = pd.DataFrame(folder_info, columns=[\"folder\", \"files count\"])\nfolder_info_df.sort_values([\"files count\"], ascending=False).head()","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:10:52.478234Z","iopub.execute_input":"2022-04-09T10:10:52.478452Z","iopub.status.idle":"2022-04-09T10:10:52.506037Z","shell.execute_reply.started":"2022-04-09T10:10:52.478424Z","shell.execute_reply":"2022-04-09T10:10:52.505505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reading the data ","metadata":{}},{"cell_type":"code","source":"articles_df = pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv\")\ncustomers_df = pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv\")\nsample_submission_df = pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\")\ntransactions_train_df = pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:10:52.507039Z","iopub.execute_input":"2022-04-09T10:10:52.507566Z","iopub.status.idle":"2022-04-09T10:12:04.578717Z","shell.execute_reply.started":"2022-04-09T10:10:52.507534Z","shell.execute_reply":"2022-04-09T10:12:04.577883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:12:04.580703Z","iopub.execute_input":"2022-04-09T10:12:04.58092Z","iopub.status.idle":"2022-04-09T10:12:04.691295Z","shell.execute_reply.started":"2022-04-09T10:12:04.580895Z","shell.execute_reply":"2022-04-09T10:12:04.690467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_df","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:12:04.692731Z","iopub.execute_input":"2022-04-09T10:12:04.692964Z","iopub.status.idle":"2022-04-09T10:12:04.712925Z","shell.execute_reply.started":"2022-04-09T10:12:04.692936Z","shell.execute_reply":"2022-04-09T10:12:04.711792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_train_df","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:12:04.714274Z","iopub.execute_input":"2022-04-09T10:12:04.714556Z","iopub.status.idle":"2022-04-09T10:12:04.736099Z","shell.execute_reply.started":"2022-04-09T10:12:04.714515Z","shell.execute_reply":"2022-04-09T10:12:04.735275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission_df","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:12:04.737292Z","iopub.execute_input":"2022-04-09T10:12:04.737725Z","iopub.status.idle":"2022-04-09T10:12:04.750601Z","shell.execute_reply.started":"2022-04-09T10:12:04.73769Z","shell.execute_reply":"2022-04-09T10:12:04.749877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# About data \n","metadata":{"_kg_hide-output":true}},{"cell_type":"markdown","source":"* There are 4 tabels :\n 1. articles - contains informations about each article (105542 rows × 25 columns)\n 2. customers - contains informations about each customer (1371980 rows × 7 columns)\n 3. transactions (31788324 rows × 5 columns)\n 4. sample submission(1371980 rows × 2 columns)\n \n* Foreign keys for the customer and articles tables= 'customer_id` & 'article_id'\n \n* Transaction train data has entries for the date of the transaction, the customer id, the article id, a price (per transaction) and a sales channel id.","metadata":{}},{"cell_type":"code","source":"# Function to count missing values\ndef missing_data(data):\n    total = data.isnull().sum().sort_values(ascending = False)\n    percent = (data.isnull().sum()/data.isnull().count()*100).sort_values(ascending = False)\n    return pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-09T10:12:04.751424Z","iopub.execute_input":"2022-04-09T10:12:04.751627Z","iopub.status.idle":"2022-04-09T10:12:04.763081Z","shell.execute_reply.started":"2022-04-09T10:12:04.751601Z","shell.execute_reply":"2022-04-09T10:12:04.762316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def unique_values(data):\n    total = data.count()\n    tt = pd.DataFrame(total)\n    tt.columns = ['Total']\n    uniques = []\n    for col in data.columns:\n        unique = data[col].nunique()\n        uniques.append(unique)\n    tt['Uniques'] = uniques\n    return tt","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-09T10:12:04.76514Z","iopub.execute_input":"2022-04-09T10:12:04.765495Z","iopub.status.idle":"2022-04-09T10:12:04.777257Z","shell.execute_reply.started":"2022-04-09T10:12:04.765461Z","shell.execute_reply":"2022-04-09T10:12:04.776185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:12:04.779872Z","iopub.execute_input":"2022-04-09T10:12:04.780114Z","iopub.status.idle":"2022-04-09T10:12:04.878082Z","shell.execute_reply.started":"2022-04-09T10:12:04.780084Z","shell.execute_reply":"2022-04-09T10:12:04.877462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(articles_df)","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:12:04.879367Z","iopub.execute_input":"2022-04-09T10:12:04.879602Z","iopub.status.idle":"2022-04-09T10:12:05.086604Z","shell.execute_reply.started":"2022-04-09T10:12:04.879576Z","shell.execute_reply":"2022-04-09T10:12:05.08603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In articles_df only details_desc has 416 missing values that is 0.394156 %.","metadata":{}},{"cell_type":"code","source":"customers_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:12:05.087357Z","iopub.execute_input":"2022-04-09T10:12:05.087594Z","iopub.status.idle":"2022-04-09T10:12:05.372747Z","shell.execute_reply.started":"2022-04-09T10:12:05.087563Z","shell.execute_reply":"2022-04-09T10:12:05.371832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(customers_df)","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:12:05.374182Z","iopub.execute_input":"2022-04-09T10:12:05.374497Z","iopub.status.idle":"2022-04-09T10:12:06.209454Z","shell.execute_reply.started":"2022-04-09T10:12:05.374457Z","shell.execute_reply":"2022-04-09T10:12:06.208547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Active and FN has more than 50% of missing rest has a minor amount of missing data","metadata":{}},{"cell_type":"code","source":"sample_submission_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:12:06.211123Z","iopub.execute_input":"2022-04-09T10:12:06.211433Z","iopub.status.idle":"2022-04-09T10:12:06.344346Z","shell.execute_reply.started":"2022-04-09T10:12:06.211375Z","shell.execute_reply":"2022-04-09T10:12:06.343541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:12:06.345487Z","iopub.execute_input":"2022-04-09T10:12:06.34567Z","iopub.status.idle":"2022-04-09T10:12:06.355562Z","shell.execute_reply.started":"2022-04-09T10:12:06.345647Z","shell.execute_reply":"2022-04-09T10:12:06.354318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data visualization","metadata":{}},{"cell_type":"code","source":"\ntemp = articles_df.groupby([\"product_group_name\"])[\"product_type_name\"].nunique()\ndf = pd.DataFrame({'Product Group': temp.index,\n                   'Product Types': temp.values\n                  })\n\ndf = df.sort_values(['Product Types'], ascending=False)\nfig, ax = plt.subplots( figsize=(10,8))\n\nax.tick_params(axis='x', which='major', labelsize=15)\nax.tick_params(axis='y', which='major', labelsize=15)\nax.set_ylabel('Product group', weight='semibold', fontsize = 20)\nax.set_xlabel('Prodiuct type', weight='semibold', fontsize = 20)\n\nplt.title('Number of Product Types per each Product Group', weight='semibold', fontsize = 20)\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Product Group', y=\"Product Types\", data=df)\n\nax.grid(b = True, color ='grey',\n        linestyle ='-.', linewidth = 0.5,\n        alpha = 0.6)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-09T10:12:06.356892Z","iopub.execute_input":"2022-04-09T10:12:06.357334Z","iopub.status.idle":"2022-04-09T10:12:06.79112Z","shell.execute_reply.started":"2022-04-09T10:12:06.357247Z","shell.execute_reply":"2022-04-09T10:12:06.790152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stopwords = set(STOPWORDS)\n\ndef show_wordcloud(data, title = None):\n    wordcloud = WordCloud(\n        background_color='white',\n        stopwords=stopwords,\n        max_words=200,\n        max_font_size=40, \n        scale=5,\n        random_state=1\n    ).generate(str(data))\n\n    fig = plt.figure(1, figsize=(10,10))\n    plt.axis('off')\n    if title: \n        fig.suptitle(title, fontsize=14)\n        fig.subplots_adjust(top=2.3)\n\n    plt.imshow(wordcloud)\n    plt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-09T10:12:06.792357Z","iopub.execute_input":"2022-04-09T10:12:06.792703Z","iopub.status.idle":"2022-04-09T10:12:06.80474Z","shell.execute_reply.started":"2022-04-09T10:12:06.792663Z","shell.execute_reply":"2022-04-09T10:12:06.80349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_wordcloud(articles_df[\"prod_name\"], \"Wordcloud from product name\")","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:12:06.806185Z","iopub.execute_input":"2022-04-09T10:12:06.806506Z","iopub.status.idle":"2022-04-09T10:12:07.518083Z","shell.execute_reply.started":"2022-04-09T10:12:06.806466Z","shell.execute_reply":"2022-04-09T10:12:07.517361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nsns.set_theme(style=\"darkgrid\")\ntemp = articles_df.groupby([\"product_group_name\"])[\"product_type_name\"].nunique()\ndf = pd.DataFrame({'Product Group': temp.index,\n                   'Product Types': temp.values\n                  })\n\ndf = df.sort_values(['Product Types'], ascending=False)\n\nf, ax = plt.subplots(figsize=(10, 16))\n\n\nsns.set_color_codes(\"pastel\")\nsns.barplot(x=df['Product Types'],\n                y=df['Product Group'],\n                linewidth=0.5, saturation = 0.5)\n\n\n    \nplt.title('Number of Product Types per each Product Group', weight='bold',fontsize=15)\nplt.xlabel(\"Product type\", weight='semibold',fontsize=15)\nplt.ylabel(\"Product Group\", weight='semibold',fontsize=15)\n","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:34:30.683097Z","iopub.execute_input":"2022-04-09T10:34:30.683834Z","iopub.status.idle":"2022-04-09T10:34:31.12823Z","shell.execute_reply.started":"2022-04-09T10:34:30.683788Z","shell.execute_reply":"2022-04-09T10:34:31.12765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntemp = articles_df.groupby([\"product_type_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Product Type': temp.index,\n                   'Articles': temp.values\n                  })\ntotal_types = len(df['Product Type'].unique())\ndf = df.sort_values(['Articles'], ascending=False)[0:50]\n\nf, ax = plt.subplots(figsize=(10, 16))\n\n\nsns.set_color_codes(\"pastel\")\nsns.barplot(x=df['Articles'],\n                y=df['Product Type'],\n                linewidth=0.5, saturation = 0.5)\n\n\n    \nplt.title(f'Number of Articles per each Product Type (top 50 from total: {total_types})', weight='bold',fontsize=15)\nplt.xlabel(\"Articles\", weight='semibold',fontsize=15)\nplt.ylabel(\"Product Type\", weight='semibold',fontsize=15)\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-09T10:47:56.307383Z","iopub.execute_input":"2022-04-09T10:47:56.307665Z","iopub.status.idle":"2022-04-09T10:47:57.288371Z","shell.execute_reply.started":"2022-04-09T10:47:56.307637Z","shell.execute_reply":"2022-04-09T10:47:57.287415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntemp = articles_df.groupby([\"department_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Department Name': temp.index,\n                   'Articles': temp.values\n                  })\ntotal_depts = len(df['Department Name'].unique())\ndf = df.sort_values(['Articles'], ascending=False).head(50)\n\nf, ax = plt.subplots(figsize=(10, 16))\n\n\nsns.set_color_codes(\"pastel\")\nsns.barplot(x=df['Articles'],\n                y=df['Department Name'],\n                linewidth=0.5, saturation = 0.5)\n\n\n    \nplt.title(f'Number of Articles per each Department (top 50 from total: {total_depts})', weight='bold',fontsize=15)\nplt.xlabel(\"Articles\", weight='semibold',fontsize=15)\nplt.ylabel(\"Department Name\", weight='semibold',fontsize=15)\n","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:47:19.287319Z","iopub.execute_input":"2022-04-09T10:47:19.287974Z","iopub.status.idle":"2022-04-09T10:47:20.35307Z","shell.execute_reply.started":"2022-04-09T10:47:19.287926Z","shell.execute_reply":"2022-04-09T10:47:20.352199Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"index_group_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Index Group Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (6,6))\nplt.title(f'Number of Articles per each Index Group Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Index Group Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:49:23.614166Z","iopub.execute_input":"2022-04-09T10:49:23.614476Z","iopub.status.idle":"2022-04-09T10:49:23.858606Z","shell.execute_reply.started":"2022-04-09T10:49:23.614443Z","shell.execute_reply":"2022-04-09T10:49:23.858005Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntemp = articles_df.groupby([\"colour_group_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Colour Group Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\n\nf, ax = plt.subplots(figsize=(10, 16))\n\n\nsns.set_color_codes(\"pastel\")\nsns.barplot(x=df['Articles'],\n                y=df['Colour Group Name'],\n                linewidth=0.5, saturation = 0.5)\n\n\n    \nplt.title(f'Number of Articles per each Colour Group Name', weight='bold',fontsize=15)\nplt.xlabel(\"Articles\", weight='semibold',fontsize=15)\nplt.ylabel('Colour Group Name', weight='semibold',fontsize=15)\n","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:51:42.028703Z","iopub.execute_input":"2022-04-09T10:51:42.029046Z","iopub.status.idle":"2022-04-09T10:51:43.173162Z","shell.execute_reply.started":"2022-04-09T10:51:42.029011Z","shell.execute_reply":"2022-04-09T10:51:43.172285Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"perceived_colour_value_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Perceived Colour Group Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (6,6))\nplt.title(f'Number of Articles per each Perceived Colour Group Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Perceived Colour Group Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:53:06.461216Z","iopub.execute_input":"2022-04-09T10:53:06.461519Z","iopub.status.idle":"2022-04-09T10:53:06.731549Z","shell.execute_reply.started":"2022-04-09T10:53:06.461485Z","shell.execute_reply":"2022-04-09T10:53:06.73088Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"perceived_colour_master_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Perceived Colour Master Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize = (12,6))\nplt.title(f'Number of Articles per each Perceived Colour Master Name')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Perceived Colour Master Name', y=\"Articles\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:53:31.21754Z","iopub.execute_input":"2022-04-09T10:53:31.218441Z","iopub.status.idle":"2022-04-09T10:53:31.597565Z","shell.execute_reply.started":"2022-04-09T10:53:31.218388Z","shell.execute_reply":"2022-04-09T10:53:31.596473Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"index_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Index Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize=(15,10))\nexplode_values = (0, 0, 0.2, 0, 0, 0, 0.2, 0, 0, 0, 0, 0.2,0.2)\n    \ncircle = plt.Circle((0,0), 0.7, color='white')\nplt.rcParams['text.color'] = 'black'\n    \nplt.pie(df['Articles'],\n        labels=df['Index Name'],\n        shadow=True,autopct='%.1f%%' )\n    \np = plt.gcf()\np.gca().add_artist(circle)\nplt.title(f'Number of Articles per each Index Name', size=20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-09T10:59:25.989581Z","iopub.execute_input":"2022-04-09T10:59:25.989867Z","iopub.status.idle":"2022-04-09T10:59:26.296386Z","shell.execute_reply.started":"2022-04-09T10:59:25.989836Z","shell.execute_reply":"2022-04-09T10:59:26.295416Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = articles_df.groupby([\"garment_group_name\"])[\"article_id\"].nunique()\ndf = pd.DataFrame({'Garment Group Name': temp.index,\n                   'Articles': temp.values\n                  })\ndf = df.sort_values(['Articles'], ascending=False)\nplt.figure(figsize=(15,10))\nexplode_values = (0, 0, 0.2, 0, 0, 0, 0.2, 0, 0, 0, 0, 0.2,0.2)\n    \ncircle = plt.Circle((0,0), 0.7, color='white')\nplt.rcParams['text.color'] = 'black'\n    \nplt.pie(df['Articles'],\n        labels=df['Garment Group Name'],\n        shadow=True,autopct='%.1f%%' )\n    \np = plt.gcf()\np.gca().add_artist(circle)\nplt.title(f'Number of Articles per each Garment Group Name', size=20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-09T11:03:03.618683Z","iopub.execute_input":"2022-04-09T11:03:03.619727Z","iopub.status.idle":"2022-04-09T11:03:04.072452Z","shell.execute_reply.started":"2022-04-09T11:03:03.619681Z","shell.execute_reply":"2022-04-09T11:03:04.071473Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Customers data","metadata":{}},{"cell_type":"code","source":"temp = customers_df.groupby([\"age\"])[\"customer_id\"].count()\ndf = pd.DataFrame({'Age': temp.index,\n                   'Customers': temp.values\n                  })\ndf = df.sort_values(['Age'], ascending=False)\nplt.figure(figsize = (16,6))\nplt.title(f'Number of Customers per each Age')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Age', y=\"Customers\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-09T11:07:23.83216Z","iopub.execute_input":"2022-04-09T11:07:23.83248Z","iopub.status.idle":"2022-04-09T11:07:26.231718Z","shell.execute_reply.started":"2022-04-09T11:07:23.832443Z","shell.execute_reply":"2022-04-09T11:07:26.230731Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = customers_df.groupby([\"fashion_news_frequency\"])[\"customer_id\"].count()\ndf = pd.DataFrame({'Fashion News Frequency': temp.index,\n                   'Customers': temp.values\n                  })\ndf = df.sort_values(['Customers'], ascending=False)\nplt.figure(figsize=(15,10))\nexplode_values = (0, 0, 0.2, 0, 0, 0, 0.2, 0, 0, 0, 0, 0.2,0.2)\n    \ncircle = plt.Circle((0,0), 0.7, color='white')\nplt.rcParams['text.color'] = 'black'\n    \nplt.pie(df['Customers'],\n        labels=df['Fashion News Frequency'],\n        shadow=True,autopct='%.1f%%' )\n    \np = plt.gcf()\np.gca().add_artist(circle)\nplt.title(f'Number of Articles per each Garment Group Name', size=20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-09T11:09:06.032573Z","iopub.execute_input":"2022-04-09T11:09:06.03292Z","iopub.status.idle":"2022-04-09T11:09:06.43258Z","shell.execute_reply.started":"2022-04-09T11:09:06.032882Z","shell.execute_reply":"2022-04-09T11:09:06.431447Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = customers_df.groupby([\"club_member_status\"])[\"customer_id\"].count()\ndf = pd.DataFrame({'Club Member Status': temp.index,\n                   'Customers': temp.values\n                  })\ndf = df.sort_values(['Customers'], ascending=False)\nplt.figure(figsize = (6,6))\nplt.title(f'Number of Customers per each Club Member Status')\nsns.set_color_codes(\"pastel\")\ns = sns.barplot(x = 'Club Member Status', y=\"Customers\", data=df)\ns.set_xticklabels(s.get_xticklabels(),rotation=90)\nlocs, labels = plt.xticks()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-09T11:10:32.456198Z","iopub.execute_input":"2022-04-09T11:10:32.457039Z","iopub.status.idle":"2022-04-09T11:10:32.838302Z","shell.execute_reply.started":"2022-04-09T11:10:32.456999Z","shell.execute_reply":"2022-04-09T11:10:32.837455Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transactions data","metadata":{}},{"cell_type":"code","source":"df = transactions_train_df.sample(100_000)\nfig, ax = plt.subplots(1, 1, figsize=(14, 7))\nsns.kdeplot(np.log(df.loc[df[\"sales_channel_id\"]==1].price.value_counts()))\nsns.kdeplot(np.log(df.loc[df[\"sales_channel_id\"]==2].price.value_counts()))\nax.legend(labels=['Sales channel 1', 'Sales channel 1'])\nplt.title(\"Logaritmic distribution of price frequency in transactions, grouped per sales channel (100k sample)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-09T11:11:50.953376Z","iopub.execute_input":"2022-04-09T11:11:50.953683Z","iopub.status.idle":"2022-04-09T11:11:53.525782Z","shell.execute_reply.started":"2022-04-09T11:11:50.953652Z","shell.execute_reply":"2022-04-09T11:11:53.5249Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf = transactions_train_df.sample(100_000).groupby([\"t_dat\", \"sales_channel_id\"])[\"article_id\"].count().reset_index()\ndf[\"t_dat\"] = df[\"t_dat\"].apply(lambda x: datetime.strptime(x, '%Y-%m-%d'))\ndf.columns = [\"Date\", \"Sales Channel Id\", \"Transactions\"]\nfig, ax = plt.subplots(1, 1, figsize=(16,6))\ng1 = ax.plot(df.loc[df[\"Sales Channel Id\"]==1, \"Date\"], df.loc[df[\"Sales Channel Id\"]==1, \"Transactions\"], label=\"Sales Channel 1\", color=\"Darkblue\")\ng2 = ax.plot(df.loc[df[\"Sales Channel Id\"]==2, \"Date\"], df.loc[df[\"Sales Channel Id\"]==2, \"Transactions\"], label=\"Sales Channel 2\", color=\"Magenta\")\nplt.xlabel(\"Date\")\nplt.ylabel(\"Transactions\")\nax.legend()\nplt.title(f\"Transactions per day, grouped by Sales Channel (100k sample; to get the real volume, please consider that real transaction count is {round(transactions_train_df.shape[0]/10.e6,2)}M)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-09T11:12:20.445236Z","iopub.execute_input":"2022-04-09T11:12:20.445552Z","iopub.status.idle":"2022-04-09T11:12:22.845984Z","shell.execute_reply.started":"2022-04-09T11:12:20.445513Z","shell.execute_reply":"2022-04-09T11:12:22.845173Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = transactions_train_df.groupby([\"t_dat\", \"sales_channel_id\"])[\"article_id\"].nunique().reset_index()\ndf[\"t_dat\"] = df[\"t_dat\"].apply(lambda x: datetime.strptime(x, '%Y-%m-%d'))\ndf.columns = [\"Date\", \"Sales Channel Id\", \"Unique Articles\"]\nfig, ax = plt.subplots(1, 1, figsize=(16,6))\ng1 = ax.plot(df.loc[df[\"Sales Channel Id\"]==1, \"Date\"], df.loc[df[\"Sales Channel Id\"]==1, \"Unique Articles\"], label=\"Sales Channel 1\", color=\"Blue\")\ng2 = ax.plot(df.loc[df[\"Sales Channel Id\"]==2, \"Date\"], df.loc[df[\"Sales Channel Id\"]==2, \"Unique Articles\"], label=\"Sales Channel 2\", color=\"Green\")\nplt.xlabel(\"Date\")\nplt.ylabel(\"Unique Articles / Day\")\nax.legend()\nplt.title(f\"Unique articles per day, grouped by Sales Channel\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-09T11:12:42.469367Z","iopub.execute_input":"2022-04-09T11:12:42.469952Z","iopub.status.idle":"2022-04-09T11:12:58.419416Z","shell.execute_reply.started":"2022-04-09T11:12:42.469913Z","shell.execute_reply":"2022-04-09T11:12:58.418472Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = transactions_train_df.sample(100_000).groupby([\"t_dat\"])[\"article_id\"].count().reset_index()\ndf[\"t_dat\"] = df[\"t_dat\"].apply(lambda x: datetime.strptime(x, '%Y-%m-%d'))\ndf.columns = [\"Date\", \"Transactions\"]\nfig, ax = plt.subplots(1, 1, figsize=(16,6))\nplt.plot(df[\"Date\"], df[\"Transactions\"], color=\"Darkgreen\")\nplt.xlabel(\"Date\")\nplt.ylabel(\"Transactions\")\nplt.title(f\"Transactions per day (100k sample; to get the real volume, please consider that real transaction count is {round(transactions_train_df.shape[0]/10.e6,2)}M)\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-04-09T11:13:51.1553Z","iopub.execute_input":"2022-04-09T11:13:51.155654Z","iopub.status.idle":"2022-04-09T11:13:53.409021Z","shell.execute_reply.started":"2022-04-09T11:13:51.155621Z","shell.execute_reply":"2022-04-09T11:13:53.408117Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To be continued.......","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}