{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<img src=\"https://upload.wikimedia.org/wikipedia/he/thumb/5/52/H%26M_logo.svg/1200px-H%26M_logo.svg.png\">\n\n<center><h1>- Data Exploration -</h1></center>\n\n>  🛍️ **Competition Goal:** For each customer within the training data we need to predict up to 12 products that the customer will buy in the next 7-day period *after* the training time period. We can predict up to *12 products* that the customer will likely be purchasing in the 7-day period.\n\n# 1.Importing Libraries","metadata":{}},{"cell_type":"code","source":"# Libraries\nimport os\nimport gc\nimport wandb\nimport time\nimport random\nimport math\nimport glob\nfrom scipy import spatial\nfrom tqdm import tqdm\nimport warnings\nimport cv2\nimport pandas as pd\nimport numpy as np\nfrom numpy import dot, sqrt\nimport seaborn as sns\nimport matplotlib as mpl\nimport matplotlib.patches as patches\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom matplotlib.offsetbox import AnnotationBbox, OffsetImage\nfrom IPython.display import display_html\nfrom wordcloud import WordCloud, STOPWORDS\nfrom PIL import Image\nplt.rcParams.update({'font.size': 16})\n\n# Environment check\nwarnings.filterwarnings(\"ignore\")\nos.environ[\"WANDB_SILENT\"] = \"true\"\nCONFIG = {'competition': 'HandM', '_wandb_kernel': 'aot'}\n\n# Custom colors\nclass clr:\n    S = '\\033[1m' + '\\033[95m'\n    E = '\\033[0m'\n    \nmy_colors = [\"#AF0848\", \"#E90B60\", \"#CB2170\", \"#954E93\", \"#705D98\", \"#5573A8\", \"#398BBB\", \"#00BDE3\"]\nprint(clr.S+\"Notebook Color Scheme:\"+clr.E)\nsns.palplot(sns.color_palette(my_colors))\nplt.show()\n\nbk_image = plt.imread(\"../input/hm-fashion-recommender-dataset/background.jpg\")","metadata":{"execution":{"iopub.status.busy":"2022-04-04T15:16:01.145045Z","iopub.execute_input":"2022-04-04T15:16:01.14583Z","iopub.status.idle":"2022-04-04T15:16:03.325305Z","shell.execute_reply.started":"2022-04-04T15:16:01.14574Z","shell.execute_reply":"2022-04-04T15:16:03.324521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 🐝 W&B Fork & Run\n\nIn order to run this notebook you will need to input your own **secret API key** within the `! wandb login $secret_value_0` line. \n\n🐝**How do you get your own API key?**\n\nSuper simple! Go to **https://wandb.ai/site** -> Login -> Click on your profile in the top right corner -> Settings -> Scroll down to API keys -> copy your very own key (for more info check [this amazing notebook for ML Experiment Tracking on Kaggle](https://www.kaggle.com/ayuraj/experiment-tracking-with-weights-and-biases)).\n\n<center><img src=\"https://i.imgur.com/fFccmoS.png\" width=500></center>","metadata":{}},{"cell_type":"code","source":"# 🐝 Secrets\nfrom kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\nsecret_value_0 = user_secrets.get_secret(\"wandb\")\n\n! wandb login $secret_value_0","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.248088Z","iopub.status.idle":"2022-04-02T11:32:58.248528Z","shell.execute_reply.started":"2022-04-02T11:32:58.248299Z","shell.execute_reply":"2022-04-02T11:32:58.248323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### ⬇ Helper Functions","metadata":{}},{"cell_type":"code","source":"def adjust_id(x):\n    '''Adjusts article ID code.'''\n    x = str(x)\n    if len(x) == 9:\n        x = \"0\"+x\n    \n    return x\n\n\ndef insert_image(path, zoom, xybox, ax):\n    '''Insert an image within matplotlib'''\n    imagebox = OffsetImage(mpimg.imread(path), zoom=zoom)\n    ab = AnnotationBbox(imagebox, xy=(0.5, 0.7), frameon=False, pad=1, xybox=xybox)\n    ax.add_artist(ab)\n    \n    \ndef show_values_on_bars(axs, h_v=\"v\", space=0.4):\n    '''Plots the value at the end of the a seaborn barplot.\n    axs: the ax of the plot\n    h_v: weather or not the barplot is vertical/ horizontal'''\n    \n    def _show_on_single_plot(ax):\n        if h_v == \"v\":\n            for p in ax.patches:\n                _x = p.get_x() + p.get_width() / 2\n                _y = p.get_y() + p.get_height()\n                value = int(p.get_height())\n                ax.text(_x, _y, format(value, ','), ha=\"center\") \n        elif h_v == \"h\":\n            for p in ax.patches:\n                _x = p.get_x() + p.get_width() + float(space)\n                _y = p.get_y() + p.get_height()\n                value = int(p.get_width())\n                ax.text(_x, _y, format(value, ','), ha=\"left\")\n\n    if isinstance(axs, np.ndarray):\n        for idx, ax in np.ndenumerate(axs):\n            _show_on_single_plot(ax)\n    else:\n        _show_on_single_plot(axs)\n\n\n# === 🐝 W&B ===\ndef save_dataset_artifact(run_name, artifact_name, path):\n    '''Saves dataset to W&B Artifactory.\n    run_name: name of the experiment\n    artifact_name: under what name should the dataset be stored\n    path: path to the dataset'''\n    \n    run = wandb.init(project='HandM', \n                     name=run_name, \n                     config=CONFIG)\n    artifact = wandb.Artifact(name=artifact_name, \n                              type='dataset')\n    artifact.add_file(path)\n\n    wandb.log_artifact(artifact)\n    wandb.finish()\n    print(\"Artifact has been saved successfully.\")\n    \n    \ndef create_wandb_plot(x_data=None, y_data=None, x_name=None, y_name=None, title=None, log=None, plot=\"line\"):\n    '''Create and save lineplot/barplot in W&B Environment.\n    x_data & y_data: Pandas Series containing x & y data\n    x_name & y_name: strings containing axis names\n    title: title of the graph\n    log: string containing name of log'''\n    \n    data = [[label, val] for (label, val) in zip(x_data, y_data)]\n    table = wandb.Table(data=data, columns = [x_name, y_name])\n    \n    if plot == \"line\":\n        wandb.log({log : wandb.plot.line(table, x_name, y_name, title=title)})\n    elif plot == \"bar\":\n        wandb.log({log : wandb.plot.bar(table, x_name, y_name, title=title)})\n    elif plot == \"scatter\":\n        wandb.log({log : wandb.plot.scatter(table, x_name, y_name, title=title)})\n        \n        \ndef create_wandb_hist(x_data=None, x_name=None, title=None, log=None):\n    '''Create and save histogram in W&B Environment.\n    x_data: Pandas Series containing x values\n    x_name: strings containing axis name\n    title: title of the graph\n    log: string containing name of log'''\n    \n    data = [[x] for x in x_data]\n    table = wandb.Table(data=data, columns=[x_name])\n    wandb.log({log : wandb.plot.histogram(table, x_name, title=title)})\n    \n    \n# 🐝 Log Cover Photo\nrun = wandb.init(project='HandM', name='CoverPhoto', config=CONFIG)\ncover = plt.imread(\"../input/hm-fashion-recommender-dataset/pics/Kaggle Covers.png\")\nwandb.log({\"example\": wandb.Image(cover)})\nwandb.finish()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-02T11:32:58.249957Z","iopub.status.idle":"2022-04-02T11:32:58.250389Z","shell.execute_reply.started":"2022-04-02T11:32:58.25016Z","shell.execute_reply":"2022-04-02T11:32:58.250183Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Dataset\n\n🛍️ **There are 3 metadata .csv files and 1 image file:**\n* `images` - folder containing the photo of *almost* all `article_ids`\n* `articles.csv` - description features of all `article_ids` **(105,542 datapoints)**\n* `customers.csv` - description features of the customer profiles **(1,371,980 datapoints)**\n* `transactions_train.csv` - file containing the `customer_id`, the article that was bought and at what price **(31,788,324 datapoints)**","metadata":{}},{"cell_type":"code","source":"%%time\n\n# Read in the data\narticles = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\ncustomers = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/customers.csv\")\ntransactions = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")\nss = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.251745Z","iopub.status.idle":"2022-04-02T11:32:58.25219Z","shell.execute_reply.started":"2022-04-02T11:32:58.251946Z","shell.execute_reply":"2022-04-02T11:32:58.251969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(clr.S+\"ARTICLES:\"+clr.E, articles.shape)\ndisplay_html(articles.head(3))\nprint(\"\\n\", clr.S+\"CUSTOMERS:\"+clr.E, customers.shape)\ndisplay_html(customers.head(3))\nprint(\"\\n\", clr.S+\"TRANSACTIONS:\"+clr.E, transactions.shape)\ndisplay_html(transactions.head(3))\nprint(\"\\n\", clr.S+\"SAMPLE_SUBMISSION:\"+clr.E, ss.shape)\ndisplay_html(ss.head(3))","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.253679Z","iopub.status.idle":"2022-04-02T11:32:58.254111Z","shell.execute_reply.started":"2022-04-02T11:32:58.253879Z","shell.execute_reply":"2022-04-02T11:32:58.253903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Articles\n\n## I. Preprocessing\n\n🛍️ **Important Notes**:\n* There are *more* `article_ids` than actual images:\n    * unique article ids: 105,542\n    * unique images: 105,100\n* The `path` processing was taking too long, so the fastest (takes 1 second) way to do it was to create a variable that contains all article ids within the `images` folder (remember, `set()` is faster than a `list`), and then to correct any path that was invalid within the `articles.csv` file.\n* There are only 416 missing values within the `desc` column - product description","metadata":{}},{"cell_type":"code","source":"# 🐝 W&B Experiment\nrun = wandb.init(project='HandM', name='Articles', config=CONFIG)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.255618Z","iopub.status.idle":"2022-04-02T11:32:58.256051Z","shell.execute_reply.started":"2022-04-02T11:32:58.255822Z","shell.execute_reply":"2022-04-02T11:32:58.255846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(clr.S+\"There are no missing values in any columns but 'Detail Description':\"+clr.E,\n      articles.isna().sum()[-1], \"total missing values\")\n\n# Replace missing values\narticles.fillna(value=\"No Description\", inplace=True)\n\n# Adjust the article ID and product code to be string & add \"0\"\narticles[\"article_id\"] = articles[\"article_id\"].apply(lambda x: adjust_id(x))\narticles[\"product_code\"] = articles[\"article_id\"].apply(lambda x: x[:3])","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.257439Z","iopub.status.idle":"2022-04-02T11:32:58.257871Z","shell.execute_reply.started":"2022-04-02T11:32:58.257632Z","shell.execute_reply":"2022-04-02T11:32:58.257655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get all paths from the image folder\nall_image_paths = glob.glob(f\"../input/h-and-m-personalized-fashion-recommendations/images/*/*\")\n\nprint(clr.S+\"Number of unique article_ids within articles.csv:\"+clr.E, len(articles), \"\\n\"+\n      clr.S+\"Number of unique images within the image folder:\"+clr.E, len(all_image_paths), \"\\n\"+\n      clr.S+\"=> not all article_ids have a corresponding image!!!\"+clr.E, \"\\n\")\n\n# 🐝 Log Distinct article IDs\nwandb.log({\"article_ids\":len(articles)})\n\n# Get all valid article ids\n# Create a set() - as it moves faster than a list\nall_image_ids = set()\n\nfor path in tqdm(all_image_paths):\n    article_id = path.split('/')[-1].split('.')[0]\n    all_image_ids.add(article_id)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.259897Z","iopub.status.idle":"2022-04-02T11:32:58.260899Z","shell.execute_reply.started":"2022-04-02T11:32:58.260642Z","shell.execute_reply":"2022-04-02T11:32:58.260668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# An image path example: ../input/h-and-m-personalized-fashion-recommendations/images/010/0108775015.jpg\n\n# Create full path to the article image\nimages_path = \"../input/h-and-m-personalized-fashion-recommendations/images/\"\narticles[\"path\"] = images_path + articles[\"product_code\"] + \"/\" + articles[\"article_id\"] + \".jpg\"\n\n# Adjust the incorrect paths and set them to None\nfor k, article_id in tqdm(enumerate(articles[\"article_id\"])):\n    if article_id not in all_image_ids:\n        articles.loc[k, \"path\"] = None","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.262035Z","iopub.status.idle":"2022-04-02T11:32:58.262947Z","shell.execute_reply.started":"2022-04-02T11:32:58.262703Z","shell.execute_reply":"2022-04-02T11:32:58.262729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## II. Explore","metadata":{}},{"cell_type":"code","source":"print(clr.S+\"Total Number of unique Product Names:\"+clr.E, articles[\"prod_name\"].nunique())\n\n# Data\nprod_name = articles[\"prod_name\"].value_counts().reset_index().head(15)\ntotal_prod_names = articles[\"prod_name\"].nunique()\nclrs = [\"#CB2170\" if x==max(prod_name[\"prod_name\"]) else '#954E93' for x in prod_name[\"prod_name\"]]\n\n# Get images\nprod_name_images = articles[articles[\"prod_name\"].isin(prod_name[\"index\"].tolist())].groupby(\"prod_name\")[\"path\"].first().reset_index()\nimage_paths = prod_name_images[\"path\"].tolist()\nimage_names = prod_name_images[\"prod_name\"].tolist()\n\n# Plot\nfig, ax = plt.subplots(figsize=(25, 13))\nplt.title('- Most Frequent Product Names -', size=22, weight=\"bold\")\n\nsns.barplot(data=prod_name, x=\"prod_name\", y=\"index\", ax=ax,\n            palette=clrs)\nx0,x1 = ax.get_xlim()\ny0,y1 = ax.get_ylim()\nplt.imshow(bk_image, zorder=0, extent=[x0, x1, y0, y1], alpha=0.35, aspect='auto')\n\nshow_values_on_bars(axs=ax, h_v=\"h\", space=0.4)\nplt.ylabel(\"Product Name\", size = 16, weight=\"bold\")\nplt.xlabel(\"\")\nplt.xticks([])\nplt.yticks(size=16)\nplt.tick_params(size=16)\n\ninsert_image(path='../input/hm-fashion-recommender-dataset/pics/dragonfly.jpg', zoom=0.45, xybox=(92, 11), ax=ax)\n\nsns.despine(left=True, bottom=True)\nplt.show();\n\nprint(\"\\n\")\n\n# Plot\nfig, axs = plt.subplots(3, 5, figsize=(23, 8))\nfig.suptitle('- Example Images -', size=22, weight=\"bold\")\naxs = axs.flatten()\n\nfor k, (path, name) in enumerate(zip(image_paths, image_names)):\n    axs[k].set_title(f\"{name}\", size = 16)\n    img = plt.imread(path)\n    axs[k].imshow(img)\n    axs[k].axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-02T11:32:58.264099Z","iopub.status.idle":"2022-04-02T11:32:58.265008Z","shell.execute_reply.started":"2022-04-02T11:32:58.264764Z","shell.execute_reply":"2022-04-02T11:32:58.264789Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Log Barplot to W&B\ncreate_wandb_plot(x_data=prod_name[\"index\"], y_data=prod_name[\"prod_name\"],\n                  x_name=\"Product Name\", y_name=\"Frequency\", \n                  title=\"- Most Frequent Product Names -\", log=\"prod_name\", plot=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.26612Z","iopub.status.idle":"2022-04-02T11:32:58.26704Z","shell.execute_reply.started":"2022-04-02T11:32:58.266798Z","shell.execute_reply":"2022-04-02T11:32:58.266823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"code","source":"print(clr.S+\"Total Number of unique Product Types:\"+clr.E, articles[\"product_type_name\"].nunique())\n\n# Data\nprod_type = articles[\"product_type_name\"].value_counts().reset_index().head(15)\ntotal_prod_types = articles[\"product_type_name\"].nunique()\nclrs = [\"#00BDE3\" if x==max(prod_type[\"product_type_name\"]) else '#398BBB' for x in prod_type[\"product_type_name\"]]\n\n# Get images\nprod_type_images = articles[articles[\"product_type_name\"].isin(prod_type[\"index\"].tolist())].groupby(\"product_type_name\")[\"path\"].first().reset_index()\nimage_paths = prod_type_images[\"path\"].tolist()\nimage_names = prod_type_images[\"product_type_name\"].tolist()\n\n# Plot\nfig, ax = plt.subplots(figsize=(25, 13))\nplt.title('- Most Frequent Product Types -', size=22, weight=\"bold\")\n\nsns.barplot(data=prod_type, x=\"product_type_name\", y=\"index\", ax=ax,\n            palette=clrs)\nx0,x1 = ax.get_xlim()\ny0,y1 = ax.get_ylim()\nplt.imshow(bk_image, zorder=0, extent=[x0, x1, y0, y1], alpha=0.35, aspect='auto')\n\nshow_values_on_bars(axs=ax, h_v=\"h\", space=0.4)\nplt.ylabel(\"Product Type\", size = 16, weight=\"bold\")\nplt.xlabel(\"\")\nplt.xticks([])\nplt.yticks(size=16)\nplt.tick_params(size=16)\n\ninsert_image(path='../input/hm-fashion-recommender-dataset/pics/blue.jpg', zoom=0.45, xybox=(11000, 11), ax=ax)\n\nsns.despine(left=True, bottom=True)\nplt.show();\n\nprint(\"\\n\")\n\n# Plot\nfig, axs = plt.subplots(3, 5, figsize=(23, 8))\nfig.suptitle('- Example Images -', size=22, weight=\"bold\")\naxs = axs.flatten()\n\nfor k, (path, name) in enumerate(zip(image_paths, image_names)):\n    axs[k].set_title(f\"{name}\", size = 16)\n    img = plt.imread(path)\n    axs[k].imshow(img)\n    axs[k].axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-02T11:32:58.268215Z","iopub.status.idle":"2022-04-02T11:32:58.270917Z","shell.execute_reply.started":"2022-04-02T11:32:58.270653Z","shell.execute_reply":"2022-04-02T11:32:58.27068Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Log Barplot to W&B\ncreate_wandb_plot(x_data=prod_type[\"index\"], y_data=prod_type[\"product_type_name\"],\n                  x_name=\"Product Type\", y_name=\"Frequency\", \n                  title=\"- Most Frequent Product Types -\", log=\"prod_type\", plot=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.272298Z","iopub.status.idle":"2022-04-02T11:32:58.273216Z","shell.execute_reply.started":"2022-04-02T11:32:58.272952Z","shell.execute_reply":"2022-04-02T11:32:58.272977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"code","source":"print(clr.S+\"Total Number of unique Product Group:\"+clr.E, articles[\"product_group_name\"].nunique())\n\n# Data\nprod_group = articles[\"product_group_name\"].value_counts().reset_index()\ntotal_prod_groups = articles[\"product_group_name\"].nunique()\nclrs = [\"#E90B60\" if x==max(prod_group[\"product_group_name\"]) else '#AF0848' for x in prod_group[\"product_group_name\"]]\n\n# Get images\nprod_group_images = articles[articles[\"product_group_name\"].isin(prod_group[\"index\"].tolist())].groupby(\"product_group_name\")[\"path\"].first().reset_index()\nimage_paths = prod_group_images[\"path\"].tolist()\nimage_names = prod_group_images[\"product_group_name\"].tolist()\n\n# Plot\nfig, ax = plt.subplots(figsize=(25, 13))\nplt.title('- Most Frequent Product Groups -', size=22, weight=\"bold\")\n\nsns.barplot(data=prod_group, x=\"product_group_name\", y=\"index\", ax=ax,\n            palette=clrs)\nx0,x1 = ax.get_xlim()\ny0,y1 = ax.get_ylim()\nplt.imshow(bk_image, zorder=0, extent=[x0, x1, y0, y1], alpha=0.35, aspect='auto')\n\nshow_values_on_bars(axs=ax, h_v=\"h\", space=0.4)\nplt.ylabel(\"Product Group\", size = 16, weight=\"bold\")\nplt.xlabel(\"\")\nplt.xticks([])\nplt.yticks(size=16)\nplt.tick_params(size=16)\n\ninsert_image(path='../input/hm-fashion-recommender-dataset/pics/chloe.jpg', zoom=0.45, xybox=(40000, 14), ax=ax)\n\nsns.despine(left=True, bottom=True)\nplt.show();\n\nprint(\"\\n\")\n\n# Plot\nfig, axs = plt.subplots(4, 6, figsize=(23, 10))\nfig.suptitle('- Example Images -', size=22, weight=\"bold\")\naxs = axs.flatten()\n\nfor k, (path, name) in enumerate(zip(image_paths, image_names)):\n    axs[k].set_title(f\"{name}\", size = 16)\n    img = plt.imread(path)\n    axs[k].imshow(img)\n    axs[k].axis(\"off\")\n\nfor a in [-1, -2, -3, -4, -5]: axs[a].set_visible(False)\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-02T11:32:58.274392Z","iopub.status.idle":"2022-04-02T11:32:58.275328Z","shell.execute_reply.started":"2022-04-02T11:32:58.275051Z","shell.execute_reply":"2022-04-02T11:32:58.275076Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Log Barplot to W&B\ncreate_wandb_plot(x_data=prod_group[\"index\"], y_data=prod_group[\"product_group_name\"],\n                  x_name=\"Product Group\", y_name=\"Frequency\", \n                  title=\"- Most Frequent Product Group -\", log=\"prod_group\", plot=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.276468Z","iopub.status.idle":"2022-04-02T11:32:58.277415Z","shell.execute_reply.started":"2022-04-02T11:32:58.277159Z","shell.execute_reply":"2022-04-02T11:32:58.277186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"code","source":"def change_color(x):\n    '''Change color name.'''\n    if (\"light\" in x.lower().strip()) or \\\n        (\"dark\" in x.lower().strip()) or \\\n        (\"greyish\" in x.lower().strip()) or \\\n        (\"yellowish\" in x.lower().strip()) or \\\n        (\"greenish\" in x.lower().strip()) or \\\n        (\"off\" in x.lower().strip()) or \\\n        (\"other\" in x.lower().strip()):\n        x = x.split(\" \")[-1]\n        \n    return x\n\narticles[\"colour_group_name\"] = articles[\"colour_group_name\"].apply(lambda x: change_color(x))","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.278554Z","iopub.status.idle":"2022-04-02T11:32:58.279461Z","shell.execute_reply.started":"2022-04-02T11:32:58.279206Z","shell.execute_reply":"2022-04-02T11:32:58.279232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Appearance and color\nprint(clr.S+\"Total Number of unique Product Appearances:\"+clr.E, articles[\"graphical_appearance_name\"].nunique())\nprint(clr.S+\"Total Number of unique Product Colors (after preprocess):\"+clr.E, articles[\"colour_group_name\"].nunique())\n\n# --- Data 1 ---\nprod_appearance = articles[\"graphical_appearance_name\"].value_counts().reset_index().head(15)\ntotal_prod_appearances = articles[\"graphical_appearance_name\"].nunique()\nclrs1 = [\"#AF0848\" if x==max(prod_appearance[\"graphical_appearance_name\"]) else '#E90B60' for x in prod_appearance[\"graphical_appearance_name\"]]\n\n\n# Get images\nprod_appearance_images = articles[articles[\"graphical_appearance_name\"].isin(prod_appearance[\"index\"].tolist())].groupby(\"graphical_appearance_name\")[\"path\"].first().reset_index()\nimage_paths1 = prod_appearance_images[\"path\"].tolist()\nimage_names1 = prod_appearance_images[\"graphical_appearance_name\"].tolist()\n\n# --- Data 2 ---\nprod_color = articles[\"colour_group_name\"].value_counts().reset_index().head(15)\ntotal_prod_color = articles[\"colour_group_name\"].nunique()\nclrs2 = [\"#CB2170\" if x==max(prod_color[\"colour_group_name\"]) else '#954E93' for x in prod_color[\"colour_group_name\"]]\n\n# Get images\nprod_color_images = articles[articles[\"colour_group_name\"].isin(prod_color[\"index\"].tolist())].groupby(\"colour_group_name\")[\"path\"].first().reset_index()\nimage_paths2 = prod_color_images[\"path\"].tolist()\nimage_names2 = prod_color_images[\"colour_group_name\"].tolist()\n\n# Plot\nfig, (ax1, ax2) = plt.subplots(nrows=1, ncols=2, figsize=(25, 13))\n\nax1.set_title('- Most Frequent Product Appearances -', size=22, weight=\"bold\")\nsns.barplot(data=prod_appearance, x=\"graphical_appearance_name\", y=\"index\", ax=ax1,\n            palette=clrs2)\nx0,x1 = ax1.get_xlim()\ny0,y1 = ax1.get_ylim()\nax1.imshow(bk_image, zorder=0, extent=[x0, x1, y0, y1], alpha=0.35, aspect='auto')\n\nshow_values_on_bars(axs=ax1, h_v=\"h\", space=0.4)\nax1.set_ylabel(\"Product Appearance\", size = 16, weight=\"bold\")\nax1.set_xlabel(\"\")\nax1.set_xticks([])\n# ax1.set_yticks(size=16)\n# ax1.set_tick_params(size=16)\n\n# insert_image(path='../input/hm-fashion-recommender-dataset/pics/blue.jpg', zoom=0.45, xybox=(11000, 11), ax=ax1)\n\n\nax2.set_title('- Most Frequent Product Colors -', size=22, weight=\"bold\")\nsns.barplot(data=prod_color, x=\"colour_group_name\", y=\"index\", ax=ax2,\n            palette=clrs2)\nx0,x1 = ax2.get_xlim()\ny0,y1 = ax2.get_ylim()\nax2.imshow(bk_image, zorder=0, extent=[x0, x1, y0, y1], alpha=0.35, aspect='auto')\n\nshow_values_on_bars(axs=ax2, h_v=\"h\", space=0.4)\nax2.set_ylabel(\"Product Colors\", size = 16, weight=\"bold\")\nax2.set_xlabel(\"\")\nax2.set_xticks([])\n# ax1.set_yticks(size=16)\n# ax1.set_tick_params(size=16)\n\n# insert_image(path='../input/hm-fashion-recommender-dataset/pics/blue.jpg', zoom=0.45, xybox=(11000, 11), ax=ax1)\n\nsns.despine(left=True, bottom=True)\nplt.show();\n\nprint(\"\\n\")\n\n# Plot\nfig, axs = plt.subplots(3, 5, figsize=(23, 8))\nfig.suptitle('- Example Images [Appearance] -', size=22, weight=\"bold\")\naxs = axs.flatten()\n\nfor k, (path, name) in enumerate(zip(image_paths1, image_names1)):\n    axs[k].set_title(f\"{name}\", size = 16)\n    img = plt.imread(path)\n    axs[k].imshow(img)\n    axs[k].axis(\"off\")\n\nplt.tight_layout()\nplt.show()\n\n# Plot\nfig, axs = plt.subplots(3, 5, figsize=(23, 8))\nfig.suptitle('- Example Images [Color] -', size=22, weight=\"bold\")\naxs = axs.flatten()\n\nfor k, (path, name) in enumerate(zip(image_paths2, image_names2)):\n    axs[k].set_title(f\"{name}\", size = 16)\n    img = plt.imread(path)\n    axs[k].imshow(img)\n    axs[k].axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-02T11:32:58.280705Z","iopub.status.idle":"2022-04-02T11:32:58.281486Z","shell.execute_reply.started":"2022-04-02T11:32:58.281231Z","shell.execute_reply":"2022-04-02T11:32:58.281257Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Log Barplot to W&B\ncreate_wandb_plot(x_data= prod_appearance[\"index\"], y_data=prod_appearance[\"graphical_appearance_name\"],\n                  x_name=\"Product Appearance\", y_name=\"Frequency\", \n                  title=\"- Most Frequent Product Appearance -\", log=\"prod_appearance\", plot=\"bar\")\n\ncreate_wandb_plot(x_data= prod_color[\"index\"], y_data=prod_color[\"colour_group_name\"],\n                  x_name=\"Product Color\", y_name=\"Frequency\", \n                  title=\"- Most Frequent Product Color -\", log=\"prod_color\", plot=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.282846Z","iopub.status.idle":"2022-04-02T11:32:58.283771Z","shell.execute_reply.started":"2022-04-02T11:32:58.283526Z","shell.execute_reply":"2022-04-02T11:32:58.283552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---\n\n🛍️ **Important Notes**:\n* In order for the wordcloud to take the shape of the image you should input a `.jpg` image with **white** background (not black and not transparent - because the function will interpret the transparent background as black).\n* More custom fonts like I used below can be found here: https://www.dafont.com/","metadata":{}},{"cell_type":"code","source":"def similar_color_func(word=None, font_size=None,\n                       position=None, orientation=None,\n                       font_path=None, random_state=None):\n    '''Creates a custom function for the color of the wordcloud.'''\n    \n    h = 270 # 0 - 360 <- the color hue\n    s = 40 # 0-100 <- the color saturation\n    l = random_state.randint(30, 70) # 0 - 100 <- gradient\n    \n    return \"hsl({}, {}%, {}%)\".format(h, s, l)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.284882Z","iopub.status.idle":"2022-04-02T11:32:58.285565Z","shell.execute_reply.started":"2022-04-02T11:32:58.285314Z","shell.execute_reply":"2022-04-02T11:32:58.28534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(clr.S+\"Total Number of unique Article Descriptions:\"+clr.E, articles[\"detail_desc\"].nunique(), \"\\n\")\n\n# Get descriptions and convert them to a string\ntext = articles[\"detail_desc\"].unique()\ntext = \" \".join(text)\n\n# Get the mask - the form of the wordcloud\nmask = np.array(Image.open('../input/hm-fashion-recommender-dataset/pics/mask.jpg'))\n\n# Create wordcloud object\nwc = WordCloud(mask=mask, background_color=\"white\", max_words=2000,\n               stopwords=STOPWORDS, max_font_size=256,\n               random_state=42, width=mask.shape[1],\n               height=mask.shape[0], font_path=\"../input/hm-fashion-recommender-dataset/MorningRainbow.ttf\",\n               color_func=similar_color_func)\nwc.generate(text)\n\n# Plot\nfig = plt.figure(figsize=(15, 15))\nplt.title(\"- Most Common Words found within Article Descriptions -\",\n           size=22, weight=\"bold\")\nplt.imshow(wc, interpolation=\"bilinear\")\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.286932Z","iopub.status.idle":"2022-04-02T11:32:58.287707Z","shell.execute_reply.started":"2022-04-02T11:32:58.287448Z","shell.execute_reply":"2022-04-02T11:32:58.287474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Save wordcloud to Dashboard\nfig.canvas.draw()\nimage_from_plot = np.frombuffer(fig.canvas.tostring_rgb(), dtype=np.uint8)\nimage_from_plot = image_from_plot.reshape(fig.canvas.get_width_height()[::-1] + (3,))\n\nwandb.log({\"wordcloud\": wandb.Image(image_from_plot)})\n\nwandb.finish()","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.289056Z","iopub.status.idle":"2022-04-02T11:32:58.289743Z","shell.execute_reply.started":"2022-04-02T11:32:58.289498Z","shell.execute_reply":"2022-04-02T11:32:58.289525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Save the updated `articles` file.\n# articles.to_parquet('articles.pqt', index=False)\n\nsave_dataset_artifact(run_name=\"save_articles\", artifact_name=\"articles\",\n                      path=\"../input/hm-fashion-recommender-dataset/articles.pqt\")","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.291007Z","iopub.status.idle":"2022-04-02T11:32:58.291659Z","shell.execute_reply.started":"2022-04-02T11:32:58.291414Z","shell.execute_reply":"2022-04-02T11:32:58.29144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Customers\n\n🛍️ **Important Notes**:\n* In this dataset we have quite a few missing values:\n    * for columns `FN` and `Active` I replaced all missing values with 0\n    * for `club_member_status` and `fashion_news_frequency` I have set all missing values with `UNKNOWN`\n    * for `age` I have imputed all missing values with the median age (which is 36)\n* I have created an `age_interval` as well that splits all ages in decades","metadata":{}},{"cell_type":"code","source":"# 🐝 W&B Experiment\nrun = wandb.init(project='HandM', name='Customers', config=CONFIG)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.292889Z","iopub.status.idle":"2022-04-02T11:32:58.293505Z","shell.execute_reply.started":"2022-04-02T11:32:58.293284Z","shell.execute_reply":"2022-04-02T11:32:58.293308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_age_interval(x):\n    if x <= 25:\n        return [16, 25]\n    elif x <= 35:\n        return [26, 35]\n    elif x <= 45:\n        return [36, 45]\n    elif x <= 55:\n        return [46, 55]\n    elif x <= 65:\n        return [56, 65]\n    else:\n        return [66, 99]","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.294766Z","iopub.status.idle":"2022-04-02T11:32:58.295457Z","shell.execute_reply.started":"2022-04-02T11:32:58.29521Z","shell.execute_reply":"2022-04-02T11:32:58.295237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(clr.S+\"Missing values within customers dataset:\"+clr.E)\nprint(customers.isna().sum())\n\n# 🐝 Log Distinct customer IDs\nwandb.log({\"customer_ids\":len(customers)})\n\n# Fill FN and Active - the only available value is \"1\"\ncustomers[\"FN\"].fillna(0, inplace=True)\ncustomers[\"Active\"].fillna(0, inplace=True)\n\n# Set unknown the club member status & news frequency\ncustomers[\"club_member_status\"].fillna(\"UNKNOWN\", inplace=True)\n\ncustomers[\"fashion_news_frequency\"] = customers[\"fashion_news_frequency\"].replace({\"None\":\"NONE\"})\ncustomers[\"fashion_news_frequency\"].fillna(\"UNKNOWN\", inplace=True)\n\n# Set missing values in age with the median\ncustomers[\"age\"].fillna(customers[\"age\"].median(), inplace=True)\ncustomers[\"age_interval\"] = customers[\"age\"].apply(lambda x: create_age_interval(x))","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.296731Z","iopub.status.idle":"2022-04-02T11:32:58.297415Z","shell.execute_reply.started":"2022-04-02T11:32:58.297175Z","shell.execute_reply":"2022-04-02T11:32:58.297201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(24, 10))\nplt.suptitle('- Customer Profile -', size=22, weight=\"bold\")\n\nax1 = plt.subplot(2,2,1)\nax2 = plt.subplot(2,2,2)\nax3 = plt.subplot(2,1,2)\n\nsns.countplot(data=customers, x=\"club_member_status\", ax=ax1,\n              order=customers['club_member_status'].value_counts().index,\n              palette=my_colors[2:])\nshow_values_on_bars(axs=ax1, h_v=\"v\", space=0.4)\nax1.set_title(\"Club Member Status\", size=18, weight=\"bold\")\nax1.set_yticks([])\nax1.set_xlabel(\"\")\nax1.set_ylabel(\"\")\n\nsns.countplot(data=customers, x=\"fashion_news_frequency\", ax=ax2,\n              order=customers['fashion_news_frequency'].value_counts().index,\n              palette=my_colors[2:])\nshow_values_on_bars(axs=ax2, h_v=\"v\", space=0.4)\nax2.set_title(\"Fashion News frequency\", size=18, weight=\"bold\")\nax2.set_yticks([])\nax2.set_xlabel(\"\")\nax2.set_ylabel(\"\")\n\nsns.distplot(customers[\"age\"], color=my_colors[-3], ax=ax3,\n             hist_kws=dict(edgecolor=my_colors[-3]))\nax3.set_title(\"Age Distribution\", size=18, weight=\"bold\")\nax3.set_ylabel(\"\")\n\nfor ax in [ax1, ax2]:\n    x0,x1 = ax.get_xlim()\n    y0,y1 = ax.get_ylim()\n    ax.imshow(bk_image, zorder=0, extent=[x0, x1, y0, y1], alpha=0.35, aspect='auto')\n    \n# insert_image(path='../input/hm-fashion-recommender-dataset/pics/vans.jpg', zoom=0.5, xybox=(60, 0.00), ax=ax3)\n\nsns.despine(left=True, bottom=True)\nplt.subplots_adjust(left=None, bottom=None, right=None, top=None, wspace=None, hspace=0.99);","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-02T11:32:58.298677Z","iopub.status.idle":"2022-04-02T11:32:58.299368Z","shell.execute_reply.started":"2022-04-02T11:32:58.299105Z","shell.execute_reply":"2022-04-02T11:32:58.299147Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Log Barplot to W&B\ndt = customers[\"club_member_status\"].value_counts().reset_index()\ncreate_wandb_plot(x_data= dt[\"index\"], y_data=dt[\"club_member_status\"],\n                  x_name=\"Status\", y_name=\"Frequency\", \n                  title=\"- Club Member Status -\", log=\"member_status\", plot=\"bar\")\n\ndt = customers[\"fashion_news_frequency\"].value_counts().reset_index()\ncreate_wandb_plot(x_data= dt[\"index\"], y_data=dt[\"fashion_news_frequency\"],\n                  x_name=\"News\", y_name=\"Frequency\", \n                  title=\"- Fashion News Frequency -\", log=\"news_freq\", plot=\"bar\")\n\ncreate_wandb_hist(x_data=customers[\"age\"], x_name=\"Age\", \n                  title=\"Age Distribution\", log=\"age_dist\")\n\nwandb.finish()","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.300622Z","iopub.status.idle":"2022-04-02T11:32:58.301288Z","shell.execute_reply.started":"2022-04-02T11:32:58.301025Z","shell.execute_reply":"2022-04-02T11:32:58.301049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🪄🐝 Save the updated `customers` file.\n# customers.to_parquet('customers.pqt', index=False)\n\nsave_dataset_artifact(run_name=\"save_customers\", artifact_name=\"customers\",\n                      path=\"../input/hm-fashion-recommender-dataset/customers.pqt\")","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.302531Z","iopub.status.idle":"2022-04-02T11:32:58.303215Z","shell.execute_reply.started":"2022-04-02T11:32:58.30295Z","shell.execute_reply":"2022-04-02T11:32:58.302977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Transactions\n\n🛍️ **Important Notes**:\n* Denims, Trousers and Undergarments are sold the most.\n* The **prices are altered**, with the highest one being 0.59 and the lowest being 0.0000169.\n* The most expensive items are leather garments.\n* The average order has around 23 units and costs ~0.649.\n* The units/order is directly correlated with the price/order: as the units increase, the price within the order increases too.","metadata":{}},{"cell_type":"code","source":"# 🐝 W&B Experiment\nrun = wandb.init(project='HandM', name='Transactions', config=CONFIG)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.304464Z","iopub.status.idle":"2022-04-02T11:32:58.305129Z","shell.execute_reply.started":"2022-04-02T11:32:58.304883Z","shell.execute_reply":"2022-04-02T11:32:58.304908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(clr.S+\"Missing values within transactions dataset:\"+clr.E)\nprint(transactions.isna().sum())\n\n# 🐝 Log length of transactions\nwandb.log({\"transaction_ids\":len(transactions)})\n\n# Adjust article_id (as did for articles dataframe)\ntransactions[\"article_id\"] = transactions[\"article_id\"].apply(lambda x: adjust_id(x))","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.306404Z","iopub.status.idle":"2022-04-02T11:32:58.307096Z","shell.execute_reply.started":"2022-04-02T11:32:58.30684Z","shell.execute_reply":"2022-04-02T11:32:58.306868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"code","source":"# Get data\ntop_sold_products = transactions[\"article_id\"].value_counts().reset_index().head(15)\ntop_sold_products.columns = [\"article_id\", \"count\"]\ntop_sold_products = pd.merge(top_sold_products, articles, on=\"article_id\")[[\"article_id\", \"count\", \"prod_name\"]]\n\nclrs = [\"#E90B60\" if x==max(top_sold_products[\"count\"]) else '#AF0848' for x in top_sold_products[\"count\"]]\n\n# Get images\nimage_paths = [path for path in articles[articles[\"article_id\"].isin(top_sold_products[\"article_id\"].tolist())][\"path\"].tolist() \n               if path != None]\nimage_names = articles[articles[\"path\"].isin(image_paths)][\"prod_name\"].tolist()\n\n\n# Plot\nfig, ax = plt.subplots(figsize=(25, 13))\nplt.title('- Products that sell the most (in UNITS) -', size=22, weight=\"bold\")\n\nsns.barplot(data=top_sold_products, x=\"count\", y=\"prod_name\", ax=ax,\n            palette=clrs)\nx0,x1 = ax.get_xlim()\ny0,y1 = ax.get_ylim()\nplt.imshow(bk_image, zorder=0, extent=[x0, x1, y0, y1], alpha=0.35, aspect='auto')\n\nshow_values_on_bars(axs=ax, h_v=\"h\", space=0.4)\nplt.ylabel(\"Product Name\", size = 16, weight=\"bold\")\nplt.xlabel(\"\")\nplt.xticks([])\nplt.yticks(size=16)\nplt.tick_params(size=16)\n\n# insert_image(path='../input/hm-fashion-recommender-dataset/pics/chloe.jpg', zoom=0.45, xybox=(40000, 14), ax=ax)\n\nsns.despine(left=True, bottom=True)\nplt.show();\n\nprint(\"\\n\")\n\n# Plot\nfig, axs = plt.subplots(3, 5, figsize=(23, 10))\nfig.suptitle('- Images -', size=22, weight=\"bold\")\naxs = axs.flatten()\n\nfor k, (path, name) in enumerate(zip(image_paths, image_names)):\n    axs[k].set_title(f\"{name}\", size = 16)\n    img = plt.imread(path)\n    axs[k].imshow(img)\n    axs[k].axis(\"off\")\n\nfor a in [-1, -2]: axs[a].set_visible(False)\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-02T11:32:58.308401Z","iopub.status.idle":"2022-04-02T11:32:58.309043Z","shell.execute_reply.started":"2022-04-02T11:32:58.308804Z","shell.execute_reply":"2022-04-02T11:32:58.308829Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Log Barplot to W&B\ncreate_wandb_plot(x_data= top_sold_products[\"prod_name\"], y_data=top_sold_products[\"count\"],\n                  x_name=\"Prod Name\", y_name=\"Count\", \n                  title=\"- Products that sell the most (UNITS) -\", log=\"sold_most\", plot=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.310308Z","iopub.status.idle":"2022-04-02T11:32:58.310977Z","shell.execute_reply.started":"2022-04-02T11:32:58.310719Z","shell.execute_reply":"2022-04-02T11:32:58.310745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"code","source":"print(clr.S+\"Maximum Price is:\"+clr.E, transactions[\"price\"].max(), \"\\n\" +\n      clr.S+\"Minimum Price is:\"+clr.E, transactions[\"price\"].min(), \"\\n\" +\n      clr.S+\"Average Price is:\"+clr.E, transactions[\"price\"].mean())\n\n# Get data\ntop_sold_products = transactions.groupby(\"article_id\")[\"price\"].max().reset_index()\\\n                                        .sort_values(\"price\", ascending=False).head(15)\ntop_sold_products.columns = [\"article_id\", \"price\"]\ntop_sold_products = pd.merge(top_sold_products, articles, on=\"article_id\")[[\"article_id\", \"price\", \"prod_name\"]]\n\nclrs = [\"#E90B60\" if x==max(top_sold_products[\"price\"]) else '#AF0848' for x in top_sold_products[\"price\"]]\n\n# Get images\nimage_paths = [path for path in articles[articles[\"article_id\"].isin(top_sold_products[\"article_id\"].tolist())][\"path\"].tolist() \n               if path != None]\nimage_names = articles[articles[\"path\"].isin(image_paths)][\"prod_name\"].tolist()\n\n# Plot\nfig, axs = plt.subplots(3, 5, figsize=(23, 10))\nfig.suptitle('- Most Expensive Products -', size=22, weight=\"bold\")\naxs = axs.flatten()\n\nfor k, (path, name) in enumerate(zip(image_paths, image_names)):\n    prc = top_sold_products[top_sold_products[\"prod_name\"]==name][\"price\"].values[0]\n    axs[k].set_title(f\"{name} : {round(prc, 3)}\", size = 16)\n    img = plt.imread(path)\n    axs[k].imshow(img)\n    axs[k].axis(\"off\")\n\n# for a in [-1, -2]: axs[a].set_visible(False)\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-02T11:32:58.312283Z","iopub.status.idle":"2022-04-02T11:32:58.312948Z","shell.execute_reply.started":"2022-04-02T11:32:58.312688Z","shell.execute_reply":"2022-04-02T11:32:58.312713Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"code","source":"# Data\nbasket = transactions.groupby(\"customer_id\").agg({'article_id':'count', \n                                                  'price': 'sum'}).reset_index()\nbasket.columns = [\"customer_id\", \"units\", \"order_price\"]\n\nprint(clr.S+\"=== UNITS/ORDER ===\"+clr.E)\nprint(clr.S+\"Maximum Units/Order is:\"+clr.E, basket[\"units\"].max(), \"\\n\" +\n      clr.S+\"Minimum Units/Order is:\"+clr.E, basket[\"units\"].min(), \"\\n\" +\n      clr.S+\"Average Units/Order is:\"+clr.E, basket[\"units\"].mean(), \"\\n\")\n\nprint(clr.S+\"=== SPENDING/ORDER ===\"+clr.E)\nprint(clr.S+\"Maximum Spending/Order is:\"+clr.E, basket[\"order_price\"].max(), \"\\n\" +\n      clr.S+\"Minimum Spending/Order is:\"+clr.E, basket[\"order_price\"].min(), \"\\n\" +\n      clr.S+\"Average Spending/Order is:\"+clr.E, basket[\"order_price\"].mean())\n\n# Plot\nplt.figure(figsize=(24, 15))\nplt.suptitle('- Order Attributes -', size=22, weight=\"bold\")\n\nax1 = plt.subplot(2,2,1)\nax2 = plt.subplot(2,2,2)\nax3 = plt.subplot(2,1,2)\n\nsns.distplot(basket[\"units\"], color=my_colors[-3], ax=ax1,\n             hist_kws=dict(edgecolor=my_colors[-3]))\nax1.set_title(\"Units/Order Distribution\", size=18, weight=\"bold\")\nax1.set_ylabel(\"\")\n\nsns.distplot(basket[\"order_price\"], color=my_colors[-5], ax=ax2,\n             hist_kws=dict(edgecolor=my_colors[-5]))\nax2.set_title(\"Spending/Order Distribution\", size=18, weight=\"bold\")\nax2.set_ylabel(\"\")\n\nsns.scatterplot(data=basket, x=\"units\", y=\"order_price\", hue=\"units\", palette=\"mako\", \n                legend=None, ax=ax3)\nax3.set_title(\"Units x Price Correlation\", size=18, weight=\"bold\")\nax3.set_ylabel(\"Spending per Order\")\nax3.set_xlabel(\"Units per Order\")\n\nsns.despine(left=True, bottom=True)\nplt.subplots_adjust(left=None, bottom=None, right=None, top=None, wspace=None, hspace=None);","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-02T11:32:58.314264Z","iopub.status.idle":"2022-04-02T11:32:58.314919Z","shell.execute_reply.started":"2022-04-02T11:32:58.314661Z","shell.execute_reply":"2022-04-02T11:32:58.314686Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Log parameters\nwandb.log({\"average_units_per_order\" : basket[\"units\"].mean(),\n           \"average_spending_per_order\" : basket[\"order_price\"].mean()})\n\nwandb.finish()","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.316171Z","iopub.status.idle":"2022-04-02T11:32:58.316857Z","shell.execute_reply.started":"2022-04-02T11:32:58.316584Z","shell.execute_reply":"2022-04-02T11:32:58.31661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Save the updated `transactions` file.\n# transactions.to_parquet('transactions.pqt', index=False)\n\nsave_dataset_artifact(run_name=\"save_transactions\", artifact_name=\"transactions\",\n                      path=\"../input/hm-fashion-recommender-dataset/transactions.pqt\")","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.318118Z","iopub.status.idle":"2022-04-02T11:32:58.318786Z","shell.execute_reply.started":"2022-04-02T11:32:58.318532Z","shell.execute_reply":"2022-04-02T11:32:58.318557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del articles, customers, transactions, ss\ndel top_sold_products, prod_appearance, prod_color, prod_name\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.320018Z","iopub.status.idle":"2022-04-02T11:32:58.320689Z","shell.execute_reply.started":"2022-04-02T11:32:58.320442Z","shell.execute_reply":"2022-04-02T11:32:58.320467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<img src=\"https://i.imgur.com/nMuocgz.png\">\n\n# 5. Market Basket Analysis\n\n**What is a Recommender System**?\n\nA recommender system is not more or less than an algorithm that tries to predict the preference on an object or concept based on somebody's preferences for other objects or concepts.\n\nThis can apply to **anything**: movies, songs, books, amazon orders, clothing or just Google Engine searches.\n\n<center><img src=\"https://i.imgur.com/qKG6v9l.png\" width=500></center>\n\n> 🛍 **Turicreate**: we will be using `turicreate` model in order to create recommendations for users based on their previous purchases. For more details about the library you can [read the documentation](https://github.com/apple/turicreate). My main inspiration was this amazing article [How to Build a Recommendation System for Purchase Data (Step-by-Step)](https://medium.datadriveninvestor.com/how-to-build-a-recommendation-system-for-purchase-data-step-by-step-d6d7a78800b6).\n\n> 🛍 **RAPIDS**: as RAPIDS outperformes pandas whenever we have large datasets, I will be using it to prepare the data.","metadata":{}},{"cell_type":"code","source":"!pip install turicreate --user\n\nimport turicreate as tc\nimport cudf\nimport cuml\nimport cupy\n\nfrom cuml.model_selection import train_test_split","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-04-02T11:32:58.321985Z","iopub.status.idle":"2022-04-02T11:32:58.322663Z","shell.execute_reply.started":"2022-04-02T11:32:58.322415Z","shell.execute_reply":"2022-04-02T11:32:58.322441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### More Helper Functions","metadata":{}},{"cell_type":"code","source":"def create_predictions_format(data):\n    '''\n    data: a pandas dataframe that contains customer_id and article_id columns\n            that we wish to format as in submission file\n    return: data[[\"customer_id\", \"preds\"]]\n    '''\n    \n    # Adjust ID\n    data[\"article_id\"] = data[\"article_id\"].apply(lambda x: adjust_id(x))\n\n    # Group article_ids\n    all_preds = data.groupby(\"customer_id\")[\"article_id\"].unique().to_dict()\n    data[\"preds\"] = data[\"customer_id\"].map(all_preds)\n    data[\"preds\"] = data[\"preds\"].apply(lambda x: \" \".join([str(y) for y in x]))\n\n    # Unicize\n    data = data.groupby(\"customer_id\")[\"preds\"].first().reset_index()\n    \n    return data\n\n\ndef get_frequent_purchases(transactions, n=50):\n    '''\n    This function looks at customer level and retrieves most frequent (> 50%) purchased items\n    transactions: original cudf .csv or .pqt file\n    return : temp[[\"customer_id\", \"preds\"]]\n    '''\n    \n    # Compute count per each customer and article\n    temp = transactions.groupby([\"customer_id\", \"article_id\"])[\"t_dat\"].count().reset_index()\n    temp.columns = [\"customer_id\", \"article_id\", \"count\"]\n\n    # Compute total count per each customer\n    temp2 = transactions.groupby([\"customer_id\"])[\"t_dat\"].count().reset_index()\n    temp2.columns = [\"customer_id\", \"full_count\"]\n\n    temp = temp.merge(temp2, on=\"customer_id\", how=\"left\")\n    temp[\"perc\"] = (temp[\"count\"] / temp[\"full_count\"])*100\n\n    # Select only articles that represented at least 50% of the entire purchase\n    temp = temp[temp[\"perc\"] >= n].reset_index(drop=True).to_pandas()\n\n    temp = create_predictions_format(temp)\n\n    return  cudf.DataFrame(temp)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-02T11:32:58.32412Z","iopub.status.idle":"2022-04-02T11:32:58.324865Z","shell.execute_reply.started":"2022-04-02T11:32:58.324582Z","shell.execute_reply":"2022-04-02T11:32:58.324609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read in RAPIDS dataframes\ntransactions = cudf.read_parquet(\"../input/hm-fashion-recommender-dataset/transactions.pqt\")","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.326239Z","iopub.status.idle":"2022-04-02T11:32:58.326966Z","shell.execute_reply.started":"2022-04-02T11:32:58.326683Z","shell.execute_reply":"2022-04-02T11:32:58.326711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Faster & Lighter dataframe\n\nI will be using Chris's Deotte [trick within this discussion post](https://www.kaggle.com/competitions/h-and-m-personalized-fashion-recommendations/discussion/308635) to make the dataframes lighter.","metadata":{}},{"cell_type":"code","source":"# Keep only last 16 digits from customer_id and convert to int\ntransactions['customer_id'] = transactions['customer_id'].str[-16:].str.hex_to_int().astype('int64')\n# Convert article_id from object to int32\n# transactions['article_id'] = transactions['article_id'].astype('int32')\n# Convert date from object to datetime\ntransactions['t_dat'] = cudf.to_datetime(transactions['t_dat'])\n\ntransactions = transactions[['t_dat','customer_id','article_id']]","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.328335Z","iopub.status.idle":"2022-04-02T11:32:58.329071Z","shell.execute_reply.started":"2022-04-02T11:32:58.328802Z","shell.execute_reply":"2022-04-02T11:32:58.328829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Decrease dataset size\n\n🛍 I will decrease the dataset size as following:\n* Select only **TOP_N most frequently bought articles** - as there are 104,547 unique article IDs, the training dataframe would be extremely wide and the notebook would run out of memory. Hence, for showcasing purposes I have set the TOP_N used in this notebook to be very low (only 200 ids are used).\n* Get only customers with many transactions (as customers with only 1 purchased item are harder to evaluate by the model).","metadata":{}},{"cell_type":"code","source":"# ------ PARAMETERS ------\nTOP_CUSTOMERS = 300000\nTOP_N = 200\n# ------------------------","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.330434Z","iopub.status.idle":"2022-04-02T11:32:58.331181Z","shell.execute_reply.started":"2022-04-02T11:32:58.330892Z","shell.execute_reply":"2022-04-02T11:32:58.330921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select only most frequent article ids\nmost_frequent_articles = transactions[\"article_id\"].value_counts().reset_index()\nmost_frequent_articles.columns = [\"article_id\", \"count\"]\nprint(clr.S+\"Total Unique IDs in Transactions:\"+clr.E, len(most_frequent_articles))\nprint(clr.S+\"Total Unique IDs that are selected:\"+clr.E, TOP_N)\n# Get top n most frequent products\nmost_frequent_articles = cupy.asarray(most_frequent_articles.head(TOP_N)[\"article_id\"])\n\ntransactions = transactions[transactions[\"article_id\"].isin(most_frequent_articles)].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.332673Z","iopub.status.idle":"2022-04-02T11:32:58.333432Z","shell.execute_reply.started":"2022-04-02T11:32:58.333162Z","shell.execute_reply":"2022-04-02T11:32:58.333191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get customers with many transactions on these TOP_N items\ncustomers_top_trans = list(transactions[\"customer_id\"].value_counts().reset_index()\\\n                            .head(TOP_CUSTOMERS)[\"index\"].unique().to_pandas())\ntransactions = transactions[transactions[\"customer_id\"].isin(customers_top_trans)].reset_index(drop=True)\n\nprint(clr.S+\"Total unique users to recommend:\"+clr.E, transactions[\"customer_id\"].nunique())","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.334821Z","iopub.status.idle":"2022-04-02T11:32:58.335562Z","shell.execute_reply.started":"2022-04-02T11:32:58.33528Z","shell.execute_reply":"2022-04-02T11:32:58.33531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del most_frequent_articles, customers_top_trans\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.336955Z","iopub.status.idle":"2022-04-02T11:32:58.337689Z","shell.execute_reply.started":"2022-04-02T11:32:58.337421Z","shell.execute_reply":"2022-04-02T11:32:58.337449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6. Prepare Datasets\n\nWe will be preparing 3 distinct datasets as follows:\n\n<center><img src=\"https://i.imgur.com/ODESQDe.png\" width=700></center>\n\n### I. Train Dataset","metadata":{}},{"cell_type":"code","source":"# Count per each customer how many products of each they have bought\ntrain = transactions.groupby([\"customer_id\",\"article_id\"])[\"t_dat\"].count().reset_index()\ntrain.columns = [\"customer_id\",\"article_id\", \"purchase_count\"]\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.33907Z","iopub.status.idle":"2022-04-02T11:32:58.339806Z","shell.execute_reply.started":"2022-04-02T11:32:58.339524Z","shell.execute_reply":"2022-04-02T11:32:58.339551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### II. Dummy Dataset","metadata":{}},{"cell_type":"code","source":"dummy_train = train.copy()\ndummy_train['purchase_dummy'] = 1\n\ndummy_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.341179Z","iopub.status.idle":"2022-04-02T11:32:58.341903Z","shell.execute_reply.started":"2022-04-02T11:32:58.341625Z","shell.execute_reply":"2022-04-02T11:32:58.341652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### III. Normalized Dataset","metadata":{}},{"cell_type":"code","source":"def normalize_data(data):\n    \n    # Create matrix with customers on rows and articles ad columns\n    # Using pandas here due to an error in cudf\n    df_matrix = cudf.DataFrame(pd.pivot(data.to_pandas(), columns=\"article_id\",\n                                        index=\"customer_id\", values=\"purchase_count\"))\n    # Normalize\n    df_matrix_norm = (df_matrix-df_matrix.min())/(df_matrix.max()-df_matrix.min())\n    d = df_matrix_norm.reset_index()\n    d.index.names = ['scaled_purchase_freq']\n    \n    # Recreate df with each customer and the 'label' whether they purchased the product or not\n    final = cudf.melt(d, id_vars=['customer_id'], value_name='scaled_purchase_freq').dropna()\n    final.columns = [\"customer_id\", \"article_id\", \"scaled_purchase_freq\"]\n    final = final.reset_index(drop=True)\n    \n    return final","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.34329Z","iopub.status.idle":"2022-04-02T11:32:58.344002Z","shell.execute_reply.started":"2022-04-02T11:32:58.343726Z","shell.execute_reply":"2022-04-02T11:32:58.343753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Normalize count dataset\nnorm_train = normalize_data(data=train)\n\nnorm_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.345362Z","iopub.status.idle":"2022-04-02T11:32:58.3461Z","shell.execute_reply.started":"2022-04-02T11:32:58.345831Z","shell.execute_reply":"2022-04-02T11:32:58.34586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 7. Data Validation","metadata":{}},{"cell_type":"code","source":"def split_data(data):\n    \n    train, test = train_test_split(data, test_size=0.3)\n    train_data = tc.SFrame(train.to_pandas())\n    test_data = tc.SFrame(test.to_pandas())\n    \n    return train_data, test_data","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.347492Z","iopub.status.idle":"2022-04-02T11:32:58.348313Z","shell.execute_reply.started":"2022-04-02T11:32:58.348004Z","shell.execute_reply":"2022-04-02T11:32:58.348032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TODO: fix norm_train error\ntrain_data, test_data = split_data(train)\ntrain_data_dummy, test_data_dummy = split_data(dummy_train)\n# train_data_norm, test_data_norm = split_data(norm_train)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.349703Z","iopub.status.idle":"2022-04-02T11:32:58.350456Z","shell.execute_reply.started":"2022-04-02T11:32:58.350188Z","shell.execute_reply":"2022-04-02T11:32:58.350217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 8. Models","metadata":{}},{"cell_type":"code","source":"# ------ PARAMETERS ------\nuser_id = 'customer_id'\nitem_id = 'article_id'\nusers_to_recommend = list(train[\"customer_id\"].unique().to_pandas())\n# ------------------------","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.351825Z","iopub.status.idle":"2022-04-02T11:32:58.352549Z","shell.execute_reply.started":"2022-04-02T11:32:58.352282Z","shell.execute_reply":"2022-04-02T11:32:58.35231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model(train_data, name, user_id, item_id, target, users_to_recommend):\n    '''\n    Trains a recommender model.\n    train_data: the training tc.SFrame()\n    name: can be 'popularity', 'cosine' or 'pearson'\n    user_id & item_id: the customer and article unique IDs\n    target: the value to be predicted, can be 'purchase_count', 'purchase_dummy' or 'scaled_purchase_freq'\n    users_to_recommend: a unique list containing all customers for which we do the prediction\n    '''\n    \n    if name == 'popularity':\n        model = tc.popularity_recommender.create(train_data, \n                                                 user_id=user_id, \n                                                 item_id=item_id, \n                                                 target=target, verbose=False)\n    elif name == 'cosine':\n        model = tc.item_similarity_recommender.create(train_data, \n                                                      user_id=user_id, \n                                                      item_id=item_id, \n                                                      target=target,\n                                                      similarity_type='cosine', verbose=False)\n    elif name == 'pearson':\n            model = tc.item_similarity_recommender.create(train_data, \n                                                          user_id=user_id, \n                                                          item_id=item_id, \n                                                          target=target, \n                                                          similarity_type='pearson', verbose=False)\n    \n    # k is set to 12 => maximum items to recommend for one customer\n    recom = model.recommend(users=users_to_recommend, k=12, verbose=False)\n    \n    return model, recom","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.353934Z","iopub.status.idle":"2022-04-02T11:32:58.354677Z","shell.execute_reply.started":"2022-04-02T11:32:58.354406Z","shell.execute_reply":"2022-04-02T11:32:58.354435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8.1 Popularity Recommender System\n\n> 🛍 **Popularity Algorithm**: this one recommends the most popular articles across all customers.","metadata":{}},{"cell_type":"code","source":"# --- TRAIN ---\nname = 'popularity'\ntarget = 'purchase_count'\n\npopularity_count, _ = train_model(train_data, name, user_id, item_id, target, users_to_recommend)\n# _.to_dataframe()[\"article_id\"].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.356078Z","iopub.status.idle":"2022-04-02T11:32:58.356828Z","shell.execute_reply.started":"2022-04-02T11:32:58.356534Z","shell.execute_reply":"2022-04-02T11:32:58.356562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# --- DUMMY ---\nname = 'popularity'\ntarget = 'purchase_dummy'\n\npopularity_dummy, _ = train_model(train_data_dummy, name, user_id, item_id, target, users_to_recommend)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.358288Z","iopub.status.idle":"2022-04-02T11:32:58.359056Z","shell.execute_reply.started":"2022-04-02T11:32:58.358763Z","shell.execute_reply":"2022-04-02T11:32:58.358803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8.2 Cosine Recommender System\n\n> 🛍 **Cosine Algorithm**: uses the *collaborative filtering* methodology to find how similar is a product from another.","metadata":{}},{"cell_type":"code","source":"# --- TRAIN ---\nname = 'cosine'\ntarget = 'purchase_count'\n\ncosine_count, _ = train_model(train_data, name, user_id, item_id, target, users_to_recommend)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.360455Z","iopub.status.idle":"2022-04-02T11:32:58.361215Z","shell.execute_reply.started":"2022-04-02T11:32:58.360929Z","shell.execute_reply":"2022-04-02T11:32:58.360957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# --- DUMMY ---\nname = 'cosine'\ntarget = 'purchase_dummy'\n\ncosine_dummy, _ = train_model(train_data_dummy, name, user_id, item_id, target, users_to_recommend)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.362569Z","iopub.status.idle":"2022-04-02T11:32:58.363308Z","shell.execute_reply.started":"2022-04-02T11:32:58.363019Z","shell.execute_reply":"2022-04-02T11:32:58.363047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8.3 Pearson Recommender System\n\n> 🛍 **Pearson Algorithm**: as explained in the Cosine Algorithm, this also uses the *collaborative filtering* methodology to find how similar is a product from another.","metadata":{}},{"cell_type":"code","source":"# --- TRAIN ---\nname = 'pearson'\ntarget = 'purchase_count'\n\npearson_count, _ = train_model(train_data, name, user_id, item_id, target, users_to_recommend)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.364665Z","iopub.status.idle":"2022-04-02T11:32:58.365415Z","shell.execute_reply.started":"2022-04-02T11:32:58.365129Z","shell.execute_reply":"2022-04-02T11:32:58.365175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# --- DUMMY ---\nname = 'pearson'\ntarget = 'purchase_dummy'\n\npearson_dummy, _ = train_model(train_data_dummy, name, user_id, item_id, target, users_to_recommend)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.366799Z","iopub.status.idle":"2022-04-02T11:32:58.367513Z","shell.execute_reply.started":"2022-04-02T11:32:58.367244Z","shell.execute_reply":"2022-04-02T11:32:58.367272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8.4 Evaluate\n\n> 🛍 **Note**: The tables below show the **RMSE, Mean Precision & Mean Recall** output from the 3 algorithms (Popularity, Cosine and Pearson) on 2 types of datasets: Count and Dummy (for mor details you can check out the `eval_counts.txt` and `eval_dummy.txt` files available in [my dataset](https://www.kaggle.com/datasets/andradaolteanu/hm-fashion-recommender-dataset))\n\n<center><img src=\"https://i.imgur.com/8LP7Cr4.png\" width=900></center>\n\n*Side note: I have commented the cells below so that the notebook commits faster - all the results are in `eval_counts.txt` and `eval_dummy.txt` files available in [this dataset](https://www.kaggle.com/datasets/andradaolteanu/hm-fashion-recommender-dataset)*","metadata":{}},{"cell_type":"code","source":"# Group all models per type of dataset\nmodel_counts = [popularity_count, cosine_count, pearson_count]\nmodels_dummy = [popularity_dummy, cosine_dummy, pearson_dummy]\n\nnames_counts = ['Popularity Model on Purchase Counts', 'Cosine Similarity on Purchase Counts',\n                'Pearson Similarity on Purchase Counts']\nnames_dummy = ['Popularity Model on Purchase Dummy', 'Cosine Similarity on Purchase Dummy',\n               'Pearson Similarity on Purchase Dummy']\n\n# # Evaluate the models\n# eval_counts = tc.recommender.util.compare_models(test_data, model_counts, model_names=names_counts, verbose=False)\n# eval_dummy = tc.recommender.util.compare_models(test_data_dummy, models_dummy, model_names=names_dummy, verbose=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.368906Z","iopub.status.idle":"2022-04-02T11:32:58.369638Z","shell.execute_reply.started":"2022-04-02T11:32:58.36937Z","shell.execute_reply":"2022-04-02T11:32:58.369399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Print outputs to .txt files\n# f = open(\"eval_counts.txt\", \"a\")\n# print(eval_counts, file=f)\n# f.close()\n\n# f = open(\"eval_dummy.txt\", \"a\")\n# print(eval_dummy, file=f)\n# f.close()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-02T11:32:58.371024Z","iopub.status.idle":"2022-04-02T11:32:58.371761Z","shell.execute_reply.started":"2022-04-02T11:32:58.371491Z","shell.execute_reply":"2022-04-02T11:32:58.371519Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Save the updated `articles` file.\nsave_dataset_artifact(run_name=\"save_output_counts\", artifact_name=\"counts_results\",\n                      path=\"../input/hm-fashion-recommender-dataset/eval_counts.txt\")\n\nsave_dataset_artifact(run_name=\"save_output_dummy\", artifact_name=\"dummy_results\",\n                      path=\"../input/hm-fashion-recommender-dataset/eval_dummy.txt\")","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.373184Z","iopub.status.idle":"2022-04-02T11:32:58.373913Z","shell.execute_reply.started":"2022-04-02T11:32:58.373633Z","shell.execute_reply":"2022-04-02T11:32:58.373661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del popularity_count, cosine_count, pearson_count\ndel popularity_dummy, cosine_dummy, pearson_dummy\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.37531Z","iopub.status.idle":"2022-04-02T11:32:58.376055Z","shell.execute_reply.started":"2022-04-02T11:32:58.375785Z","shell.execute_reply":"2022-04-02T11:32:58.375813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 9. Prediction\n\n## 9.1 Predictions using the Recommender System","metadata":{}},{"cell_type":"code","source":"METHOD = \"cosine\"\nTARGET = \"purchase_count\"\n\n# 🐝 W&B Experiment\nrun = wandb.init(project='HandM', name=f'{METHOD}_customers{TOP_CUSTOMERS}_topArticles{TOP_N}', config=CONFIG)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.37747Z","iopub.status.idle":"2022-04-02T11:32:58.37822Z","shell.execute_reply.started":"2022-04-02T11:32:58.377933Z","shell.execute_reply":"2022-04-02T11:32:58.377961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Final Model Training\nfinal_model = tc.item_similarity_recommender.create(tc.SFrame(train.to_pandas()), \n                                                    user_id=user_id, \n                                                    item_id=item_id, \n                                                    target=TARGET,\n                                                    similarity_type=METHOD,\n                                                    verbose=False)\n\nrecom = final_model.recommend(users=users_to_recommend, k=12, verbose=False)\n\n# Convert to dataframe\nrecom_df = recom.to_dataframe()\n\nprint(clr.S+\"12 Recommendations for each Customer:\"+clr.E)\nrecom_df.head(12)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.379569Z","iopub.status.idle":"2022-04-02T11:32:58.380305Z","shell.execute_reply.started":"2022-04-02T11:32:58.380016Z","shell.execute_reply":"2022-04-02T11:32:58.380044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create predictions out of the dataset\nrecom_df = create_predictions_format(recom_df)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.381636Z","iopub.status.idle":"2022-04-02T11:32:58.382383Z","shell.execute_reply.started":"2022-04-02T11:32:58.382092Z","shell.execute_reply":"2022-04-02T11:32:58.38212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Log CV score\nwandb.log({\"TOP_CUSTOMERS\" : TOP_CUSTOMERS,\n              \"TOP_N\" : TOP_N,\n              \"METHOD\" : METHOD,\n              \"TARGET\": TARGET,\n              \"CV\" : 0.0051})\n\nwandb.finish()","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.383787Z","iopub.status.idle":"2022-04-02T11:32:58.384509Z","shell.execute_reply.started":"2022-04-02T11:32:58.384243Z","shell.execute_reply":"2022-04-02T11:32:58.38427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Make Submission","metadata":{}},{"cell_type":"code","source":"# Import sample submission\nss = cudf.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\")\nss['customer_id_new'] = ss['customer_id'].str[-16:].str.hex_to_int().astype('int64')\n\n\n# Merge with predicted preds\nss = ss.merge(cudf.DataFrame(recom_df[[\"customer_id\", \"preds\"]]), \n              left_on=\"customer_id_new\", right_on=\"customer_id\", how=\"left\")","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.38589Z","iopub.status.idle":"2022-04-02T11:32:58.386654Z","shell.execute_reply.started":"2022-04-02T11:32:58.386375Z","shell.execute_reply":"2022-04-02T11:32:58.386404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 9.2 Prediction using Common Sense\n\nBecause we **cannot predict every customer using the Recommender Systems** (because the dataset is too large), we will have to use other *simpler* methods to predict the rest of the customers.\n\nI was heavily inspired by Chris' Deotte's [notebook](https://www.kaggle.com/code/cdeotte/recommend-items-purchased-together-0-021/notebook) fot his next section (which is as of now **work in progress**).","metadata":{}},{"cell_type":"code","source":"# Read in RAPIDS dataframes\ntransactions = cudf.read_parquet(\"../input/hm-fashion-recommender-dataset/transactions.pqt\")\nprint(clr.S+\"Unique customer IDs:\"+clr.E, transactions[\"customer_id\"].nunique())\n\ntransactions['customer_id'] = transactions['customer_id'].str[-16:].str.hex_to_int().astype('int64')\ntransactions['t_dat'] = cudf.to_datetime(transactions['t_dat'])\ntransactions = transactions[['t_dat','customer_id','article_id']]\n\n\n# === EXCLUDE users already predicted using cosine ===\ntransactions = transactions[~transactions[\"customer_id\"].isin(users_to_recommend)]","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.387912Z","iopub.status.idle":"2022-04-02T11:32:58.388589Z","shell.execute_reply.started":"2022-04-02T11:32:58.388345Z","shell.execute_reply":"2022-04-02T11:32:58.388371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### I. Compute most frequent purchased items per customer","metadata":{}},{"cell_type":"code","source":"temp = get_frequent_purchases(transactions, n=50)\ntemp.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.389824Z","iopub.status.idle":"2022-04-02T11:32:58.390508Z","shell.execute_reply.started":"2022-04-02T11:32:58.39026Z","shell.execute_reply":"2022-04-02T11:32:58.390286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### II. Compute Last Week's Most Popular Items","metadata":{}},{"cell_type":"code","source":"def get_top_12(transactions):\n    '''\n    transactions: cudf original dataframe\n    return: string containing top 12 products\n    '''\n    \n    temp = transactions.loc[transactions[\"t_dat\"] >= cudf.to_datetime('2020-09-16')]\n    top12 = ' 0' + ' 0'.join(temp.article_id.value_counts().to_pandas().index.astype('str')[:12])\n    \n    return top12\n\ntop12 = get_top_12(transactions)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.391878Z","iopub.status.idle":"2022-04-02T11:32:58.392532Z","shell.execute_reply.started":"2022-04-02T11:32:58.392289Z","shell.execute_reply":"2022-04-02T11:32:58.392315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Final Submission","metadata":{}},{"cell_type":"code","source":"# Merge with common sense predictions\nss = ss.merge(temp, left_on=\"customer_id_new\", right_on=\"customer_id\", how=\"left\")\\\n        .drop(columns=[\"prediction\", \"customer_id_new\", \"customer_id_y\", \"customer_id\"])\n\n# Fill in final predictions (from models) with common sense preds\nss[\"preds_x\"].fillna((ss[\"preds_y\"]+top12)[:131], inplace=True)\n\n# Complete missing values with top 12\nss[\"preds_x\"].fillna(top12, inplace=True)\n\n# Drop unwanted columns and make submission\nss.drop(columns=[\"preds_y\"], inplace=True)\nss.columns = [\"customer_id\", \"prediction\"]\n\nss.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.393791Z","iopub.status.idle":"2022-04-02T11:32:58.394458Z","shell.execute_reply.started":"2022-04-02T11:32:58.394217Z","shell.execute_reply":"2022-04-02T11:32:58.394243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Submission needs to be 1371980 long\n\nprint(clr.S+\"Submission Shape:\"+clr.E, ss.shape)\nss.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.3957Z","iopub.status.idle":"2022-04-02T11:32:58.396377Z","shell.execute_reply.started":"2022-04-02T11:32:58.396121Z","shell.execute_reply":"2022-04-02T11:32:58.396162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# === Work in Progress","metadata":{"execution":{"iopub.status.busy":"2022-04-02T11:32:58.397621Z","iopub.status.idle":"2022-04-02T11:32:58.398309Z","shell.execute_reply.started":"2022-04-02T11:32:58.398044Z","shell.execute_reply":"2022-04-02T11:32:58.39807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<center><img src=\"https://i.imgur.com/0cx4xXI.png\"></center>\n\n### 🐝 W&B Dashboard\n\n> My [W&B Dashboard](https://wandb.ai/andrada/HandM).\n\n<center><video src=\"https://i.imgur.com/ni0pfbj.mp4\" width=800 controls></center>\n\n<center><img src=\"https://i.imgur.com/knxTRkO.png\"></center>\n\n### My Specs\n\n* 🖥 Z8 G4 Workstation\n* 💾 2 CPUs & 96GB Memory\n* 🎮 NVIDIA Quadro RTX 8000\n* 💻 Zbook Studio G7 on the go","metadata":{}}]}