{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Imports\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom datetime import datetime\nfrom sklearn.preprocessing import StandardScaler\nimport os","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:06:09.895388Z","iopub.execute_input":"2022-03-28T21:06:09.896727Z","iopub.status.idle":"2022-03-28T21:06:11.181292Z","shell.execute_reply.started":"2022-03-28T21:06:09.896573Z","shell.execute_reply":"2022-03-28T21:06:11.180297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:06:11.183184Z","iopub.execute_input":"2022-03-28T21:06:11.183828Z","iopub.status.idle":"2022-03-28T21:06:12.622566Z","shell.execute_reply.started":"2022-03-28T21:06:11.183774Z","shell.execute_reply":"2022-03-28T21:06:12.621664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_df = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/customers.csv')","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:06:12.623996Z","iopub.execute_input":"2022-03-28T21:06:12.624354Z","iopub.status.idle":"2022-03-28T21:06:18.627046Z","shell.execute_reply.started":"2022-03-28T21:06:12.624277Z","shell.execute_reply":"2022-03-28T21:06:18.625978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_df = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv', parse_dates=['t_dat'])","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:06:18.628895Z","iopub.execute_input":"2022-03-28T21:06:18.629162Z","iopub.status.idle":"2022-03-28T21:07:33.214849Z","shell.execute_reply.started":"2022-03-28T21:06:18.629127Z","shell.execute_reply":"2022-03-28T21:07:33.214025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Analyzing the articles dataset","metadata":{}},{"cell_type":"code","source":"articles_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-28T19:25:03.404949Z","iopub.execute_input":"2022-03-28T19:25:03.405686Z","iopub.status.idle":"2022-03-28T19:25:03.441612Z","shell.execute_reply.started":"2022-03-28T19:25:03.405638Z","shell.execute_reply":"2022-03-28T19:25:03.440784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'The articles dataset has {articles_df.shape[0]} records, each with {articles_df.shape[1]} features')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles_df.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking for missing values\narticles_df.isnull().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking number of unique values per column\narticles_df.nunique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There seem to be categories where the number of unique names is lower than the number of unique ids.\nFor instance, there are 45875 unique values in the 'prod_name' column, while there are 47224 unique values in the 'product_code' column.\nThis is most likely because distinct articles were actually named the same. I will therefore use only the columns corresponding to ids, not the ones with names.","metadata":{}},{"cell_type":"code","source":"# Checking the types of articles\nf, ax = plt.subplots(figsize=(15, 7))\nax = sns.histplot(data=articles_df, y='index_name')\nax.set_xlabel('count')\nax.set_ylabel('index name')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Analyzing the customers dataset","metadata":{}},{"cell_type":"code","source":"customers_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'The customers dataset has {customers_df.shape[0]} records, each with {customers_df.shape[1]} features')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking for missing values\ncustomers_df.isnull().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are quite a few missing values in this dataset. Luckily, the columns for age and postal_code have few and no missing values, respectively. These two should be useful for recommeding articles to a customer. ","metadata":{}},{"cell_type":"code","source":"# Checking number of unique values per column\ncustomers_df.nunique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting the age distribution\nf, ax = plt.subplots(figsize=(15, 7))\nax = sns.histplot(data=customers_df, x='age', bins=40)\nax.set_xlabel('age')\nax.set_ylabel('count')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are a lot of customers in the early 20's and there is another spike at the ages of 45-55. It could indicate a strategy of predicting differently for customers of those age groups.","metadata":{}},{"cell_type":"code","source":"median_age = customers_df['age'].median(skipna=True)\nprint(f\"The customers' median age is {median_age}\")","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:07:33.216275Z","iopub.execute_input":"2022-03-28T21:07:33.216578Z","iopub.status.idle":"2022-03-28T21:07:33.261145Z","shell.execute_reply.started":"2022-03-28T21:07:33.216535Z","shell.execute_reply":"2022-03-28T21:07:33.26016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 1, figsize=(15, 7))\nclub_member_statuses = customers_df.groupby('club_member_status', as_index=False)['customer_id'].count()\nax = sns.barplot(data=club_member_statuses, x='club_member_status', y='customer_id')\nplt.xlabel(\"club member status\")\nplt.ylabel(\"number of customers\")\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Given that almost all customers have an 'ACTIVE' membership, the column probably would not help the model.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 1, figsize=(15, 7))\nclub_member_statuses = customers_df.groupby('fashion_news_frequency', as_index=False)['customer_id'].count()\nax = sns.barplot(data=club_member_statuses, x='fashion_news_frequency', y='customer_id')\nplt.xlabel(\"fashion news frequency\")\nplt.ylabel(\"number of customers\")\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Analyzing the transactions dataset","metadata":{}},{"cell_type":"code","source":"transactions_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking for missing values\ntransactions_df.isnull().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are no missing values in this dataset.","metadata":{}},{"cell_type":"code","source":"# Plotting transactions per day\ntransactions_per_day = transactions_df.groupby('t_dat', as_index=False)['article_id'].count()\n\nfig, ax = plt.subplots(1, 1, figsize=(15, 7))\nplt.plot(transactions_per_day['t_dat'], transactions_per_day['article_id'])\nplt.xlabel(\"date\")\nplt.ylabel(\"number of transactions\")\nax.set_xlim(transactions_per_day['t_dat'].min(), transactions_per_day['t_dat'].max())\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_df['t_dat_month'] = transactions_df['t_dat'].apply(lambda date: date.month)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting transactions grouped by month\nfig, ax = plt.subplots(1, 1, figsize=(15, 7))\ntransactions_by_month = transactions_df.groupby('t_dat_month', as_index=False)['article_id'].count()\nax = sns.barplot(data=transactions_by_month, x='t_dat_month', y='article_id')\nplt.xlabel(\"month\")\nplt.ylabel(\"number of transactions\")\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most transactions occur during summer and the least occur during winter.","metadata":{}},{"cell_type":"code","source":"transactions_df['t_dat_weekday'] = transactions_df['t_dat'].apply(lambda date: date.weekday())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting transactions grouped by the day of the week\nimport calendar\n\nfig, ax = plt.subplots(1, 1, figsize=(15, 7))\ntransactions_by_weekday = transactions_df.groupby('t_dat_weekday', as_index=False)['article_id'].count()\ntransactions_by_weekday['t_dat_weekday'] = transactions_by_weekday['t_dat_weekday'].apply(lambda x: calendar.day_name[x])\nax = sns.barplot(data=transactions_by_weekday, x='t_dat_weekday', y='article_id')\nplt.xlabel(\"day of week\")\nplt.ylabel(\"number of transactions\")\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Creating the Dataset for the first model","metadata":{}},{"cell_type":"code","source":"# Using only recent transactions for the recommender model\nSTARTING_DATE = '2020-09-15'\nrecent_transactions_df = transactions_df[transactions_df['t_dat'] > STARTING_DATE]","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:07:33.262933Z","iopub.execute_input":"2022-03-28T21:07:33.26317Z","iopub.status.idle":"2022-03-28T21:07:33.462653Z","shell.execute_reply.started":"2022-03-28T21:07:33.263142Z","shell.execute_reply":"2022-03-28T21:07:33.461594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recent_transactions_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:07:33.463725Z","iopub.execute_input":"2022-03-28T21:07:33.463954Z","iopub.status.idle":"2022-03-28T21:07:33.483091Z","shell.execute_reply.started":"2022-03-28T21:07:33.463927Z","shell.execute_reply":"2022-03-28T21:07:33.482151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Merging the transactions with the articles dataframe and keeping only a part of the columns (the ones that are numerical, since there were less unique values\n# for names in the articles dataset)\n# Product group name is kept because there was no numeric equivalent\ntransactions_articles_merged = recent_transactions_df.merge(articles_df, on='article_id')\nkept_columns = ['customer_id', 'article_id', 'product_group_name', 'graphical_appearance_no', 'colour_group_code',\n               'perceived_colour_value_id', 'perceived_colour_master_id', 'department_no', 'index_code', 'index_group_no',\n               'section_no', 'garment_group_no']\ntransactions_articles_merged = transactions_articles_merged[kept_columns]","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:07:33.486529Z","iopub.execute_input":"2022-03-28T21:07:33.48678Z","iopub.status.idle":"2022-03-28T21:07:34.313556Z","shell.execute_reply.started":"2022-03-28T21:07:33.486749Z","shell.execute_reply":"2022-03-28T21:07:34.312602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_articles_merged.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:07:34.314891Z","iopub.execute_input":"2022-03-28T21:07:34.315696Z","iopub.status.idle":"2022-03-28T21:07:34.330022Z","shell.execute_reply.started":"2022-03-28T21:07:34.315657Z","shell.execute_reply":"2022-03-28T21:07:34.329199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encoding the fashion_news_frequency column. Firstly, consider that the missing values correspond to a 'NONE' frequency.\n# Also, two entries have the value \"None\" instead of \"NONE\".\n# We can use a label encoding for this column because the values could be ordered: NONE < Monthly < Regularly\ncustomers_df['fashion_news_frequency'] = customers_df['fashion_news_frequency'].fillna('NONE')\ncustomers_df.loc[customers_df['fashion_news_frequency'] == 'None', 'fashion_news_frequency'] = 'NONE'\n\ndef frequency_type_to_code(type):\n    if type == 'NONE':\n        return 0\n    elif type == 'Monthly':\n        return 1\n    else:\n        return 2\n\ncustomers_df['fashion_news_frequency'] = customers_df['fashion_news_frequency'].apply(lambda x: frequency_type_to_code(x))","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:07:34.331248Z","iopub.execute_input":"2022-03-28T21:07:34.331827Z","iopub.status.idle":"2022-03-28T21:07:35.836816Z","shell.execute_reply.started":"2022-03-28T21:07:34.331789Z","shell.execute_reply":"2022-03-28T21:07:35.835944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dropping some columns that don't offer enough value to the model (postal code is too varied and the other columns have few unique values but one of them\n# dominates the others in frequence, it would be hard for the model to extract insights from those columns)\ncustomers_df = customers_df.drop(['FN', 'Active', 'club_member_status', 'postal_code'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:07:35.840163Z","iopub.execute_input":"2022-03-28T21:07:35.840522Z","iopub.status.idle":"2022-03-28T21:07:35.953244Z","shell.execute_reply.started":"2022-03-28T21:07:35.840477Z","shell.execute_reply":"2022-03-28T21:07:35.952368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:07:35.954335Z","iopub.execute_input":"2022-03-28T21:07:35.954591Z","iopub.status.idle":"2022-03-28T21:07:35.966058Z","shell.execute_reply.started":"2022-03-28T21:07:35.954562Z","shell.execute_reply":"2022-03-28T21:07:35.964849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filling the missing values in the age column with the median age, since there are few missing values in this column\ncustomers_df['age'] = customers_df['age'].fillna(median_age)","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:07:35.967665Z","iopub.execute_input":"2022-03-28T21:07:35.967968Z","iopub.status.idle":"2022-03-28T21:07:35.991822Z","shell.execute_reply.started":"2022-03-28T21:07:35.967927Z","shell.execute_reply":"2022-03-28T21:07:35.990933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Merging with the customers dataset and keeping only age and fashion news frequency as features from the customers dataset\nall_merged = transactions_articles_merged.merge(customers_df, on='customer_id')\nkept_columns.extend(['age', 'fashion_news_frequency'])\nall_merged = all_merged[kept_columns]","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:07:35.992927Z","iopub.execute_input":"2022-03-28T21:07:35.993152Z","iopub.status.idle":"2022-03-28T21:07:37.13906Z","shell.execute_reply.started":"2022-03-28T21:07:35.993125Z","shell.execute_reply":"2022-03-28T21:07:37.138122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_merged.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:07:37.140424Z","iopub.execute_input":"2022-03-28T21:07:37.140854Z","iopub.status.idle":"2022-03-28T21:07:37.157008Z","shell.execute_reply.started":"2022-03-28T21:07:37.140824Z","shell.execute_reply":"2022-03-28T21:07:37.15604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The values of most columns need to be encoded. This is the case for columns that are categorical variables.\n# customer_id, article_id do not need to be encoded\n# age and news frequency can be used as they are defined, since they can be ordered (for news frequency: NONE < Monthly < Regularly)\nall_merged_ohe = pd.get_dummies(all_merged, columns=all_merged.columns[2:-2])\nall_merged_ohe.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:07:37.158177Z","iopub.execute_input":"2022-03-28T21:07:37.158431Z","iopub.status.idle":"2022-03-28T21:07:37.958199Z","shell.execute_reply.started":"2022-03-28T21:07:37.158402Z","shell.execute_reply":"2022-03-28T21:07:37.957197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_merged_ohe['customer_id'].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:07:37.96159Z","iopub.execute_input":"2022-03-28T21:07:37.962499Z","iopub.status.idle":"2022-03-28T21:07:38.055392Z","shell.execute_reply.started":"2022-03-28T21:07:37.9624Z","shell.execute_reply":"2022-03-28T21:07:38.05437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Too many customer ids for the model to finish in a reasonable time; selecting fewer transactions by using only customers that bought at least 3 times recently","metadata":{}},{"cell_type":"code","source":"# Will keep only the transactions of customers that have bought an article at least 3 times\nCUSTOMER_MIN_TRANSACTIONS = 3\n\ncustomers_num_purchases = all_merged_ohe.groupby('customer_id').size().reset_index(name='count')\ncustomers_min_purchases = customers_num_purchases[customers_num_purchases['count'] >= CUSTOMER_MIN_TRANSACTIONS]['customer_id']\n\nall_merged_ohe = all_merged_ohe[all_merged_ohe['customer_id'].isin(customers_min_purchases)]\nall_merged_ohe['customer_id'].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:07:38.056638Z","iopub.execute_input":"2022-03-28T21:07:38.056874Z","iopub.status.idle":"2022-03-28T21:07:38.999144Z","shell.execute_reply.started":"2022-03-28T21:07:38.056844Z","shell.execute_reply":"2022-03-28T21:07:38.997922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_merged_ohe.columns","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:07:39.000542Z","iopub.execute_input":"2022-03-28T21:07:39.000869Z","iopub.status.idle":"2022-03-28T21:07:39.008036Z","shell.execute_reply.started":"2022-03-28T21:07:39.000825Z","shell.execute_reply":"2022-03-28T21:07:39.006904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler = StandardScaler()\nscaled = scaler.fit_transform(all_merged_ohe.iloc[:, 2:])\nscaled_df = pd.DataFrame(scaled, columns=all_merged_ohe.columns[2:])\nscaled_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:07:39.009235Z","iopub.execute_input":"2022-03-28T21:07:39.009561Z","iopub.status.idle":"2022-03-28T21:07:40.804743Z","shell.execute_reply.started":"2022-03-28T21:07:39.009517Z","shell.execute_reply":"2022-03-28T21:07:40.803842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = all_merged_ohe[['customer_id', 'article_id']].reset_index(drop=True)\ndf2 = scaled_df.reset_index(drop=True)\nall_merged_scaled = pd.concat([df1, df2], axis=1)\nall_merged_scaled.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:07:40.806199Z","iopub.execute_input":"2022-03-28T21:07:40.806563Z","iopub.status.idle":"2022-03-28T21:07:41.483864Z","shell.execute_reply.started":"2022-03-28T21:07:40.806508Z","shell.execute_reply":"2022-03-28T21:07:41.482898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_dataframe_to_csv(dataframe, filename):\n    output_dir = '/kaggle/working'\n    filepath = os.path.join(output_dir, filename)\n    dataframe.to_csv(filepath, index=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:07:41.485441Z","iopub.execute_input":"2022-03-28T21:07:41.485691Z","iopub.status.idle":"2022-03-28T21:07:41.490697Z","shell.execute_reply.started":"2022-03-28T21:07:41.485662Z","shell.execute_reply":"2022-03-28T21:07:41.489763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_dataframe_to_csv(all_merged_scaled, 'all_merged_scaled.csv')","metadata":{"execution":{"iopub.status.busy":"2022-03-28T19:36:43.9181Z","iopub.execute_input":"2022-03-28T19:36:43.918439Z","iopub.status.idle":"2022-03-28T19:39:54.16675Z","shell.execute_reply.started":"2022-03-28T19:36:43.918403Z","shell.execute_reply":"2022-03-28T19:39:54.165536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dataset for the second model","metadata":{}},{"cell_type":"code","source":"# Selecting a small part of the dataset\nsampled = recent_transactions_df.sample(n=200000)[['customer_id', 'article_id']]","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:09:40.617657Z","iopub.execute_input":"2022-03-28T21:09:40.618545Z","iopub.status.idle":"2022-03-28T21:09:40.662493Z","shell.execute_reply.started":"2022-03-28T21:09:40.618501Z","shell.execute_reply":"2022-03-28T21:09:40.661725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sampled.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:09:42.063899Z","iopub.execute_input":"2022-03-28T21:09:42.064202Z","iopub.status.idle":"2022-03-28T21:09:42.073389Z","shell.execute_reply.started":"2022-03-28T21:09:42.064163Z","shell.execute_reply":"2022-03-28T21:09:42.072618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_dataframe_to_csv(sampled, 'small_transactions.csv')","metadata":{"execution":{"iopub.status.busy":"2022-03-28T21:09:47.636656Z","iopub.execute_input":"2022-03-28T21:09:47.637235Z","iopub.status.idle":"2022-03-28T21:09:48.548905Z","shell.execute_reply.started":"2022-03-28T21:09:47.637182Z","shell.execute_reply":"2022-03-28T21:09:48.548124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Computing the top 12 most popular items recently bought (to recommend by default if we didn't compute a recommendation for the customer)","metadata":{}},{"cell_type":"code","source":"article_popularity_df = recent_transactions_df.groupby('article_id').count().reset_index().iloc[:, :2]\narticle_popularity_df.columns = ['article_id', 'count']\narticle_popularity_df = article_popularity_df.sort_values(by='count', ascending=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For the customers that are included in the dataset but for whom recommendations could not be made due to the size of the problem,\n# a default recommendation of the top 12 bought products in the timeframe will be made\ntop12_popular = article_popularity_df['article_id'].head(12)\nsave_dataframe_to_csv(top12_popular, 'top12_popular.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving the list of all customers in the dataset as they are not all part of the datasets used in training the models\nall_customers = pd.DataFrame(customers_df['customer_id'], columns=['customer_id'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_dataframe_to_csv(all_customers, 'all_customers.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}