{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Commons","metadata":{}},{"cell_type":"code","source":"# Data processing \nimport numpy as np\nimport pandas as pd\n\n# Data visualisation\nimport seaborn as sns\n\n%matplotlib inline\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\n\nmpl.rc('axes',  labelsize=14)\nmpl.rc('xtick', labelsize=12)\nmpl.rc('ytick', labelsize=12)\n\n# Commons\nimport gc\nimport os\nimport math\nimport random\nRANDOM_STATE = 42\nnp.random.seed(RANDOM_STATE)\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-29T07:42:16.970367Z","iopub.execute_input":"2022-03-29T07:42:16.970658Z","iopub.status.idle":"2022-03-29T07:42:17.753716Z","shell.execute_reply.started":"2022-03-29T07:42:16.970626Z","shell.execute_reply":"2022-03-29T07:42:17.752926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ARTICLES_CSV_PATH      = '../input/h-and-m-personalized-fashion-recommendations/articles.csv'\nCUSTOMERS_CSV_PATH     = '../input/h-and-m-personalized-fashion-recommendations/customers.csv'\nTRANSACTIONS_CSV_PATH  = '../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv'\nSAMPLE_SUBMISSION_PATH = '../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv'\nARTICLES_IMAGES_PATH   = '../input/h-and-m-personalized-fashion-recommendations/images/'","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:42:17.755318Z","iopub.execute_input":"2022-03-29T07:42:17.755566Z","iopub.status.idle":"2022-03-29T07:42:17.759373Z","shell.execute_reply.started":"2022-03-29T07:42:17.755532Z","shell.execute_reply":"2022-03-29T07:42:17.758727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def stop_execution():\n    \"\"\" \n    Stop the execution of the program\n    \"\"\"\n    raise SystemExit()\n\n    \ndef is_kaggle_gpu_enabled():\n    \"\"\"\n    Return whether GPU is enabled in the running Kaggle kernel\n    \"\"\"\n    \n    import tensorflow as tf\n    return len(tf.config.list_physical_devices('GPU')) > 0\n\n\ndef csv_to_df(csv_path, cudf=False):\n    \"\"\" \n    Convert a `csv` file to `pd/cudf Dataframe`\n    If the given file path is not correct, then `None` is returned\n    \n    @param csv_path: Path to the csv file\n    @return: Parsed csv file or `None`\n    @rtype: pd/cudf.Dataframe    \n    \"\"\"\n    \n    if not os.path.isfile(csv_path):\n        print(f\"The file '{csv_path}' doesn't exist!\")\n        return None\n    \n    if cudf == True:\n        return cudf.read_csv(csv_path)\n    return pd.read_csv(csv_path)\n\n\ndef group_data_by(data, groupby, countby):\n    \"\"\"\n    Group data based on the column `groupby`\n    and count values by after the `countby` param\n    \n    @param data: Dataframe to be grouped\n    @param groupby: Key used to group the data\n    @param countby: Count the instances based on this key\n    @return: df with 2 columns: `groupby` and `count`\n    @rtype: Dataframe\n    \"\"\"\n    \n    # Check if the param `groupby` is a nested list \n    if not any(isinstance(x, list) for x in groupby):\n        group_by = groupby\n    else:\n        group_by = [groupby]\n    \n    grouped_data = data                  \\\n        .groupby(group_by)               \\\n        .count()[countby]                \\\n        .sort_values(ascending=False)    \\\n        .reset_index()\n    return grouped_data.rename(columns={countby: 'count'})","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:42:17.760797Z","iopub.execute_input":"2022-03-29T07:42:17.761336Z","iopub.status.idle":"2022-03-29T07:42:17.777091Z","shell.execute_reply.started":"2022-03-29T07:42:17.761298Z","shell.execute_reply":"2022-03-29T07:42:17.776193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check if Kaggle GPU's kernel is enabled\nif not is_kaggle_gpu_enabled():\n    print('Enable Kaggle GPU before running this notebook!')\n    stop_execution()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:42:17.778587Z","iopub.execute_input":"2022-03-29T07:42:17.779065Z","iopub.status.idle":"2022-03-29T07:42:20.619841Z","shell.execute_reply.started":"2022-03-29T07:42:17.779021Z","shell.execute_reply":"2022-03-29T07:42:20.619023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# In will use `RAPIDS cuDF` instead of `pd df`\n# for faster dataframe manipulations (requires GPU)\nimport cudf","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:42:20.622518Z","iopub.execute_input":"2022-03-29T07:42:20.623037Z","iopub.status.idle":"2022-03-29T07:42:22.934646Z","shell.execute_reply.started":"2022-03-29T07:42:20.622994Z","shell.execute_reply":"2022-03-29T07:42:22.933243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Recommend using only dataframe manipulations","metadata":{}},{"cell_type":"code","source":"# For this approach, I will try to implement a recommendation system\n# using only the existing dataframes, without involving a ML algorithm\n\n# 1. I will recommend items previously purchased by the current `customer_id`\n# 2. Then, items that are bought together with previous purchases\n# 3. At the end, we can also recommend popular items","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:42:22.938713Z","iopub.execute_input":"2022-03-29T07:42:22.939784Z","iopub.status.idle":"2022-03-29T07:42:22.946626Z","shell.execute_reply.started":"2022-03-29T07:42:22.939702Z","shell.execute_reply":"2022-03-29T07:42:22.944857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load the data","metadata":{}},{"cell_type":"code","source":"articles     = cudf.DataFrame.from_pandas(csv_to_df(ARTICLES_CSV_PATH))\ncustomers    = cudf.DataFrame.from_pandas(csv_to_df(CUSTOMERS_CSV_PATH))\ntransactions = cudf.DataFrame.from_pandas(csv_to_df(TRANSACTIONS_CSV_PATH))\n\n# Sanity check\nif any(df is None for df in [articles,\n                             customers,\n                             transactions]):\n    stop_execution()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:42:22.948384Z","iopub.execute_input":"2022-03-29T07:42:22.949834Z","iopub.status.idle":"2022-03-29T07:43:08.820717Z","shell.execute_reply.started":"2022-03-29T07:42:22.949749Z","shell.execute_reply":"2022-03-29T07:43:08.819876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reduce some memory","metadata":{}},{"cell_type":"code","source":"# Make a deep copy of the `transactions` df\ntrain = cudf.DataFrame(transactions)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:08.822205Z","iopub.execute_input":"2022-03-29T07:43:08.822464Z","iopub.status.idle":"2022-03-29T07:43:08.857527Z","shell.execute_reply.started":"2022-03-29T07:43:08.822429Z","shell.execute_reply":"2022-03-29T07:43:08.856868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:08.859516Z","iopub.execute_input":"2022-03-29T07:43:08.859995Z","iopub.status.idle":"2022-03-29T07:43:08.865976Z","shell.execute_reply.started":"2022-03-29T07:43:08.859958Z","shell.execute_reply":"2022-03-29T07:43:08.865004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I will use this 'trick' to reduce the memory\n# Here are all the explanations why I've used that's kind of mapping\n# https://www.kaggle.com/c/h-and-m-personalized-fashion-recommendations/discussion/308635\n\n# Reduce `customer_id` size (64bytes -> 8bytes)\ntrain['customer_id'] = train['customer_id'] \\\n    .str[-16:]                              \\\n    .str.hex_to_int()                       \\\n    .astype('int64')                        \\\n\n# Reduce `article_id` size (10bytes -> 4bytes)\ntrain['article_id'] = train['article_id'].astype('int32')\n\n# Convert `t_dat` to `cudf datetime`\ntrain['t_dat'] = cudf.to_datetime(train['t_dat'])","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:08.867649Z","iopub.execute_input":"2022-03-29T07:43:08.867911Z","iopub.status.idle":"2022-03-29T07:43:09.276148Z","shell.execute_reply.started":"2022-03-29T07:43:08.867876Z","shell.execute_reply":"2022-03-29T07:43:09.275482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:09.277509Z","iopub.execute_input":"2022-03-29T07:43:09.277749Z","iopub.status.idle":"2022-03-29T07:43:09.302803Z","shell.execute_reply.started":"2022-03-29T07:43:09.277714Z","shell.execute_reply":"2022-03-29T07:43:09.302161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save this version of `train reduced memory`\n# V1) Make a deep copy of the current df and store it in the RAM memory,\n# which is not a good idea: train_mem_red = cudf.DataFrame(train)\n# V2) Save it to a file and load when needed (lazy initialization)\ntrain.to_parquet('train.pqt')","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:09.303813Z","iopub.execute_input":"2022-03-29T07:43:09.304553Z","iopub.status.idle":"2022-03-29T07:43:10.130651Z","shell.execute_reply.started":"2022-03-29T07:43:09.304515Z","shell.execute_reply":"2022-03-29T07:43:10.129788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## (i) Recommend previously purchased items","metadata":{}},{"cell_type":"code","source":"# For this, we will only need the columns `t_dat`, `customer_id` and `article_id`\n# So, we can drop the `price` and `sales_channel_id` attributes\ntrain.drop(columns=['price', 'sales_channel_id'], inplace=True)\ntrain.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:10.132125Z","iopub.execute_input":"2022-03-29T07:43:10.132377Z","iopub.status.idle":"2022-03-29T07:43:10.153816Z","shell.execute_reply.started":"2022-03-29T07:43:10.13234Z","shell.execute_reply":"2022-03-29T07:43:10.152908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For each customer, compute the last date of purchases, individually\ntrain_grouped_customer_id = train.groupby('customer_id')\ntrain_last_purchase = train_grouped_customer_id['t_dat'].max()\ntrain_last_purchase = train_last_purchase.reset_index()\ntrain_last_purchase = train_last_purchase.rename(columns={'t_dat': 'max_dat'})\ntrain_last_purchase.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:10.157838Z","iopub.execute_input":"2022-03-29T07:43:10.158036Z","iopub.status.idle":"2022-03-29T07:43:10.222588Z","shell.execute_reply.started":"2022-03-29T07:43:10.158012Z","shell.execute_reply":"2022-03-29T07:43:10.221772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's now merge the `max_dat` column to the `train` df\ntrain = train.merge(train_last_purchase, on=['customer_id'], how='left')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:10.223821Z","iopub.execute_input":"2022-03-29T07:43:10.22415Z","iopub.status.idle":"2022-03-29T07:43:10.304393Z","shell.execute_reply.started":"2022-03-29T07:43:10.224114Z","shell.execute_reply":"2022-03-29T07:43:10.303421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Delete unused objects\ndel train_grouped_customer_id\ndel train_last_purchase\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:10.305779Z","iopub.execute_input":"2022-03-29T07:43:10.306033Z","iopub.status.idle":"2022-03-29T07:43:10.589061Z","shell.execute_reply.started":"2022-03-29T07:43:10.305999Z","shell.execute_reply":"2022-03-29T07:43:10.588199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can now compute the difference between the `max_dat` and `t_dat`\n# In other words, calculate last_purchase_date - curr_purchase_date, for each customer's transaction\ntrain['diff_dat'] = train['max_dat'] - train['t_dat']\n\n# Convert to days\ntrain['diff_dat'] = train['diff_dat'].dt.days\n\n# Now, select purchases within the last 2 weeks \ntrain = train.loc[train['diff_dat'] <= 14]","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:10.59088Z","iopub.execute_input":"2022-03-29T07:43:10.591379Z","iopub.status.idle":"2022-03-29T07:43:10.639034Z","shell.execute_reply.started":"2022-03-29T07:43:10.591328Z","shell.execute_reply":"2022-03-29T07:43:10.638362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:10.640445Z","iopub.execute_input":"2022-03-29T07:43:10.640948Z","iopub.status.idle":"2022-03-29T07:43:10.671947Z","shell.execute_reply.started":"2022-03-29T07:43:10.640909Z","shell.execute_reply":"2022-03-29T07:43:10.671296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ~5.6mil customers who bought articles in the last 2 weeks\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:10.67337Z","iopub.execute_input":"2022-03-29T07:43:10.673983Z","iopub.status.idle":"2022-03-29T07:43:10.679624Z","shell.execute_reply.started":"2022-03-29T07:43:10.673941Z","shell.execute_reply":"2022-03-29T07:43:10.678748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Group data by `customer_id` and `article_id` and count by `t_dat`\ntrain_grouped = group_data_by(data=train,\n                              groupby=['customer_id', 'article_id'],\n                              countby='t_dat')\ntrain_grouped.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:10.681379Z","iopub.execute_input":"2022-03-29T07:43:10.681956Z","iopub.status.idle":"2022-03-29T07:43:10.757039Z","shell.execute_reply.started":"2022-03-29T07:43:10.681917Z","shell.execute_reply":"2022-03-29T07:43:10.756374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Merge `train_grouped` and `train`\n# Sort in descending order by `count` and then by `t_dat`\ntrain = train.merge(train_grouped, on=['customer_id', 'article_id'], how='left')\ntrain = train.drop_duplicates(['customer_id', 'article_id'])\ntrain = train.sort_values(['count','t_dat'], ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:10.758358Z","iopub.execute_input":"2022-03-29T07:43:10.758825Z","iopub.status.idle":"2022-03-29T07:43:10.9381Z","shell.execute_reply.started":"2022-03-29T07:43:10.758789Z","shell.execute_reply":"2022-03-29T07:43:10.937407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:10.940353Z","iopub.execute_input":"2022-03-29T07:43:10.942373Z","iopub.status.idle":"2022-03-29T07:43:10.97176Z","shell.execute_reply.started":"2022-03-29T07:43:10.942295Z","shell.execute_reply":"2022-03-29T07:43:10.970966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:10.97328Z","iopub.execute_input":"2022-03-29T07:43:10.973633Z","iopub.status.idle":"2022-03-29T07:43:10.979175Z","shell.execute_reply.started":"2022-03-29T07:43:10.973593Z","shell.execute_reply":"2022-03-29T07:43:10.978337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## (ii) Recommend items that are bought together with previous purchases","metadata":{}},{"cell_type":"code","source":"# This function is inspired from:\n# https://www.kaggle.com/code/cdeotte/customers-who-bought-this-frequently-buy-this/notebook#Find-Items-Purchased-Together\ndef dict_items_freq(data):\n    \"\"\"\n    Generate a dictionary of type {itemX: [items_bought_together_with_itemX]}\n    List of recommended itemes has length maximum of 3\n    \n    @param `data`: Dataframe with 2 mandatory columns: `customer_id` `article_id`\n    @return: Computed dictionary of 'pairs'\n    \"\"\"\n    \n    # Compute a df with the best-selling articles in descending order\n    best_selling = data['article_id'].value_counts()\n    \n    # Create a dictionary with the following `signature`: item -> [bought_together_items]\n    pairs = {}\n    for _, i in enumerate(best_selling.index.values[1000:1032]):\n        # Select the users who bought the current `item`\n        users = data.loc[data['article_id'] == i.item(), 'customer_id'].unique()\n        \n        # Compute a df with the most similar articles with the current `item` (excluding itself)\n        best_selling_similars = data.loc[(data['customer_id'].isin(users)) &             \\\n                                        (data['article_id'] != i.item()), 'article_id']  \\\n                                     .value_counts()\n        \n        # Assign a list of 3 `bought_together_items` to the current `item`\n        pairs[i.item()] = [best_selling_similars.index[0],\n                           best_selling_similars.index[1],\n                           best_selling_similars.index[2]]\n    return pairs","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:10.980744Z","iopub.execute_input":"2022-03-29T07:43:10.981051Z","iopub.status.idle":"2022-03-29T07:43:10.989735Z","shell.execute_reply.started":"2022-03-29T07:43:10.980963Z","shell.execute_reply":"2022-03-29T07:43:10.988766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate the dictionary of items frequently purchased together\npairs = dict_items_freq(train)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:10.991265Z","iopub.execute_input":"2022-03-29T07:43:10.991541Z","iopub.status.idle":"2022-03-29T07:43:12.937877Z","shell.execute_reply.started":"2022-03-29T07:43:10.991507Z","shell.execute_reply":"2022-03-29T07:43:12.937112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert from `cudf` to `pandas` for `mapping` each item to another 3 `bought together` items\ntrain = train.to_pandas()\n# Create a new column from dictionary (map each item to a new column)\ntrain['article_id_aux'] = train['article_id'].map(pairs)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:12.939558Z","iopub.execute_input":"2022-03-29T07:43:12.939852Z","iopub.status.idle":"2022-03-29T07:43:13.39955Z","shell.execute_reply.started":"2022-03-29T07:43:12.939811Z","shell.execute_reply":"2022-03-29T07:43:13.398767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a df with the following columns: `customer_id` -> [`recommend_3_bought_together_items`]\ntrain_aux = train[['customer_id', 'article_id_aux']].copy()\ntrain_aux = train_aux.rename({'article_id_aux': 'article_id'}, axis=1)\ntrain_aux = train_aux.loc[train_aux['article_id'].notnull()]\ntrain_aux = train_aux.drop_duplicates(['customer_id'])","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:13.401046Z","iopub.execute_input":"2022-03-29T07:43:13.401315Z","iopub.status.idle":"2022-03-29T07:43:13.805043Z","shell.execute_reply.started":"2022-03-29T07:43:13.401274Z","shell.execute_reply":"2022-03-29T07:43:13.804266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_aux.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:13.806519Z","iopub.execute_input":"2022-03-29T07:43:13.806774Z","iopub.status.idle":"2022-03-29T07:43:13.817469Z","shell.execute_reply.started":"2022-03-29T07:43:13.80674Z","shell.execute_reply":"2022-03-29T07:43:13.816691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_aux.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:13.818651Z","iopub.execute_input":"2022-03-29T07:43:13.819049Z","iopub.status.idle":"2022-03-29T07:43:13.828479Z","shell.execute_reply.started":"2022-03-29T07:43:13.819011Z","shell.execute_reply":"2022-03-29T07:43:13.82764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Keep only the `customer_id` and `article_id` columns\ntrain = train[['customer_id','article_id']]\n# Concatenate the `train` and `train_aux` on rows \ntrain = pd.concat([train, train_aux], axis=0, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:13.830578Z","iopub.execute_input":"2022-03-29T07:43:13.831179Z","iopub.status.idle":"2022-03-29T07:43:14.082069Z","shell.execute_reply.started":"2022-03-29T07:43:13.83114Z","shell.execute_reply":"2022-03-29T07:43:14.081296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:14.084429Z","iopub.execute_input":"2022-03-29T07:43:14.084975Z","iopub.status.idle":"2022-03-29T07:43:14.094292Z","shell.execute_reply.started":"2022-03-29T07:43:14.084933Z","shell.execute_reply":"2022-03-29T07:43:14.093282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:14.095884Z","iopub.execute_input":"2022-03-29T07:43:14.096153Z","iopub.status.idle":"2022-03-29T07:43:14.105166Z","shell.execute_reply.started":"2022-03-29T07:43:14.096117Z","shell.execute_reply":"2022-03-29T07:43:14.104342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Delete unused objects\ndel train_aux\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:14.106529Z","iopub.execute_input":"2022-03-29T07:43:14.106808Z","iopub.status.idle":"2022-03-29T07:43:14.302295Z","shell.execute_reply.started":"2022-03-29T07:43:14.10675Z","shell.execute_reply":"2022-03-29T07:43:14.301433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert recommendations into a single string\ntrain['article_id'] = ' 0' + train['article_id'].astype('str')\npredictions = cudf.DataFrame(train                            \\\n                       .groupby('customer_id')['article_id']  \\\n                       .sum()                                 \\\n                       .reset_index())\npredictions.rename(columns={'article_id': 'prediction'}, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:14.304923Z","iopub.execute_input":"2022-03-29T07:43:14.305665Z","iopub.status.idle":"2022-03-29T07:43:19.101453Z","shell.execute_reply.started":"2022-03-29T07:43:14.305618Z","shell.execute_reply":"2022-03-29T07:43:19.100657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:19.102614Z","iopub.execute_input":"2022-03-29T07:43:19.102895Z","iopub.status.idle":"2022-03-29T07:43:19.122859Z","shell.execute_reply.started":"2022-03-29T07:43:19.102857Z","shell.execute_reply":"2022-03-29T07:43:19.121883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## (iii) Recommend last 2 weeks popular items","metadata":{}},{"cell_type":"code","source":"# Customizable variables\nlast_n_days = 14\ntop_n_popular_items = 10\n\n# Open the `reduced memory train df`\ntrain = cudf.read_parquet('train.pqt')\n\n# Convert `t_dat` field to `cudf datetime`\n# Serialization doesn't keep the `datetime` format\ntrain['t_dat'] = cudf.to_datetime(train['t_dat'])\n\n# Keep only the `last_n_days` entries\ntrain = train.loc[train['t_dat'] >= cudf.to_datetime(train['t_dat'].max() - np.timedelta64(last_n_days, 'D'))]\n\n# Compute the `top_n_popular_items` popular items within the`last_n_days`\ntop = ' 0' + ' 0'.join(train['article_id'].value_counts().to_pandas().index.astype('str')[:top_n_popular_items])\nprint(f'Top {top_n_popular_items} popular items within the last {last_n_days} days:')\nprint(top)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:19.124509Z","iopub.execute_input":"2022-03-29T07:43:19.124992Z","iopub.status.idle":"2022-03-29T07:43:19.949161Z","shell.execute_reply.started":"2022-03-29T07:43:19.124953Z","shell.execute_reply":"2022-03-29T07:43:19.948231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generate submission file","metadata":{}},{"cell_type":"code","source":"# Open the `submission_sample` df\nsubmission = cudf.read_csv(SAMPLE_SUBMISSION_PATH)\nsubmission = submission[['customer_id']]\n\n# Reduce `customer_id` size (64bytes -> 8bytes)\nsubmission['customer_id_aux'] = submission['customer_id'] \\\n                                .str[-16:]                \\\n                                .str.hex_to_int()         \\\n                                .astype('int64')\n\n# Rename `customer_id` to `customer_id_aux` for being able to merge them\npredictions.rename({'customer_id': 'customer_id_aux'}, axis=1, inplace=True)\n\n# Merge the `submission` df with the `predictions` df,\n# filling the missing values with the empty string\nsubmission = submission.merge(predictions, on='customer_id_aux', how='left').fillna('')\n\n# Remove the `customer_id_aux` column\ndel submission['customer_id_aux']\n\n# Append the `top` popular items to the current predictions\nsubmission['prediction'] = submission['prediction'] + top\n\n# Convert to string and remove redundant spaces\nsubmission['prediction'] = submission['prediction'].str.strip()\n\n# Each `article_id` from the `prediction` column has exactly 10 characters\n# So, if we want to predict 1 item, we need 10 characters\n# 2 items -> 2 * 10 + 1 (space between them)\n# 3 items -> 3 * 10 + 2 (spaces between them)\n# ...\n# n items -> n * 10 + (n - 1) = 11 * n - 1\npredict_n_items = 12\ncharacters_required = 11 * predict_n_items - 1\nsubmission['prediction'] = submission['prediction'].str[:characters_required]\n\n# Generate the `submission.csv` file\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:19.950797Z","iopub.execute_input":"2022-03-29T07:43:19.951073Z","iopub.status.idle":"2022-03-29T07:43:21.736643Z","shell.execute_reply.started":"2022-03-29T07:43:19.951034Z","shell.execute_reply":"2022-03-29T07:43:21.735718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:21.737806Z","iopub.execute_input":"2022-03-29T07:43:21.738471Z","iopub.status.idle":"2022-03-29T07:43:21.744228Z","shell.execute_reply.started":"2022-03-29T07:43:21.738431Z","shell.execute_reply":"2022-03-29T07:43:21.743545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:21.745572Z","iopub.execute_input":"2022-03-29T07:43:21.747597Z","iopub.status.idle":"2022-03-29T07:43:21.773569Z","shell.execute_reply.started":"2022-03-29T07:43:21.747555Z","shell.execute_reply":"2022-03-29T07:43:21.770549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del transactions\ndel customers\ndel articles\ndel train\ndel predictions\ndel submission\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:21.777358Z","iopub.execute_input":"2022-03-29T07:43:21.777938Z","iopub.status.idle":"2022-03-29T07:43:22.06606Z","shell.execute_reply.started":"2022-03-29T07:43:21.777902Z","shell.execute_reply":"2022-03-29T07:43:22.065331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# alg1_score: 0.0210","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:22.071353Z","iopub.execute_input":"2022-03-29T07:43:22.071709Z","iopub.status.idle":"2022-03-29T07:43:22.075198Z","shell.execute_reply.started":"2022-03-29T07:43:22.071679Z","shell.execute_reply":"2022-03-29T07:43:22.074255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. ALS Model","metadata":{}},{"cell_type":"markdown","source":"**In this part of the notebook I will use the [implicit](https://github.com/benfred/implicit/) library. It's a module used for recommender\nsystems and has GPU support, which is nice.**\n\n**It has a lot of models, but I will use the `AlternatingLeastSquares` algorithm. You can read more about this approach [here](https://towardsdatascience.com/prototyping-a-recommender-system-step-by-step-part-2-alternating-least-square-als-matrix-4a76c58714a1).**\n\n**There were done a lot of changes recently in the API, so be aware of that. This notebook uses the version 0.5.2, so everything newer than 0.5.0 should work well.**\n\n**Older versions of the package doesn't have support for batch (multi-users) recommendations, hence the process it's slower when recommending items for each user individually.**","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"# For this approach, I will use the `implicit` library\n!pip install --upgrade implicit\n# Also, install the `scipy` module\n!pip install --upgrade scipy","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:22.077472Z","iopub.execute_input":"2022-03-29T07:43:22.078389Z","iopub.status.idle":"2022-03-29T07:43:37.486186Z","shell.execute_reply.started":"2022-03-29T07:43:22.078349Z","shell.execute_reply":"2022-03-29T07:43:37.485248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This notebook was tested on v0.5.2 of `implicit` package\n# Versions newer than 0.5.0 work well with GPU out-of-the-box\n# There are big improvements in terms of efficiency\nimport implicit\nprint(f'`implicit` version: {implicit.__version__}')\nif implicit.__version__ < '0.5.0':\n    print('Requires `implicit` 0.5.0 or newer')\n    stop_execution()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:37.490038Z","iopub.execute_input":"2022-03-29T07:43:37.490289Z","iopub.status.idle":"2022-03-29T07:43:37.520716Z","shell.execute_reply.started":"2022-03-29T07:43:37.49026Z","shell.execute_reply":"2022-03-29T07:43:37.519915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.sparse import csr_matrix\nfrom implicit.als import AlternatingLeastSquares\nfrom implicit.evaluation import mean_average_precision_at_k","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:37.52223Z","iopub.execute_input":"2022-03-29T07:43:37.522546Z","iopub.status.idle":"2022-03-29T07:43:37.530912Z","shell.execute_reply.started":"2022-03-29T07:43:37.522505Z","shell.execute_reply":"2022-03-29T07:43:37.530024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load the data","metadata":{}},{"cell_type":"code","source":"# Load the dataframes\narticles     = cudf.DataFrame.from_pandas(csv_to_df(ARTICLES_CSV_PATH))\ncustomers    = cudf.DataFrame.from_pandas(csv_to_df(CUSTOMERS_CSV_PATH))\ntransactions = cudf.DataFrame.from_pandas(csv_to_df(TRANSACTIONS_CSV_PATH))\n\n# Sanity check\nif any(df is None for df in [articles,\n                             customers,\n                             transactions]):\n    stop_execution()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:43:37.532327Z","iopub.execute_input":"2022-03-29T07:43:37.532845Z","iopub.status.idle":"2022-03-29T07:44:19.504697Z","shell.execute_reply.started":"2022-03-29T07:43:37.532782Z","shell.execute_reply":"2022-03-29T07:44:19.503898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert `t_dat` to `cudf datetime`\ntransactions['t_dat'] = cudf.to_datetime(transactions['t_dat'])","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:19.506159Z","iopub.execute_input":"2022-03-29T07:44:19.506441Z","iopub.status.idle":"2022-03-29T07:44:19.531638Z","shell.execute_reply.started":"2022-03-29T07:44:19.506403Z","shell.execute_reply":"2022-03-29T07:44:19.530948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert `article_ids` to `str`\ntransactions = transactions.astype({'article_id': 'str'})\narticles = articles.astype({'article_id': 'str'})","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:19.532854Z","iopub.execute_input":"2022-03-29T07:44:19.533119Z","iopub.status.idle":"2022-03-29T07:44:19.585567Z","shell.execute_reply.started":"2022-03-29T07:44:19.533083Z","shell.execute_reply":"2022-03-29T07:44:19.58493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:19.587337Z","iopub.execute_input":"2022-03-29T07:44:19.587743Z","iopub.status.idle":"2022-03-29T07:44:19.613004Z","shell.execute_reply.started":"2022-03-29T07:44:19.587709Z","shell.execute_reply":"2022-03-29T07:44:19.61233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:19.614164Z","iopub.execute_input":"2022-03-29T07:44:19.614596Z","iopub.status.idle":"2022-03-29T07:44:19.619783Z","shell.execute_reply.started":"2022-03-29T07:44:19.614559Z","shell.execute_reply":"2022-03-29T07:44:19.618985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Use only the last month of transactions","metadata":{}},{"cell_type":"code","source":"# Let's use only the last month of transactions for this implementation\n# `cudf` doesn't have support for timestamps substraction. This is a possible workaround\ntransactions = transactions[transactions['t_dat'] >= transactions['t_dat'].max() - np.timedelta64(30, 'D')]","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:19.62089Z","iopub.execute_input":"2022-03-29T07:44:19.62176Z","iopub.status.idle":"2022-03-29T07:44:19.646946Z","shell.execute_reply.started":"2022-03-29T07:44:19.621718Z","shell.execute_reply":"2022-03-29T07:44:19.646289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:19.64795Z","iopub.execute_input":"2022-03-29T07:44:19.648264Z","iopub.status.idle":"2022-03-29T07:44:19.65392Z","shell.execute_reply.started":"2022-03-29T07:44:19.648228Z","shell.execute_reply":"2022-03-29T07:44:19.653219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create a 1:1 mapping from current ids to some incremental ids","metadata":{}},{"cell_type":"code","source":"# Convert the columns `customer_id` and `article_id` in 2 lists\nusers_list = customers['customer_id'].unique().to_arrow().to_pylist()\nitems_list = articles['article_id'].unique().to_arrow().to_pylist()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:19.655092Z","iopub.execute_input":"2022-03-29T07:44:19.65567Z","iopub.status.idle":"2022-03-29T07:44:21.518958Z","shell.execute_reply.started":"2022-03-29T07:44:19.655632Z","shell.execute_reply":"2022-03-29T07:44:21.51816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sanity checks\nassert len(users_list) == customers['customer_id'].nunique()\nassert len(items_list) == articles['article_id'].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:21.521187Z","iopub.execute_input":"2022-03-29T07:44:21.521983Z","iopub.status.idle":"2022-03-29T07:44:21.568361Z","shell.execute_reply.started":"2022-03-29T07:44:21.521941Z","shell.execute_reply":"2022-03-29T07:44:21.567712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Map each customer's id (and article's id) to an index starting from 0\nusers_map = {}\nfor idx, user in enumerate(users_list):\n    users_map[user] = idx\n    \nitems_map = {}\nfor idx, item in enumerate(items_list):\n    items_map[item] = idx","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:21.569826Z","iopub.execute_input":"2022-03-29T07:44:21.570117Z","iopub.status.idle":"2022-03-29T07:44:22.173552Z","shell.execute_reply.started":"2022-03-29T07:44:21.570079Z","shell.execute_reply":"2022-03-29T07:44:22.172768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create new columns, using the bijection generated above\ntransactions['user_id'] = transactions['customer_id'].map(users_map)\ntransactions['item_id'] = transactions['article_id'].map(items_map)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:22.175023Z","iopub.execute_input":"2022-03-29T07:44:22.175319Z","iopub.status.idle":"2022-03-29T07:44:22.592757Z","shell.execute_reply.started":"2022-03-29T07:44:22.175279Z","shell.execute_reply":"2022-03-29T07:44:22.591972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:22.594203Z","iopub.execute_input":"2022-03-29T07:44:22.594484Z","iopub.status.idle":"2022-03-29T07:44:22.622356Z","shell.execute_reply.started":"2022-03-29T07:44:22.594437Z","shell.execute_reply":"2022-03-29T07:44:22.621557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sanity checks: Make sure that the 1:1 mapping is correct\nassert transactions['article_id'].nunique()  == transactions['item_id'].nunique()\nassert transactions['customer_id'].nunique() == transactions['user_id'].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:22.623733Z","iopub.execute_input":"2022-03-29T07:44:22.624Z","iopub.status.idle":"2022-03-29T07:44:22.870123Z","shell.execute_reply.started":"2022-03-29T07:44:22.623963Z","shell.execute_reply":"2022-03-29T07:44:22.869387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The purpose of mapping each customer to a smaller (integer) index\n# is to handle indices easier in the computed CSR matrix of type (users x items)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:22.871372Z","iopub.execute_input":"2022-03-29T07:44:22.871654Z","iopub.status.idle":"2022-03-29T07:44:22.875608Z","shell.execute_reply.started":"2022-03-29T07:44:22.871617Z","shell.execute_reply":"2022-03-29T07:44:22.874852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute the CSR matrix\ndef users_items_csr(data):\n    \"\"\"\n    Generate a CSR matrix given the input `data` df\n    CSR matrix details: https://en.wikipedia.org/wiki/Sparse_matrix#Compressed_sparse_row_(CSR,_CRS_or_Yale_format)\n    \n    @param `data`: Dataframe with mandatory 2 columns: `user_id` and `item_id`\n    @return: Computed CSR matrix of type (users x items)\n    \n    Variable `values` is an array full of ones, because our `csr_matrix` will be a binary matrix\n    We only know if the user bought a certain article\n    We don't know what 'rating' the user has given to that \n    \"\"\"\n    \n    users = data['user_id'].values.get() # Rows\n    items = data['item_id'].values.get() # Cols\n    values = np.ones(data.shape[0])      # Values\n    return csr_matrix((values, (users, items)), shape=(len(users_list), len(items_list)))","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:22.876944Z","iopub.execute_input":"2022-03-29T07:44:22.877679Z","iopub.status.idle":"2022-03-29T07:44:22.886158Z","shell.execute_reply.started":"2022-03-29T07:44:22.877634Z","shell.execute_reply":"2022-03-29T07:44:22.885388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's check if the `AlternatingLeastSquares` model works with our data format\n# We need to `fit` the model with a csr/coo matrix of type (users x items) \nals = AlternatingLeastSquares(iterations=3)\ncsr_mat = users_items_csr(transactions)\nals.fit(csr_mat)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:22.88724Z","iopub.execute_input":"2022-03-29T07:44:22.887522Z","iopub.status.idle":"2022-03-29T07:44:24.062717Z","shell.execute_reply.started":"2022-03-29T07:44:22.887468Z","shell.execute_reply":"2022-03-29T07:44:24.061942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Perfect, everything went ok.\n# Now we can start to manipulate our main df\ntransactions.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:24.064047Z","iopub.execute_input":"2022-03-29T07:44:24.064789Z","iopub.status.idle":"2022-03-29T07:44:24.07101Z","shell.execute_reply.started":"2022-03-29T07:44:24.064746Z","shell.execute_reply":"2022-03-29T07:44:24.070322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Use the entire dataset (last month)","metadata":{}},{"cell_type":"code","source":"# I will use the last week as validation set and the first 3 weeks as training set \ndef split_data(data, validation_days=7):\n    \"\"\"\n    Make a simple data splitting, holding the last `validation_days` for validating the model\n    \n    @param `data`: Dataframe with mandatory `t_dat` column (this will be the comparator)\n    @param `validation_days`: How many days to hold for validating the model (out of 30 days)\n    @return: 2 dataframes of type (first_day to x_day) and (x_day + 1 to last_day)\n    \"\"\"\n    \n    last_days_timestamp = data['t_dat'].max() - np.timedelta64(validation_days, 'D')\n    train = data[data['t_dat'] <  last_days_timestamp]\n    val   = data[data['t_dat'] >= last_days_timestamp]\n    \n    return train, val","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:24.072166Z","iopub.execute_input":"2022-03-29T07:44:24.073025Z","iopub.status.idle":"2022-03-29T07:44:24.083442Z","shell.execute_reply.started":"2022-03-29T07:44:24.072868Z","shell.execute_reply":"2022-03-29T07:44:24.082725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_users_items_matrices(data, validation_days=7):\n    \"\"\"\n    Split the data into train and validation and compute a CSR matrix from each one of them\n    \n    @param `data`: Dataframe with mandatory `t_dat`, `user_id` and `item_id` columns\n    @param `validation_days`: How many days to hold for validating the model (out of 30 days)\n    @return: Computed train and validation CSR matrices\n    \"\"\"\n    \n    # First, let's split the data by `train` and `validations` sets\n    train_data, val_data = split_data(data, validation_days=validation_days)\n    \n    # Then compute the CSR matrices\n    train_csr = users_items_csr(train_data)\n    val_csr   = users_items_csr(val_data)\n    \n    # Return the matrices\n    return train_csr, val_csr","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:24.084737Z","iopub.execute_input":"2022-03-29T07:44:24.085456Z","iopub.status.idle":"2022-03-29T07:44:24.102824Z","shell.execute_reply.started":"2022-03-29T07:44:24.085419Z","shell.execute_reply":"2022-03-29T07:44:24.102104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_csr, X_val_csr = get_users_items_matrices(transactions)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:24.103889Z","iopub.execute_input":"2022-03-29T07:44:24.104476Z","iopub.status.idle":"2022-03-29T07:44:24.217132Z","shell.execute_reply.started":"2022-03-29T07:44:24.104439Z","shell.execute_reply":"2022-03-29T07:44:24.216345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### In order to find the users x items matrix, the following problem is solved:\n\n## ![image.png](attachment:09bb7f67-eaa2-431a-91bd-9579267eefbc.png)","metadata":{},"attachments":{"09bb7f67-eaa2-431a-91bd-9579267eefbc.png":{"image/png":"iVBORw0KGgoAAAANSUhEUgAABOQAAACLCAYAAADBLEaAAAAgAElEQVR4nOzdf1Rb95ng/3eSaZVjd5QmWflkWlQnNkka5HiNnNQrj6cW441FPTWuJwxNV8XtEOyW4tQwmsQMiU3JD4qbo4IbE1qbsq0JpzFD17XoupZT13LrY43rWmQdi2wS4YlHZOzhpiHRfs36duvm+wcCrkACCUkg8PM6x8cILpeLkO793OfzfJ7nhg8//PBDhBBCCCGEEEIIIYQQ0+LGmT4AIYQQQgghhBBCCCGuJxKQE0IIIYQQQgghhBBiGklATgghhBBCCCGEEEKIaSQBOSGEEEIIIYQQQgghppEE5IQQQgghhBBCCCGEmEYSkBNCCCGEEEIIIYQQYhpJQE4IIYQQQgghhBBCiGkkATkhhBBCCCGEEEIIIaaRBOSEEEIIIYQQQgghhJhGEpATQgghhBBCCCGEEGIaSUBOCCGEEEIIIYQQQohpJAE5IYQQQgghhBBCCCGmkQTkhBBCCCGEEEIIIYSYRhKQE0IIIYQQQgghhBBiGklATgghhBBCCCGEEEKIaSQBOSGEEEIIIYQQQgghppEE5IQQQgghhBBCCCGEmEYSkBNCCCGEEEIIIYQQYhpJQE4IIYQQQgghhBBCiGkkATkhhBBCCCGEEEIIIaaRBOSEEEIIIYQQQgghhJhGEpATQgghhBBCCCGEEGIaSUBOCCGEEEIIIYQQQohpJAE5IYQQQgghhBBCCCGmkQTkhBBCCCGEEEIIIYSYRhKQE0IIIYQQQgghhBBiGklATgghhBBCCCGEEEKIaSQBOSGEEEIIIYQQQgghptGfzfQBCCGEEEKIxCjeDvYfPo7vQi8Br4pxVQ6mVV+g+BEb2fqZPjohhBBCCDGZGz788MMPZ/oghBBCCCFEHAYDdHzrUao6gzE2MFPc2kyt1TCthyWEEEIIIRIjATkhhBBCiFkhhGfHQ5ScNFH81S9StMaCUddPwOelY/8eOrxKeDsr9cdbKVo4owcrhBBCCCEmIAE5IYQQQohZQO1uovAfgmxtq8eWNfarIXwvlFDY4APA8HgXp8tM036MQgghhBAiPtLUQQghhBAi46n43PtZ/LgjSjAOQI/50a0Uhx8pZ/zEWtQqhBBCCDHnhAK4f1xH1SY7hXmLWJFvx/5EnWYFQeaRDDkhhBBCiIznw/mwj/yflhI77y2EZ8cyStoBg4Ou0+UTbCuEEEIIMTeETjopczQRK/ZmWFNL655iTLrpPa7JSIacEEIIIUTGM7G5baJgHIAe/YLhzQ1Is9XZ409/+hMyRy6EEEJMQU8LJZs6UP6Lg8b2o5x+9VVO/LyVxkobxvAmyrEaSr7nRU1gtx9++CF/+tOf0nHEIyQgJ4QQQgiR8XTo58Wx2R+H/jPen82CibcUGeL999/n5ZdfRlUTuU0QYm5TvB04d5Rht69lxaLVFG4qo2avm0Bopo9MCJFZFFwv7kP3eDNdu8spsGRj0Osx5lgpeKyZE79upCDceF5pPoAngdWrqqry8ssv8/7776fn0JGAnBBCCCHEHBGk700AIwV/ZSLDVmWIKAYHB/nud7/LjTfeiE4nfzEhGAzQ8cRqVtiraGp34/UGUAjiO+mmrb6MtcsKqUnkjloIMbddPM6BS5uoLjFHH/dkFbD1m5bwAxfnA/HvWqfTceONN/Ld736XwcHBFBzseBKQE0IIIYSYCy538ws3GGzlFD0gwZ1MNzg4iNPpBKCgoIAbbrhhho9IiJkWwvNtO1VnFlNc00zXr1/l1dNH6WyupcgSTnHBR1vJdjouzuiBCiEyRNB3nOxvTFwbLjs3b6TkR3Ag/oD+DTfcQEFBAQBOpzMtQTkJyAkhhEiTIK7D/pk+CCFmnP+wa1o6ngY9P8NtsLD5GxtGaqaIzPThhx/yy1/+kp6eHrZs2cK8efGsRxZiblO723CezKO5rZXar9gwZenRG7Ix24qpb3+FzkpzeEsPThlfCCEAo20X1WsmqZp72wKGGtQbMN6aWIXdefPmsWXLFs6dO8fRo0dTXu9VAnJCCCHSIIj7iUf5lVSxEoIF/IpHn3CnNyh32Y2zwY/t8XpK75fsuEz329/+lueff55HHnmErKysmT4cITKAis+9n8WPO7BFfUvoMT+6leLwI+WMf1omOoQQGW6efvISHeH6uhjM3P2JxMdIWVlZbNq0ie9973v89re/Tfj7J/JnKd2bEEIIgYrvhUp2XnPw83WGyTcXc17oWA3LNrelZF+23adpXj+7XleGdQ62Hiuk8gUD7Y/FqHGSlCCu53fSvfppXiqU3LhM9+6779La2sqKFStYs2bNTB+OEBnCz6kzm9lcNcH5fZ6ZPDu0tQN+BenvIISIy+W3cAMG8+fIXTi1XeTl5XHixAn27t3LokWLMBhSMxaVgJwQCQp2VVC4zcXMlZN10HmhHPPkGwoxI4Jd2yl7ycSuIzZmV9hEpEug58jQBwYLpY9vpWi1iQXDUam+Q1R8vgZP+KHhsXZeedQ0+s1qP/7DLVTVdhDEQt7S2fiqMlLw1C7O5pex/c5OGtenNmgWaK/BedOTvPRtmyxVzXDXrl2js7OTU6dO8eKLL/Kxj30svm8c9NFUXIizO73HN5HZGAwXs4mJzW1mJl5Mpkc/nHhvMkyy7Vyl4nvBTmGDb+YOYV0jp/cUyBgvxdTLfnz/OhRmNnzaQvat47cJ9Xrx9wPoyTabMMzmhPiBAN7/PXRHrb/LjOmO8b9MPM9JPHzeDsBI0aapj5M+9rGP8bd/+7d87Wtf48CBA5SVlXHTTTdNcW+jZMmqEAkyrn+SpzfKJUiIqPpcOJ/1kl/nwDrFi6aYa/z43AoGi4POV9qpLrSQbdCj1w/941JgJBgHsOFB88jX9Pqh+kGWr9RSX2YAw0pMU5zZnHG3WnHU5ePdVoerL3W7DXZV8I1jK2n4VgHG5MeFIs16enpob2/HZrORm5sb/zfOM1P6LYdMxok5TIc+nlKK4aVnxvuzr9OiGDrMW2pxJHD6ELND6Mwe7HY7drsd99vRtwkctoe32YN3YHqPL+Xedo/8vnvORM93jec5mdSglyMvKRgLq9lkSS6CuWzZMj73uc9x4MABenp6ktrXMAnICZEwA7ZnGimVmJwQYwTp2FmBy+SgdLLiquL6cbmXs0oxu/aUY47yshjJngPAxpLsaIMlHdmfNsNGM9npOs5poF9TisPqpmJnR0pqH4V+10TVy3dTv6cUs/QEyHhXr17lwIEDvPPOOzz00EPxZ8eF6e4vp6HOmqajE2I2CNL3JoCRgr8ypWH5/yyhM1H+3XrkbCDEZFT8P3bSMq+c+m8lv3Jn3rx5rFmzhnfeeYe2traUdF2VJatCTMU8C47vO/A97CRawri17gStjySxcGgwhHLJj/fwAfa85CIwc+tjhYhb8GAdVR4D5e3S4VGMCvnPoj5eGiNjMkjgjOYEl7OcxXfE3pfNlD3Lb8CMbNhcjtNeRc3LlqSuE2p3E2UNH7B5V3XsYJyqEBwwYJzgORXT59y5c/zyl7/kL//yLxPLjtMwPlJL45lCKg5GGRgYSmk/Xo1lysFZlVCon6DXQ8eBFto8UjI/JaZxWdZMmbaldpe7+YUbDLZyih6Y3VeDpC0sonb3qRhldAyU7v811aum/hypoRD9fV48PzlAS7tHGmiIWUntbqHmRzoc39+axLUxUm5uLn/5l3/JL37xC77whS+wcuXKpPYnGXKZrs9DXclqFq1YS8WP/agzfTxihC63lOrK6ItHPNWVNHUn8deap8ew2ELBY40cPdJOuaSli0w34KGl3g25m8lPMh1czC3Byzq+uCpG4Enp4dRJzeM1ZkzRt6T/vVv43KysHxdJZynCYQVPgxP35antQ+1poeyfgmzaVY01ZoNOFd/enbjfm+qRznHXgrjrS1i7YhEr8p14k5/kntDVq1dxuVz09/djNpu5/fbbp7gnIwWVT1IQ7a2gtFCxw5XEjbMOvd6IyVZMbetRjtZJjaiUmK5lWTNoupbaBT0/w22wsPkbMvEHYFzv4MmoZXQUWhzbkyqPoNPrMebYKH6mlaOv1Ec/5wiRyUIenP9wirymZspzU3dvYjAYWLFiBVeuXOFnP/tZ0llyEpDLaEE6dpbQ4gmCEsBVW8Ier4TkMocO82O1VEcNlvlw7tiDLxUD/FstOHY3yoVQZDAV30tO2hQo2lIUM6Airk8mezW2GBla6htncWkeFy+NvSDV9JV6CmZr/bgIRvK/WASKi51t3sQn2vpcbP/7DhZ8dQP6oBevN9o/Dx3PllDZ8zlsOen4HWa7IK4nCinb6yGggPJmE/YfpLdA+htvvMEvf/lL5s+fz4oVK5IrBJ1VwNN1xVGDZcrBCuoOpiKXRUf2I7tojjHxKMS0u+zG2eDH9ng9pffLxN+QoYZBxVFPBi4q6pMJ0I/SLS5i1/elhqWYRQZ9NJXVEXq8nvIHUltG54YbbsBisTB//nx+/etf89ZbbyW1PwnIZbQQil/7WCE4IA2+M4uJ0t2N2KJ9qaeJsqRmqjWyCnA8LpUiRIa67GZfgx8oIu8zUjtOxC/wplfzyIIp+/p4/egt+RQDSvM+Dl1M4Bv7XFQ8XIFLCdBRbR/JSBn/r4Sq1gDWQqtkkUQROtbCcwdvxpyreXZeOI73Wnp+3rVr1/jVr35Ff38/999/P4sXL056n/o11TQ/Fn36w+1IMkt/hA7zFgflMiEoZlwQ1/M76V79NNWFclaLcKuV6qby6JOhhyuofMGXkhVWutxSHGVyMhCzwLUgrh1VBO0/pH5des4XWVlZ3H///fT39/OrX/2Ka9emPoCQgFxGM1HwlGa5QK6DTVY5EWacrAKqd0df1qEcfA5nV2qqLhjXfJGilOxJiNTyu/bhBrDnY5lldW/ETArSe0Yz6zSbO6gmSm8mzw7goc3tn2zrISEvzm0VuOKtKWrIJy9aF43rnp+OBj9F+7vo/OkJToxcv30EU9j9Vuvdd9/lzJkzAOTk5HDbbbelYK86zF+L1WnRh/PZltRk6ess5D8qec9iZgXaa3De9CQvfdsmkwxR6B7YSm2MbFZfQx0tKQrQW/5ms6yCSKM/DAygqiqqqqLEyMG58q4a3maAD/4wvceXciFl5PcdGIj+y8TznEQK4nriUX6xtIHaCYJxoYtBkklzuu2228jJGVqC8Jvf/IZ///d/n/K+JCCX4YzrG/n1qS7af3qUVzvKpYtahjKuf5KnY9RwcG2rpOm1FFwIbzWzcl3yuxEipVQvR344FFAotpqR238Rt1AvZ92ax7O8g2pi9JitxQD46zvwxDMq1Ftw/PQCFy7E+e90bYxGGte5bg9tuZvZvGrobGVc78CxCsCLkqZ6e6+99hr/8i//AgwF5D7ykY+kZsfzzJR+K8Yysm4nZTvcUYq9J870YL7UkhMzJthVwTeOraThWwUYk1jpPbfpMG+ZIED/9e1TrlkaIcdMvpwM0ub/vHWe/v5++vv7ef1S9G2CPf3hbc7zxvvTe3wpd+n1kd/3/Fv/J+om8Twno4J4aqs4kF1P41cm6MLc56LuR71JHDh85CMfGQnIvfrqq/T09Ex5XxKQmwV0d5iw5Gajl4tQBjNge6aR0qgXKR/Ob6ViptrAYrPMS4nMEvIcokkBsLHcJOE4kYBeP0c0D2d/B9XE6E3Lw+UO2jjinVvlKNTL/pF6doEpFm9PF9/J4xRssGomD4yYrEPX1g+upL5O7//7f/9vJBg3f/587rrrrpTuX3d/OQ110UtaKAd38lwqsvQXm8hPfi9CJCz0uyaqXr6b+j2lmZWUMBAYOcf5L2dIfW+difLv1hP1bKC42Pl8KsroZGNam/ROhEiDEL4XKinpW8mmZSq+qPV1vXjdLZQVOzH8jSXpJIJ77rmHBQsWAHD69Gn+8IeppSz+WZLHIYQYNs+C4/sOfA87GVcauttJ2bdNvPKMNak3f/Y9liS+W4hUC+E91jH0Yc5yFsco3C9ENME3Tmmyd0wsX3ydTbvfkUveKnCfhI4uDw7b3OloGTqzB/u2ofRHx08vkJ1BmXqmr7Ri0keGfo0LzUCcS4cT9N57743MnH/6058eGbynkvGRWhrPFFJxcGw+3FCW/pLFnZQm09xDvxjTKuhP5iCFSJDa3URZwwds3lUdOxinKgQHDBine/zxthu73QmAbfdpmtdnyNl7YRG1u09RuM01LjtWOVhBpWkxnSXJTO7rWWyywMnJtxRi+qj4XiihsMEH+Cg7NsnmudVsfSD5KeBbb70Vo9FIf38/r732Goqi8MlPfjLh/UiGnBAppMstpTpGDQelfTs7k5yp1lkcvPpqqXQ5EplhwMvxzvDHa8xSV0QkQKHnzJiGDsnXuZ9ljGQPLy86/Au8qVhOJCal0+vHZWLqb0t9kGxYX18fr732GgCf+tSn+PjHP56Gn2KkoPLJGN3YfdTtaEoyS9/Ihj2vsmtthgQdxJyn9rRQ9k9BNu2qxpoVcyt8e3fiTtNS89nKuN7Bk1HL6IDv2ZqkG74YNzbzap1tzkwgidkv2FlGWUP8ndLNf2NJyT3Lxz/+cT71qU8BcPbsWd55550p7Ucy5IRIKR3mx2qp9qynrnvs1xRc22pYubSVoqkWLr9Jh15WBYoMEfKdIpwfR8G9UmZZJEANcPag5vHG5WRfT+tVw4z3FgAuwM1Zf4iCO+QEP9f8r//1v7hy5QowNJv+0Y9+ND0/KKuAp+vO4t3cNr5uXLeTsu+Z+XWVZcrLwqMFMoVIiz4X2/++gwWVteiDXrxR57JVgsf20fTOF3npsek+wExnpOCpXZw9WULbuJOBD+fXnZiPV2OZ6hJgnR69nAxEBjEWtnK6cPp/7s0338wdd4ym554/f57PfOYzCe8nroBcqNfDEddxjnf76TnpG1p/vtCMxWQiv7CUDVZjfMvwVAXf0UMcckHRvtJwZDKE78dO9rk8uPt0WDdupfqbBWTHOEko3S4O/Y9fcOSkG99FMOZayMkxk7dmA/mW7BSeIFRCvT48hw+wJ/g52r8zOhMw9hhYaMa2agObvlmMZdx0QYig5whthw/hO+PFdxEM91jJ/9Jmtn7ZgiGRunDXVFR06Cb5nlCvd+i43/5r2p2jS2BCPW46fvIzzXNnw1qwKfHjEJMwUbq7kbOfrcA97mseqv6hiew2adAhZr9Az3AFMAN3G2WuVCSg109EftyDOdflbLvBeDcGQAHazvipXjP1gInIPKqq8vbbb488/sQnPpG6hg5R6NdU0/yYj8IXxi+/VfZWsN3USeN6mTwRGazPRcXD4W7S1faRSb/oDBTvs0rX1WhutVLdVI6vqGn8YnylhYodS+h0FshzJ0QSbrrpJm6//faRx2+//TaqqqLTJTaSm3jJ6oAXp30Fyx4qoSO0nK272zlx4QIX3nqVrn8woR5uo6ZkNQ/ZJ0iFv6YS7HbR9EQhq+9bQeG2OtqOqagA14K4HA9RWNuGuzsISgDP3grqDkeZChkM0PHEalY8/BynPvFF6jte5cRP67Hixd3eRFXJWpZ91k7T75IojDyoEPC6aNlRwtoV97HsITsVDS4Cg+HU3pCftq2rWfFwBXXt4WAcwEUf7vYa7J+vwNU3urvQmy7q7A+xuqSKlk7vyPbKmx7aau18/usdBCY9KJVQr4e2HSWsXXkfLeeibHJNJdjjpqO+isK8RaPHPZyRHH7uln2+LOK4g93uoeN4ooNghtQjnTOyCqjeHaMeULeTmh/4kKdczG5+fO7hqVczEo8TiVB6z2puEgysTFeGpRoi2OPB9UIVhVs7ohe0HvTTtm0tKxYtYtEKe9LLeRJyRzYjlUFP+uMYE4i0uHY1LbsdHBzk3/7t30Yeawfu6aHD/LVYnRYVXNvqIsaps4IaIhQK4vd6cXe20FJfQ1u0ZnaKD9feGso22Snb0YYvw5qJiDiEvDi3hYNx8TDkk2eWrOJYdA9spTZWGZ2DFdQdTEHDl5kk54YpUZUAPncbNZsLcXqjj3eCx+ooyVvEokWLWP2EOwXNQOYu7XX9X//1X0cy4hMRO0Nu0EdTqZ2mbsBaT0ONJop+kx7T+loa/tjPaocbxRu9YH3g5RK+0eAhEPXEGsLzrUIqfEbMC5XR4BbguaSANmY/6KNpcxlOr4K17gStj4S/ZiiidvfNfDA8k6J4cRY9xKlHisi785aRb9c/WERR7uQn7KD3CD03GVmghw/GHPPVPhdVxRV47yqmtnkDFnM2C3QfEDjYRGVteJCvuKjYuZLc1iI4VkNldYDsbzxN1y4LRj2oio9DDTUjAUflWBV1nRZaC8ffiKiXfbhdh/jZT9rwXBz35chn8twh3GdC8Md+gmO2vTrgxVlqp+NKAY7dtZj/kw71XT9HXt5Hh3fol1QOVlFpMiVZ5FOMZVz/JE97vJSNK7IM/hfK2J4tM9ViFlP6OD8y8FlC1lQKKqshQuoHBP19KO8GCFwKEjjpJ9ivsvLb7ZTnhmeYQn46nnfSctRDgGzKG7twWCSPKK2uqYSu9BP099H3jp9gIMDx13zo1r1Iqz07+veoCn5fgOFpMf1dFkxRXxcqAb82fzg/dfXjrin4f+vD/y+nOH7SMzTZF2Z5qpholcIUzz5qusKhMMWL81sd2A4VE+O3TC2DgbuHP+45T58CJgluT69BH03fbkrLrgcGBlCU0TFAVlbMYlipM89M6bccHN8QpcEUbiq2NZGV8Vn6Cp6G5zjQo9B7zjvmPqKY1m9qH4fw/3g736jV3DSe9OI+B0en63087VRCoRD9vQH6+gIEAgHOdvvpWbyVozXWGFm2IQJe/8hyZt0nTJgXZlgwS2/B8dMLOGb6OOYMHeYttTg863GOK6MDbkclTXdqxlqzgpwbEhW66MN3zs+p3xzBc0L7nBXR/J1of3s/ruoWPOHtgp1ltNhepXZNhp0vMoT2uv7uu+8yMDDAbbfdltA+Ygbk/C/VjL55c7OjprQa122i1OGmBVDanXR8yRrRxSn7kVaOPgKEPNQ9VELLyAvgKn1dOzmgb+TV4xb014K4n62k7Mc+wEBxjvYtouL93lAwDkM5mzeOOZKsAnY1vUXvSEqugvflpoilMLbdGyiKOls45vdZUzz0e1oXc/W11VQNd5B5ez+VxXry6k5TH7EmVY/5K/W8eC3A+mfDwx5PGy21pzhyzkjjkXYs2s5ieiule15CX7KaKk948x+58ReWRhQW9LeX0BZcSTb99E4SjAPQ5xZRmgtgQ/+m5rhDv8JZ6iX0N128UmLSBEstWNeYMRYXjvyNfT/w4CsxSbOAlDJge6aR0pN2zWt/mILrWSd/ndtIwTSMz4VIub4AruGPbcYpLTcMnjxCzzwjvLafunrNgMlQytZ7hwYJak8LZX9fNzIwgABNO9vIf6VUmkiky5su6v5nP7r3TuFq92hmRk04/nGCSYSeDkrszvANn4Xqn5sx3RFtsBfAp60fZ1vO4lSN80IfoLzbT/C9ID3d2jldEytzs6PeqOr/4m5MaHpsvuand4Bp6gxqwGiDofoGLgJ9jdgkIDeNVHw/qIl6s5oKAwMDvP766wDMnz+fG2+cnl5quvvLaag7y+pqz/gvdjup+6GF9sfMGbw82oC1shErQE8L6z9fN/r+XLcc00gwMYj7iS9T1hklf+O14/guFpM91Zq9mSrkoeXFAHCejoOuiICE7e9Msf+mlz047aOlVIqaX8U8154bMZ7ORPl36zmbV8X4s4EP57MtWDI+QK8l54bEqKghhf5LQfrPBSIDmOtWYo46zlmA8UHg8OhnvBeDIKPuqG688Ubmz5/PlStXeOONN3j//fcT30f0T/vxujQrzt9WxheIBdBlscQ2+j1ne2PkGOst5G3UfqID52/+mvrHLUNBopuM2Go6Of3zdtp//vPICGxPG3V7w/vduBJzlCuN7oHNbNUW8jOU0/76BS5cGPqXeCtqI9kPah72QN53mykfXyAOANO6TYw8DfhpOwxP7nZEBuM0+863F2v2fZbeMcsHTPZW6qtKKa1q5ofPWIjfmOM+6YUvd9IaEYwLm2em6KujR41yBF+0NF+RnHkWHN93RA90Ki4qtjXhl7WrYhZS+s6PPrgJbp7CPoxrirBZLNi2bKZI+4U1Q4MqtbuJkohgXJhORZX3TfrcU0B1ZSmOZ5qp36L5vMGC+d7Yt/D+7iOjYwWDGdPiGNte9HNK8zc1Pbg4dXVsbs3Gur4YxzO7cKzTfH6CY9fllvPivmLNeTrIB0lUv0jMzaCp4Xq+L961WiIVlMPbKYtSby1VLl26NPLxpz71KW655ZYJtk4t4yO1NMbqtNhQxvbDs+S1Nl8fMYa1rsoNTwAFcTkKKTtjofanp7nw1lFqV2m/cY5eJPRWSqtKKa1q5IeV2nsEK3lLY9/vhPxnNXWNrZg/Ldku142FRdROUEanbIc7+n1+ppNzQxx0GO63UbSlmsZvb4r4yujzNZaBgu90Ub1m9KuB/tB19awl4pZbbhnptApDndUTFd9U3cAHRK+uETmQ9PX1x9iBDt187eMFFH91fCFJQ44FS472paHi/Z/7RiLfljuzYsz86LFuLB99USlNHD+TypdNHpaJ0nnvWMwS7WE/+NdYJsh60v+FURNjdhOc4CyYnbMy/sMcZxObxmYUahgWL9cch5/+9+Stlg663FJqH4+Re9jtpGav1JMTs09/n2Yx1L1ZyRXk7+tDE97DZjGjH/BQ93Un3ijnR8MqM6bMTe2YQ/oJaidqHlwSs+ESBPF7NIGNNUti/I1UfP+jLSKLPevWdNwYBnlLM7s7HOSNxbimloaRG5bpfHEZyLp39JH7banUMm36XDxX60rrjWgoNG2R3SiMFDz+NAVRT84KrtrnZkU9OeXccc35wsTyexcAKr4XKql4u4jmtlqKcw1wUzYm7aQ0JoxTKaUwa6j0va05k+YsJzvm76viP9M2+nDVyhjlBMRcZVz/JE/HCNArBxDAiPwAACAASURBVHfyXNfsu/bIuSExwbe1I20LK5dGK+IRNs9E6e7mkXqk0vwxflOpIRcjIGfC9rXhgamRoi/mxTV7rfwx3rBCPuacybeCAP4To0Ml/fzYORg680o2aB5735zO0sjh1M543bYAbbzurUszNC8x5jiCAzM5cJzLdJjKGqi3Rv+qr6GMumPy3IvZRCU0ttBmEkJvRM7c5y29ivvZ7RxZVUvX2QucqNO8eQwFOL6UyUut5pBQL/6Tow9jz6YCF70c0m5rzhmduVYV/B4XbXud1JSsHdcB0l1fRVVDCx1uN943UzQ90eOLWJ5jXZozaTd44/rNbM4ByJ65wXp/CLkaTIcgHTsTKB4/Rb///e9HPr7llluYP3/+BFunwR02djlLo79vFRcV21rGd2DMKGPqTYYzXZXD2yl7KYv672zFlqUb2VbV3getMmKY0xeKQMT5ecKJKtXH8b2abe83kT2nnxsxXriMTqwA/bZKWmbVSik5NyQmRK9P+3yZMN05yZMwz0zxlqH1K9nGWElRYv78+RHZ79rrfrxi1pAzrm/k16ueJHSTHoN+pv4EKqrm5DA0cxzjdkCXzfKN0BKuS+O/MntyjtRrM30EIv2MFD3dyKmHo90AKPh7AqhrJMggZosQiiazwmJIJj9u/Mz9gnMtbH+7iMaWYky3Ao800jX/ED6MmP+LVYreTxP13ClG/zITz6YGzxzRzFRbWJmj2VZnwGQtwGQFtjioTcfBjj2ec8c1gYZJZoJHLGDBncBtCyYN3qWSwWCB4WdvQJaFpN9QBkVVlPJqqfThhx/ypz/9aeTxRz/6UT760Y+m94dGoVvloLnSR2HD+BYP9J2ntw9MGVvL1h9Zb3LNckwhNztr+yhytlIUsSw+MkBletCUuqXwmeiin+Oae6S83Nj141TfKQ5pHuctnaDWnJi7wmV0fA9Ha/gS5HxvEHJmy7tGzg0JUf2catc8nmTVwDD9gqFnynjr9JVbmG3GXtv/9Kc/8eGHH3LDDTfEvY/YXVYB3a2G6OGvayECvzlE2/4W2tI8oIkQnjmOPlA2YLhz9JFpvlxqkhci+Dsf3f4A58+00XJ4TDqzIZuCjUUsyR66STYPz0QoPjy+PoIXz3PqJy24w40pDPcUsOFvl5C90IT5M5aJC2ZfC+J5yc0pXweHugJjlpQYsDxSRL5lOZZV1jH7CRHwevG9dpYjP2zR1J4yYi60kf9XGyheP0MDkawCHE/9Cu+2MUtkrPU0ZHRxZSHGuspVTRrPRNnLkxszUMpROfUDHxv+UdsUR49pffGcKSfre2ERhQ1p/AHrGjm9J0a9mAQE3tQsh5pwNjWI16UZDOSsxByrfty0UOg+qV3KFe/xXIVrQJYR4zQe/s3zNaOa0NUYJUJmmo+mRYU4p/CdzocXxf99KXrtTkTtbqEuIjhlwrZOj/uwN+b3TOnnqGpEh9WZo8P8aDUOT+GY5hUGSp27Mrux1MVARL1Jq1mP9/kK+gobeXrVmLuBvl7OagJUpnuiN3GZK0IBv2YSZKKacCq+E02acaeNlf9Z6seNp+DauoKKw5NvOZZ72woWbYt3awedF8pnrImeLreU6srj4wL0hi2N7Fo/i8JUcm5ITI8vIihvs5jjm3i8dhWwkD2dg6JZTlEUVFXl5pvjvzeaMCA3jqrgfXkP+37UhuciYDBiNDBhDbTkGDCsAoZv1o6exf+UFUscrwnLPddLM+N00mN8wIrxASsFtgUED492Z8JQQONPY3QINZix2sxAAcVmHYuKmmBdPZ27izDGuwb9JiPWr5Ri/UopxdYyVjs0abZf2UVrzLbuerItNrItOdzc3YLHDQZLObsaHVgzIKvGuN6Bw+Wi6lj4E4YCGp8uuv5masQsp6CcnHyruIyZ5ffvddK/sZFOq9wwzKwgvWe0NeEmmE3tcdOmfT0szZ7Z5VAhP2c1N1Xx1xxUeMsN1rrJl7emzcmhJlpyTUiTcG1K7a2ouayardn7cU/hRnzWmGem/B/LcdqbRj5lrmzGsSqzb7Iia0SZWeDbx863i2h8yjLuPTq29MFcb1oQOBdnTbgBD4c0y1VZtZzF09JBWmQmHeZvOChvsDNyNsh10PxNy6wKUsm5ITERTbcmaQCjFQwG4lveKpISX1OHawrevRWs/ewK7LVt9N5VTG3rUV491YkjkdppCTOS93faTqAdHDkZq7qKQt8b4Q8N5eQ9KC+clFKCkenNG7+ILY5ZVaV/KKuu9JEN8QfjxjBaPqfpYgu2pYsnvWio3S72uRm6yOzLjGAcgPq7DtqGg3GYcXw/w2enhUizyEEVgJlNX7ZNPSDR20FZ3iIW5VXhTlXB8nTsM9OFejmrmQeJXYMthPtHdRF1qOKeeU2TyKW2Ey/litDjw4OVPHM8y1uHBDrLWL1oEaufcDP7SmJfb4K4nt1Om3YS2VpNbbkFBub6GzuIq300GGfY2DgLMvPH1IjCR8fLIYr+cbMme3pU3AGqOcGP79joI0NO7Jpw/n/eQ4fmselB8yQTJiG8z9tZsWgF9mZpOjYXBV37R4NxhgIad5djjmP5YuaQc0NixjTdivs5GKo7Z1i7Mq7lrQAMeHHaV7BohZ2mbjl7xGvSDLnQ71rY/ngd7otgsJTS+N+3UpAzPNROfyq+YX01jR4fFQcVQKGteg/5x6uxjH1hXDzOzw7DUJBja1xZdCJ+Qf+piL928YPx3OCoBM65ABtLkkmXuCOLJTA6u/Fnk6SADvpoedaJH1tmXWT6XGwvbxq5cbXWNVA+UfdeIea8sYMqYN0mCpJ4X/iPOsPL5DvY2VWMrSz5ha6p3qf5sQtceCzpw0qvN8/GVz+up4N9ndpPWFi+eGZnnyOW2iYwG+4/eQi/tQjLwnhff37cz4cDcZ07cRXbKL8/wYOdNcyUX7hAeZxbK11lrNg29N52/PQC5bnpO7J4BQ/WhceSYYYC6p8sxjQPfFcyu7VBcsJdB4czAA2lND5TMAuyMMfUiAJMW7ay2RLt/RwZoGLxDGfpptuYJXgxJx0GPHT8MPK1PelyPcXLgWYvCqA8vx9PoRlbhkxqp5eBgj0XKIh38+4mFj08tBjftvs0zetnx5OkdjdRObLqaBYsW49Kzg0JGdOgK+6mLgNejrQbyN8X/ySr4j1Ak3coZuN8yUNRri2tJSjmigkz5IKHq1hfNBSMM5e180p7tSYYN12MFDhfobMqnDGhtGD/myraupWRWZtQj4uaf6jCs9BG7aFWCXKkXIhev/YGx4IpO57XQbg2VM5yFk/bbISK93tlOLsNFOyuzpyLjOqnadtoQwfDxkZqH8n84bAQ6TV2UGWi/KtJZMcB2au3YlsILLThWJuaqnPp2Gem85/T1IQzmDFFrcGm4HqxLjJ7esbrx01xJlj1cuSHfmzrrQkM1rOxfnPo9Wpc58A2Z4Nxs5/6mvYmFMBAwVOOcOFvBeXt1P/MG264gRtvjG8hSjoph7dTNlIvyozj+47xk9qZaEyNKAwFbP6yNfqN4ZjSBzbz3G5aEN8SPBXfS87IjNB46scZzHyhxIIBA5ayL2KRu+m547Kb7Zol+7Nh2XpUcm5IyLhVA3E2dQkeO0BHThEbogY6ozOYv0CpxQAGC+WFlusyGHfjjTcm1NABJsiQU7ubqNzaQRAw2FtpfXz8muzpo8e8pZmj5ibWF3WgNwY58k+fp+ZNBQzZWFZb2fDYUV79q2z0U1wWKSYSwH9U8zAnD9PCOL5teAbvMfO0FWMPdm2nYq+CubIzg4qTqvj21owWU8510DwrZqfTT73sx/evIXSfMGFeeP3VdJi9dOhygJ5JN5xYj48jEcvHithwf3JDJV1OMc3Hi5M7rmnYZ2ZT6PVp68ctiVqDTT25j+fG1t2a6fpxl7s5HtFNbbLlWUOUo/tpophWayJnZh0mezMn7AkfZWw5uuvuZiHtBn20fGts3bhGntaMEdQ0rKzR6XQYkupAnQJ9Lp6rHW4kZaBg9+zJzB9bzsD6tc0xS6VENjiY+SzddAv0HBl9EGvS4eIh9jSMyfyMq36cAetT7Zx+KsmDFBkmiOv5nRGJAZm/bD06OTckxt8d0c4hvqYuqg/XjzxY/646/uWqAHdYqW4/TXWiBzmHGAwGdLrE3lnRA3Kqj5Ydw4MXE5u/FCPqPJ36XNS82I/j1yfiql0mUmjsTMQqE/G0zBiewSteOk0NNi52ULPNhZLroPnRzLnIBLsiZ6ern5muZbQhAu4O2rqOE1CzyS8sZoMte+bfyxqhM3uwb3NjeLyL0ylYWphpgp0lrH4iRPXPOynNmXjbUI+bth/t59Q7wE2g9KuY/2YzpY/ayJ7w9RIi4D5Ey4FDBOeZyFtfTLEt3R2kFrDgTkYDcn+cWl9IpfdsZO2xhLKTRNqMaYoQtSbcoI899YewbCmmb2/bSLAjrvpxA16cWytoCmTj+H5qs9pDfm3miAFLPPXjVB8dP3Bj/VpX1Poz6Xb1j5oHdy4g/gp2YnLDWfOaT+WW4yjVTjIHCU5DQ4crV64wODjIrbdO04ts0BeZmW/fFRGEzGxjyhkYitm0MdZ7WcV/RpP/ETOjd67w43OPDsqjTzoEcT1fRWhLKUV7W0ZqyMU7QSHmmvCy9eEl+4Zidj01WxMD5NyQmMjzBdY8cuNYNaAc3YdTKaa54DrsSJugwcFBrly5ktQ+oubShzwdOEcyH/IxT3IjmW5qTwslD/+ClU/XSjBuBkTOLkDB0vjenEMzePEub03SoI+mf6jCk2nFSYeDhMDw7PRkgZnUUPE3l7C2rA7PvE3U76ulOMOCcXOfH/ePPBhsm7BN+DcP4W2w89Dn6wh+ppbW9nba97dztKMe89s7WZtXQlvPROkberJtxdS3NvBF3RHqytZi/3Egxb/LeNrJH3ffVOqJhvD/VruEzMbnHkxieKiGCIWC+L1ePF1ttBxLXYl9NaQQ6Pbi7vRcH4X7e/2M5l9Em1FW8f2gho57n2RzLprMo/hmn0dqjChenC95UliNdszgmzyWx9Fe1d9eh5NSNscc1Mf+eSElgM/rpsMz9VeG0qd5HyQ4qyomNpw1P8JQQP13tsYdeFVDyaXOffKTnxz5eHBwEDUdqXhRjQlE5jpo/qcMmFyPW2Q5A8PG/An+ZoGI+kixMnpHqYTC53RPVxueKP081FCIUChEKMm/f1pcjqwfF60mXPBgHRWXHFT/1wURk16T1Y9TQyGUXh9er5uOvR34YvXRE7OKetI5Ztl6NdZZ22k3neeGIUNjPjcd9RWsrfUQ8Ta47KYqfwWLFq2m6vAsGBGOWbJrWWWafNJvwMOeZ93YKksTf50Mhgj2ePF0ufEPJvi9s5SqqgwOjv6y2ut+vKIE5FT83R3jPz1T+lxs//s6POvi6+opUi8iNR4LK03xLMEIEjijgGFlfMtbkxLC8+1w3biaJzOnbtxwkDD80FzZPH3LaENeOp73AUaKihOpyZWO7lrXZ8cu1XuEfT0G8gutEzz/Q4FT+wtesNdSXagZLOuzKXr8SWyKh5q/34778mQ/0UjBlzdhAHy1h/BeS83vEZ0Bw51J7kL1c6pd8/iRL2Cd4ns31O3B7fPSVrGa9XY7JdtqCHJLcsc34KWpuoKS/BXct2wFax+2UxbQXRfZS8E3NE18otSEC3Ztp+yklcanbIS6NQEw7bYTBB7SV2OkH+Wi5uEq06TLs9TuJmqeDVL8WPxBmtDJJqocJaxdcR/LVqyl0F5G4M9S9Mq403Bd1lxJiz4XzmddmoCvgYLKrWyYIENC1WYrDripqTiUVBBer5+ZEFhEIHI6JypT1WFvzMqMCTslj7nhjOgIHXEdVAkcbqOlvoL14XN6ybYgunFvOD9t/20Zy5YtY9nzXjItJqUGzmuygMcvP1O7m6is11H/nVIMgeOagJxm22vqmLFYEG+XB+9hJyUPFWK3l1HlCqGbPRFcEUufi+2Olplftp6qbvVpOTcMb++mpqyQtctWsPbhMqr2ulhsNkVOZCgBjr+pAEE6XnST8e2AlGBEUk32osmSasLdyE3VbC0wxjlJGcBVXUWZfTWLlixj9eftlPzzB+gzJTlmmk3luh8lIBdCiXijnCJwcfxWAAxGLmvhjzG2m7IQnh8+N5Ruf3gfNT924+0JhmetQmTixNXcEw6sDYs3wKb0cOoksNEc1/LWZAS7drK9XRkKeK3LlFuZIK4dkbPTtVumcRltXyCctZJLdiINNUa6ayl4n9+PJxWpK+nYZ8YL4TnYhLKwiA2rJjgx97RREw6cbiqKkr1wh41NZQZQXJQ97548k+gvjJgB8BLonfrRx8OQZRt90B9K/KalNzLztmj11OuU6nOt2CxWzPcMf8bGclOSdxK3Wiiva6T12aKRT9lM10Pqvkrf25q/zG0GDJpfOnishspnwfGdrVhuHdOU40HT0OzzoI+m/2bH+bsYr4pwjZELp9txJFAseHIKwYimvZPUY+tzsf3rTihrxGGL/zj0q8qpd7ZS/8jwZ5LpJB4i1D/6yJaVKdew2c5Pi2a5JoBpyy6efiTKe/hyH+fDH3ovjGYXB90H6F9rSWpZ1+233878+fMBeP311xkYGEhib/FRu5uo1GbmT+NEZaqyXyNrRE1c8yhyFYeJ5fcOBcfV7iYKN7cRGLnx1pG9rpjSqs2MnNXtK8dnzGhu4m3mxRmXVdjfpw0BGDFoJhLUnhbKvn4ci/NpihaH6D6pOZevWknOHQBBXE+sZfvBYMR+LOutWC2jwQeDbfrqP4s0GbtsfePTPDlDy9ZHutVf7GBn19TDWOk5N4QttFHb3EnDY8OfsJC3dMw1+f5S2r9TNHRd6Bkb2M48yqXIFTO6mN0DYGRp80nLSAfy+GRTUFdP8y4Hw3cGhgezZ+mS6MQNDAzw+uuvAzB//nxuv/32hPcRJSCnxxBx4fbi/J5r/AxhKEDHjhqOaF+n3YEYM4kqasTS2vP0TZrtMfR9ISV8FlG8dNSWYf/86qFZq2XLWHbfIhYt0vzLK8S+qYyahg68l5N5i0QOkGGy+khXuaq97zgTpD/mtoB6NfLmdYL6S8qltyIe97830W3v2E5hkxx331to711CV6JsP6ZVcrwBNvWNs7iYhhvYnpahgWduNbUZVJw02OXkuYORs9PxpEnHRyUUCk18EVDVqc3apKO71vXYseviEQ50grk4H3PMv7uK93/uGwqcGoqwRO3QqMNs2TD04cGddLwW7wEE0z5hscCgudQOTPJ6jCJ4TjtzX0TeZ5K97dEsTVgVX42MeIx2G40yMJuTxkzKnfTRMwCoCt7mEgo3+8M3ezro6+O85o7bnJ2FbjBAx45K3vq7erY+MN23suFmIyPHHqAv1guzz0VVcQXeVfXUl08lGOzHdyz8YVKvN5WQJkaTbbxehrDppOJ7oYa6MXXjqr8WY8nmHVksCX8Y+MERfNcYyq5r0PGFVcn9PW677TbuuuuukcfXrqU1dXmogcWzkV0UUzpRORgiNMEypNRkv46pETVJzaPAOe0y9SyMf6GDPjd1z75F0ZNFZI9t9nYxwKnwhxbT+IBb8MyR8E28ieX3Zt77UVG0U1le/L3AtRD+rhrsn9+HvrKerav0QJC3tEkTi40YCeF9vpL9d9Tz9Mbxv5s2gLEhV8Jxs5uK74d1YxrK2VKXgX1NTWhJd2q61af53ABEXNujNjHUkV1YS32ZAQxkzH1nLLp5kdn7/r5YEYoQvuYSyhqgfHiMlyBt9+fr6fzxf//v/x35+K677uK2225LeB9RAnI6TP+lOOINqxysoDC/grq9LbTsbaHuiUJWL3sU35qXaP6yZsOTTpw/9uB1t1H17Gg2R+jNQxzQzqLjZv8/++LIqDBgKSwOZ3zE4aIP70k3bS9UYV/5WUpa/VOLXPd5OKBdSoWH47+bYE+9Ho5og1bKEY54Y/12Kj5XR0RmiPtYd/RA5rUgxw+7Ij7VdvAIwVjjub7j/Cziee5gf1esxRZBXD+JXJrs/WcXvrEDrTfPRrRKjjfAFnjTC5hYvjiNN7CDPpp21OEzFNC4uzRjZvIiZ6fNOL6/K7Wz06+18NCyPfjSEnAJd9e6cJr2lHVWTsc+M5vf3YbHYGPTmglelSEvR5rDOQyrY88k6e5dTgEACvt/6Yux1fTTGbOxDD84HExwWZdCzxnNWXDdSszJ1jPRZDUYLKYUzcxpuo1Oy/L7TKBDH/G36KBs+SIW3bcC+/Mh8ncP3+wBSjBiUse3Yy2Llqyl44766JlI10ZrN3ndHbR0xjMOSEQ2ptXaa04Le348ZhwwGMS7t4zVn60gYG2kvaZokqYpMWjqOCX3etM2FLBxd4oCydcz9Xd7qGnQniutVD8z0ZJkA4ZV4Q+VJgr/61rWPlxB35djd+6L18c//vGITquXLl1KbocTiszMT30XRQX3jmWUuSY426ck+zUy89aUmz1BqQCFvohscDfPfbWQtcUHWPBP0W8qRzNsTOQtHfvODeI9HJ6EMcSuoZ1sbcFk6OZrxxV+6j6/iEV3L2P9tiPoH28ePfde7iMiN+bHJdy3aBkVl4tiTEJoa3AWsPzeTA81iIlENJRLw7J1xb2dZVvjX9I/1K3+AheON1O0eKo/Nb3nBiBiLGkqsMS4t9SR/WkzrJrg519TM2Ilnz7bhFXz2FvdhDsiUVsl1OOizv4QhT/Ssam9GcdEK3smMFri6vo6f/z+978f+dhoNKYqIAd6aylPbowMpChvumipr6Ouvo6WMwZKD3VRv86I6YFyTfBOwVVbgr3sEDrLpwi8UIY9fwXL8qsilg0A+BoKWZZXiH1rDc4Jim8brLV0nm6n3ALGe7ITiOwreJ4tYfvh+BPmg8ec1GwtZPVnKyJuMsBPU9FnWWuvom7vcFHvEL7OOqo2FbL6oZqROmEj29sfGrN9EPeOMuz5n6WwYcxN9eEKCvPtlO1w0tH9Dr5OJzVb7axduZqqsZ2/DlexeuVa7FtraOsO38pcdIePu2rMcSu4tq1m9cN2qlqHb3yCeOqrKMxbPdptZ1i3k8K8tdifaBkp5BoMaI813gBbkN4zfsCaxoYgo3XjSp0pDnglY9CL8+va2elqSlNcqyHoP4ViW0LW9XOum11UL0d+6Mew9otYJwrgaIPdd2XFPrcZDNwd/lBx+TKnXsXCbM1kST+hRCIraoCzmkGVdVVu0rO22mUMG+5P0UJ5bbfRaVh+nxn0LP7P1vGfNlgo3V1P9XpNoO02w2hQNsz8lcboN3sXvbh+4+XI8yWsfdiOvayKQyFdigP0OixfepICzYvJW7+e+/IKsW+yY394NYuWrMb+EyhuPUFrTQFT7Tmk7eaa1OstFBrNqDcsYXGmXMtmqz4X28ubNOdJAwV11RTnTHTBNGIp0LzmLwYIZJXj+HLywax58+ZFZMj9+7//e5J7jC34co2mi2IBT1amuIti+Lxtzk5z1lhEjajJOiUbyLo38jPKgJ4NzzdSHjVDV9NMKErATe120TY8oN8Yfclm8GAZn112H+tfmJmauNn3WaNcL43YqpppLNN0ub7VMO7vb1jjoPmpWJMQmmDHqpXkXA8J4XPV2IZyTzlSfJ+kEjjngtxpXpaY1nNDeJuRsaSB/AdjT6r3K32YTDHG7qoP53+9j2X3leGKa0VgGi3cwNZKbWpTB2XLV7DWbse+yT5UC/fzz+G/z0Hnz1spn/IyJk031+vs/PFv//ZvIx/feeedzJuXeOQ7xkpiIwXOV8ix7sH5Izfu7iBgxLzKQt7fFVG0zowhnOaps2yltSZI5YsuAooRc2ERm0uLsd2jhzXNWB6L/hPiF8LftQ/XHY281D40uFCHl+uFgviDY+4CVYXAm+fxu910dAdxvXiIzeviy54yrnFQuybe49JjLqzGXBjv9kZszzRjm3xDyHUksF+G1rzviWvPgBFrVT3Wqvq4tlYuabJYJpgtjBDq5awbsC9P0w3sVYJdz4XrxnXiWJUpkakgrh0VtIzUakj17PTQz/Ae9kKuI+UXQTUUIqQECPQrBF/7gOxHijCn4m75mkroUgB/oJfQ7TZs9yfyjAylxH/Q56cvGMSvs1BqNQIhgp4juN9UMa4pwpZBbcxDnkM0KUbKN0ycDRgMjs5dWxZOVBDeiHEdcBi4eJ4+BeLqq5J22Sy3A+0AAfoHIO7ois5C9YULVKfsWLQdW4tZuTRFrwdN0DTe7tJzgbGwltZADTV7PQQN2Vg3lrL5S/lYFo75Ay/MY9Mj2QReDsA9BWyuclBkNUZ/GSy0ULAQgpf3hT9hID8dSxqyCmh8JYfPvdzGz9yeofHLRR9ezNhWFdP4mAXLX5lGxjBTNboUJsmZ4JF6n8DaVGV2XseyCmg8XUBjgt9mLKyl9X9Xsb21D5O9lM3fLI67ycdEPvKRj7Bo0aKRx5cuXUJVVXQp7qardjdRWT3SQir1mfmAeuY4LdhojJbFeU0ldCVEf28ApT+I//9kU1RonlrAfWER7ReKJt8uzLyxmaL/UUbHRSPWLeVsfbQIc8xrZAD/0fCH4wJuITw/cY4Ec2OtCLnl1gXcjAH9/ElqVKbJ+PuuYjZ/dQO2sXfAOjMbnrLiftZDcKGV0vKtbN5ojn3u6/FxJDx+NVnlXDRrTUdDOdXH8b1gc8Y/GFVDCsHeAL29KjkTNjubQFrPDRA5ltwwwT1vEL+nn/zKGGMYnY4FnzAAC9AnOdZIng7zY52cfqCDfQcP4TnhJaAoBLyAJQ/r45uof9CCeez4LlF9qVo1MLuoqoqijCY43XPPPdx0U+J/9AlK++nJXl9N8/rJbpl0mL7SyNGvJDr8iUcQT20lJWcsdHaMzvTp9Pqhi6DehCXKgMNiLYAt1VR76rCXZNIN7GwzJt33QWNcbzD13CnaANtnTGlZnhjyOKk86EKx1tOZMXXjwoUwh2enc8ppfibFs9OEZ29Pgm1jKl/QQbxdvahXH8j6YAAAHPhJREFUjuOsbhsajOZU07Ulub0GDlbRcmwogDiUJWqh/nhB3N8f6nbh6YfzBypoCY8sSve/DoN+WraVUHcs/Fz/METX6fIMWbIc5MiBDsitJv+BiV+ZSp97wq9H5yVwGWwZcT7Ts9hkAbyAF38gBMle0KdMc5O1bjnJ9nMYpq0fF1936bnCiLWqlRNVk21nwFZ3FFtdvPsN0esfyWNMXwa1PhvbllpsSZ7DYtPWj0tuJjh0KTgaAEjTNXMm6B/cSnv7JoDkOzJPC+NQaYWnUr/npUuXsmDBAvr7+/n973/PH/7wh9QG5MLNSYYDuzZnOrooKrhfbgHKyRo77r7oxfWvKlePOalqH3o1m57qojTFRxBTlo364xeIa5pZE3QaO8mintzDzs7hR7FrhuqttZy4UDv1401avPddOkwlrZwoiW+vSu/Z8LkoTZMlc82dNtrbhzKP9Hdlypl7TEO5delIDADl6H5agPLJap4OeGl6/gBnfV48b4bfeFvaeT3FxxNTIucGANXPqeGSVdEavgzrcdP23mbqc2PtyERx+2mKEznWNDNYiqi2FKVwIjxSRP24VK1SmQWuXr3KO++8A8CCBQtYvHhq67En7LUxs4YCHCU/9lHU3Dqlde96az757E/9oV03rka2hb53gmV1I1R83kOANW0F0L0HXUP1EJ4uypgIvNrdQt3IUmQzjme2prRWw5Agh14Ymr3NvzOVv7kRy3ojXOxnJHclBd21sjfWU78xiKtsNRVuwLCS7ARqcOlzCygAsnqGA3LFrDT149qhCcYBES0g0+Wyi7KVY5eyT6SO9YuiRSkKaD7dOCagZuDuv5jovWIga8r1NtLLuDQPE178gO9iEGYqLKqd2TcvTlHR4uGl98Qo7CsSph3sblzOlBuTzjTNTHCymSTBN4drrsytpiG6O0xYpB4eAJ/85Ce577776O/vJxgM8t577/Hnf/7nKdp7EFfDc5ouio1URynWnyz15D6eOwysu3v86326sl9TYLSZ0JhJFtVPS0PLaFfYWOf8kUzAfm5ZmnymbebQZgblY8rQMUdGuTUbiyWzAg/jGspVpT4xgEEv+551AzbuniwL91YL5XUW+J2TRUVNQIZ3q+/xcSj8YewJshCen+wj62s/n3DEq4ZC9A+oLFhoyNzfN4VGVw2kcJXKLPD++++PZMjdd999fPKTn5zSfqLWkMsEqndPuBjlxC2NJzSgEMxZQtbcGeNOs1vQ/yfNwytxtHe+eIh9zQoG+yby03YDa6K8KYPqxkXMThso2J2O2ekQvhcqqfIA2DD+RZzfZjCij/Ptk5buWsPLlyFmPZaJBQkMz/TZTOg8dTyHg9a6gnDQxUDBNzZE3e/Uiy6rE3aRSxVV29R4sqkRzdev/n8T/F5RasakVY6FDeEsJ7+vl/grdqZWWmb2ta/dVabrpH5cmmkGu5YHc1LX7W2aaWeCLfck88oI0ntuuObKBiwS9J2T9Ho9S5cuBeD8+fP8x3/8R4r2PCYzP9eRlsx8+lxsd4SDVTEnZqcp+zUpmmOMCLip+PbWcDyneLSszJrx4xX1sh/vyzUsW7aCtQ0+PpgzwThIV5a5mD5pbygHcE1bmmcJWXFOusyWbvX+7iPh5y/2caq/24ezZxNb10b7egj/4Q7qShZx37JlfMPdf10E4yJXDZhYfB2dP/7jP/6D8+fPA7B8+XI+/vGPT2k/GRqQU/GdaAq/Kfrof2+K+3hpP1dj3KyLeOgxr9Ks1T94fHwX1ghBXN9z4sHM5i9Z41p6o/b58fclFjgxV9aydZLlgNPHT8u2itHZ6S2NKa/VEOr10LJ1vaYZyOQXweAbp4aOZ9WSOLNQ0tRdq9fPSM+dqdTgutzN8XAH4+w/97H/pSU0P1OE9ZGn6fxpO52vvELjuvEXxakXXQ7iKvssy5asp6lb8513FNB84QIXJvp3vB4rYH6qa4LtRrPjdDcn+mQMufljEzyLuiyyrQAKZ99IrO/p1Jiw/G34DHv4PIEZ6SgVpebHQABftPPKNZVgj59gPAFXTf244gcnKhws4jU62I3W3XD2SNlMsCboay2wZEzGt0itm266iRUrVjB//nyAkcF7skLH6ka7KGKl/rup7aKIquDrrMP+8OgYxxYrO39WZL8GODt8jJqAW7BrO5Vvb6J+jW4k0F68NJuhmr3+kU7QujtMI3WWDLlzbJJGk2VuWZWLAZVgdyDFXbBF2gx4qNMsW7fWpToxQEXp7qBuU+HoBMC6+MoYzZ5u9Qp9/uECEjFW9PS52F7uJf+p0hjLWfWY1m0g756hR+akJuxmkRSuGphthq/n8+fPx2KxcMMNN0xpPxm6ZDVEaCTVwk/dDiemFkf8BXZDAVwNVexnKw2PZW4kfjbQ28pp3Hh86ASstFCxI5v2Z6J0aAoF6Hj2UaoOQsHuBkonnR1V8TfbWf98OK9sYzM/d9omzZhIT6OEqVLxvVBDXffoZ24+s4eSTXtS9hOUN7wExqYdbZy4q1HozQ6cDV6M1lJqKyd/Toekp7uWdrZpKjW41MD5kQFyoPM41v2/Dt9w6DHmxr6BnXrR5VvQ33EzGPQJ1/jxu9vwGGw0rolvCkD3ZwZAARTeCipwf6znR6HvjXiPwkjR060Etm+n5XknHbm7Yrd2TxHTmmKsz1bh4RC+nmosMWtqpEtQM7O/hCzVR9MOL5bvlEdupvpp+m/rw7VVDBTs+XnUYO4w7YyuaaqtOIWGdlAeZ4OgjJS6meDheqtg4wurrqch7PXnvvvu44EHHuDEiRNcuHCBq1evcvPNU5yVAehzsbO6bTQr2dDHoR32kQzUpF3pw9s9dlLHwJI7Y5wzZ0P26+U+RlopXVJQVIVA6//f3v3HNn3feRx/Im6NxGnpuLsvQh0WDNxWxbQjpiszotSULc6AhHLNeZ2oYaIJiAZ6pOmOLGXNsq5Z1i1HuCPLXWERJctt9XxHCWsa6FHcE4ovK3W6rQa1M2jI0Q7V3aFaOlR3o9wfTkIW8sOJna+d5PWQkCLytfO143x/vD/vH3uoDBfw42eLiL/4XO83ndgWfMiJmlrCRQ1/Nkk61J14ldlakjte0d+909/LcsWdFiJHK2n8i7IRemRJ9ojQ9t09tAyYPtpzvJJN7SM9Zizi9JwJctPRwJZke5BJNK0+3r+O+2Hi64GXzz0nqP7752Cvb5RgZ9/9lJsVnx/79fenb1/CnDlXALjrtqG3sSyew5xOgCXceeuYf0TapVQ1cNtdzJmTGGy35PahWzkk855kwkcffcTFixcBWLlyJVbr+D/dWRqQM3Cs9WAc7b3Y6G5kU8EJih7dyUNr8rDPm0Nu7oAPeTxGLP4hkWA3gVMv09IaYt6uBprLR55yKMmwUPRsMx/GH6e6PUL0aCX5Zw5RtNHDg1+0YsQjdJ7p4MRJP2EclDTVUeVK5sYijP9w/2w5okef4cjXXFTcO8JDjCKeLp+AcoxxihzfM2B1uvf/ugM3nbTSzbAOVzISJxrw8lzNAXrW1FBX7iHp6dUTMl1rQA+uca6K9V38AuCswJPkRN3xN13OxVn9Bherx/iwKydoqQth2VWDK8nXOWeeHcbQlS7BjmW03+k8JxX1TVgPH8D7XC2U7mSDYwJ7WMwv4KvFlfh9UTreDFFm9o3K5Qu803ch2r6bB0Jumlrqbs4Uec/Pkf7geZS2miNsdlVgH7LsKELIr/5xaTXwojx/EmeXXApxOk0rwX3HN2PTV3Gq39qU9pnPfIaVK1fyxhtv8O677/KHP/xh3L1muBqkcUBmPgDRMIFoeNiHpIcD6zCf00mR/TqwpYNvB8t9YDgqaDqYWGQO/qnvDfVT+eViivb7aPizG+8QwRNRxnujnc0iF9r6v65/eDmd32iieUeW/h5lgEFl6wBECQeiTPjRYP6c5DacNNPqDfKcTmj3A4eo/6dl1H3dwa2xEIH2I9T/Rxz3s69Rs3KUyMKlMJ1RYO0KFo9jUvcts2f3JwQYw7Qa/cu/ySGxyWxuzfgbOrDCahxVA582+l/v7Nm3DLlJMu9JJvz+97+nuztxY+FwOLj11vFHR7O0ZBVy11TRvNd5I/AQDdO2bzdb1z/A0qV3sXDhwhv/7lrK0qUPULh1NwcvWqnoeI1WBePSZ5YNz4E3ePu1Zuq2uXFYIfBCNbu3bmL3vmME41Y8e310dbYmGYwDsOH6RhHWG79gIu+P1IFqgvohjFdPG7X9vRrMNfxJMAfD4aHhFR/uWDWblj9A5ank9nDCe3CNa1Ws7+I3sV8lWzckd/N7LU4sFiXcHSJ6bfTNb3I1RvRCZEylGpFTL+HFjseVfPamsWBJ//Gtu+f9EbZ8n8ibvV/OX8KiUW7co+27WbWimProQ+xrrsE9kcE4AHJxbizDAEJtgf5VdtPMdfBQsQWw4NzWwMljdbiGOk7c7aKi0DrgnBIhOlw7hNgFQr2l0kP1EpJxGFC+PpmniQ7stem8J5VPRpDAT6KALekWDzJ5zZgxg1WrVmG1Wnnrrbc4f368cwbjBP+1+sYURTMZtw/Tk3mSZL/mOPAccCeuO+c78VS38osjN8p8rXd7sACWPDdV3tdoGNx+pC8Yv9LGonHcaGczm7MKhwHGHUVUtf6C1h3ZUokiI4mfPUD1oMQAcxjcPurqcMJkmlZvKW7AV+3BlWch2LSD/OWF7Kg5RmRBKb7/bKZstGAcEP31aQKA7Z5FWZNAMrHCN66XXcumVf+43/72t5w/f56lS5eyevXqlJ4rSzPkoG9c93+tDeBtPEjLSf/NpXt95ttxu9wU/F0Bzun0STBZ7iIn7kon7tE3TYq1uIGTxQ0k+nY9SmTYg7udsou+Yb6XAb2r02PNbUoP2/AlI31mWijY6KGyvQXvjzooWeMZJRg2QdO1Uu3BdflGTwIWl7Ihiey4+OUQwVMtbPqWF1bWcPKILenSmcgZL21tLdT7QlDcxNvPDziVJjlltXb9QoaarXrDgCmrd9jZABwCIv8TZXB2fL9YlL5YtTFqRk6IYz9qI4qBp9Bp2sVAjmMzT69tZHf7QToCHmwOMy/lDVzPv8HF50fbzop7/0nc+4GeNnZ4IsMO/LlRSgglX1A4Lh1uapZ8NUKw51bsd0ymc/bAleASVtydwjMFTnMkCqwt7R+MIlPbwoULyc/PJxwO09nZyapVq7jllqEzAoYTOb6HHf9s+rJHQv4w559JlP1qWVvHybV1Q35vtMz6vmC8beU4FhgXuGhttSd+zueGPublfmEnra2bATAWjPUHpCbn7hJau0pSeo5k9t+6tpXWewFysU6xoKbpetrYU9Zo/iIoAAVJVg5Mtmn1udi31NC0ZbyP77ufMnDap8m144CqAccDedMkCAkff/wxXV1dANx///3cdltqtbRZHJBLyJnrwPOsA8+zEI9FiVwIE+2dTphrsWH5q1xy09nEVsx3uZvXYyWUpnBzY54oJ761IzOr0wDYsSaRJZj7V71ZdN0hLlxhlAufiZmulWoPrlhoQE+CR11JZSml0nTZstJNUc8x6n3g+Pyiic9YybGzeofBoaYotIcIVzuHfo39mUUW3OtGyfyN9vBO74lxzl+bGegwcD1Wga29nsajfkodrqzO+Il2v07s66XDfKbiBAO9pdJGGau/oDyB1A1qljwvQts3G+GJOuwZ3a8xigfp7O21aexYjX3cH40Y/qONRLFT9XhRdvbbkrSbOXMm69ev5+TJk3R1ddHT08PChQuTfnz8NwOnKJrP9jlj6OP6FMl+HU1imIuB024ldilCznxL8guNs604HCNfkeTMteGYxKXryex/7iIHjnQt+k5n8dDNZetmWmzBSOYPfdpNq++7nypg2TRZaIu82dFbNeBkg2O6hOPg0qVLnDlzBqvVyvr165k5M7Wx21lbsjqUnFwDa54DhyPxzzZPwbhJ72qIQ9/r5Cvf90yKsrDI8ed45mimzoAkfxLs9xH8aZRNxjBdK34lTOhCLInJpamvit2YZDi2g/z4my5H6T4TAGyszht02TDslNXjVOWB4ajh+GhTWAdNWYUcHOtKEwGJaAeBcwwp9GYis8hwlFAw2nTha/ExTpVNn5w8DxWbDPC9RMelDO1EEuLnDvFc4CvUben7fMQIHffS9pveT3w8SKcv8Qdh316Aqcl+U1aEcN/QGOopXlFJ5Gs12dOCYDhXQrT52gglegkTD3bijQLYKV3nGH9J16UOXvKBsa0CzzS5aJeE22+/nfz8fM6fP88vf/lLrl+/ntwDrwY59O0bUxQzwT5/6PPwkNmv7021+Zx9w1wsxAL1HAgOk9EuMuHiBF/IUNl6n7yRh8v1m27T6vvup9YuwzYl4xNxIn4vXn+k914jQqA9kXxhbNpMQdZnQKbH9evXOXPmDOFwmI0bN45pYW04WZ8hJ1PYZT+1/9jJkidqhu75lIUshQ10FTZkejfSKtnpWtH23azf2bs6n1fB8X8rG2bsN2lYFRswydBZgCPpg3wKTZfjYd5pBwwn9juTfEigg4PdBgUHN4wvoLzYQ1V5B8X7ghz8uR939aBeUlf8eH8cAsPB5qfcWR60zsVZ9h2KTu6g/qcBNlSmELCYIFF/LfX/vYSde139F5MR324K/8EPdJLT1cDiUwdpjAJ5FVQ9kt3v+ORhw1np4EhdmFsLS6nZW5L8wJmMieCtKKTSD2zMoat+MacPNhIF7OVVKQTS4gR+Wo/f8NC8Pfv+RmRizZw5k0ceeYSuri5eeeUVHnzwwf4JcyOaZafs3y9SNvqWJpsi2a+jymWOxYD/yyXnDs8Y+iWLpFsO9l0+Lu7K9H6MbrpNq4+8m+jH7XBk6aTpFMUDByje2kgUg+ixLko+bKHeT2Lo4vbp0ws3Go3y+uuvs2zZMgoLC1POjgMF5CST5jqpet6Z6b2Y9pKbrhUleGpAqUx3PQfai2jaOPRF6cAeXONaFeu50T/Oke9IvidBKk2XzwU5BvDAEqxJHVt7y87ml7EhiUavQ8vBvmsfzf9bztYX9/CMzcf3i3vLYK6G8X53Dy04KKmtG2XMepaY6+Lp6iLW76ynpchHSZZl/xj3V1E36JATi/V+qrc8hP1SI7v3+WG+m6b9ZTdPapVxysG2rZWubZnej7GIEeudteJZayfStJt6P1iKm9i3K4WG5+daqH8BPAcrcKqH0rT02c9+Fo/Hw9NPP43f78ftTldn3kwYnP3aScW/NFM2SRZZk2ehqKmLokzvhsikMd2m1cc4FzjBkFU2Y5DNvRjjV2OJ+8CVO3HmtlH9xCGihiO1oYtZ3l9zKH6/n7fffpva2lrmzUvPyU4BOZFpzuaswvGTWsKziyitfpqSIVNXDBzFFTiD9fh7yxGDl4cvSQm/1zeH0MWyO8fRP+7dvv5xNlbfk/xKdCpNl/vKblyOxcmt8vSWndn3FqTQSwrAgrPaR1e+l4NHyyk8lIMxB7gGlrsraP2FexJkE91grP0+TeWbKP6hF1ezO7savA4RaLU9UkfNr8o58OJWlvvtuB9tomaLi2mwmCsjsuF5voZ3njxAS+ly/HluNjfV4HFZU1gFjuD9YS2U+6haow/YdDVjxgy+9KUvcfbsWY4dO8aqVauYO3eyNg+bjNmvIjLhpsm0+ui5IB/Ns2O5FuC0D1hbiiuFxehs7sWYu6aE5m0Rql+optBtxZlfQ+sTntSO+ZOsv+bly5c5duwYGzduZM2aNWl7XgXkRKa5ZKdr5TrKaD6dKJgJ7lvOkfnDldkMWBVbuZq8cSwe5K6p4eII086GM2LT5Sthgr8D6z1Wcm8KzES5EAwBDlYkGQAMnWjBb7hoWJOeywzD4abK4aYqLc+WSTnYd9VQ9XAh1T9z0PxIVoXkbjbLhmf/STz7M70jkm1yFnto6PCQriYFkZ9VUz+7AV8qGXYyJcyaNYutW7fy5JNP8vLLL1NaWpqWshfzTcbsVxGZaNNhWn3sVDXLS1uw7T1OHQfxGkU0VBZl10J0WllwVjbzRmWm9yMzrl27hs/n4+OPP2bbtm3MmpW+MppJNdRBRLJAPEjgbAEPDbckcilAR++qmLNoDOWmKRuh6XLMT3VBPsUP5/PlnW1EBj80FuKtdsBYgS2ZVacrJ2ipC2Ep3oxryqfhj4eNkhYfy35eTmN3psZMiGSPeHcj5T9fRtOzU/liXcZiwYIF7Nmzh+PHj3P27NlM746ISJpMj2n1ubYVuO8wyD1TT/WvVtPsTaF0U7Le2bNnefXVV3nqqafSVqraRwE5kanIsOACIE78WjqfOIL/B0fgqar+/kfxHj9en5/I1d4t3uzAD2B42LzGzFvP3qbL83ubLg/sb5djYLEmAojRE6/SfXnQQ/smQa21JZVWHzn1El7seFKZtDgR+ifq2rFkumxolp2y/Tt5/9eh0bcVmeJCv36fnepLKIPcd999bN++ncOHD/PBBx9kendERMZhmk6rn+uirqOL1uZmfPvLcM6fqi9UPvjgAw4fPsz27du577770v78KlkVmYrmLWLZYjhx7gRvhWIUzU1Dv6J4iJZvtRB3f+fGgIF4gAMPb6UxCkb0OF1bYrT8wA8YFO0tMblp+QhNl3NslLR2UUKQ+vVBFg3qRRAKJiZBub+YTClZiBOH/RiuhpT6REyEWKgz0Xsvb9lNrzEj5jmp2ZLpnRDJPPuWsZfgy9Q3Y8YM1q1bx+XLl2lubqa8vJxPfepTmd4tEZGkaVq9TGV//OMfaW5uJi8vj3Xr1jFjxoy0/wxlyIlMSTY8z1ZgB1qqnsF7IQ1lgzk2PN+ro+TeAcG9eJzEkEoHOx/Ipe3blRyKGji+0cT3C7OvMCty1Ev08Q3YgNilIKEoQIhAWwgWl+G+f/TAZfxCDxRVUbHNmVWlZ7H3vDzzjBcMBxV7PVO2ga6IyFQyc+ZMHnvsMZYvX84nn3yS6d0RERmTwdPqKzWtXqaQTz75hOXLl/PYY49NWK/XGdevX78+Ic8sIpkXjxLynyZwKUaO4WDDRlsKEwKHFjlVS/V3D+G/ZGBdU4Bn6048SY7ciZ6qZfeLIVhTReuWiQ0hhX9WjXd2CVUuC5w7ROH6WkKbmjnpPM2m0hCbvc2U3TsZJx/GCB09RiAaJ3e+g9VOG4ay5kVERERkol0N0fLNcg4cDxOdb8f9t6WUaFq9SNIUkBOR6Sce4tCOx/GyiHlYKNhVgTtPVw4iIiIiIiJiDgXkRERERERERERETKQeciIiIiIiIiIiIiZSQE5ERERERERERMRECsiJiIiIiIiIiIiYSAE5EREREREREREREykgJyIiIiIiIiIiYiIF5EREREREREREREykgJyIiIiIiIiIiIiJFJATERERERERERExkQJyIiIiIiIiIiIiJlJATkRERERERERExEQKyImIiIiIiIiIiJhIATkRERERERERERETKSAnIiIiIiIiIiJiIgXkRERERERERERETKSAnIiIiIiIiIiIiIkUkBMRERERERERETGRAnIiIiIiIiIiIiImUkBORERERERERETERArIiYiIiIiIiIiImEgBORERERERERERERMpICciIiIiIiIiImIiBeRERERERERERERMpICciIiIiIiIiIiIiRSQExERERERERERMZECciIiIiIiIiIiIiZSQE5ERERERERERMRECsiJiIiIiIiIiIiYSAE5EREREREREREREykgJyIiIiIiIiIiYiIF5EREREREREREREykgJyIiIiIiIiIiIiJFJATERERERERERExkQJyIiIiIiIiIiIiJlJATkRERERERERExEQKyImIiIiIiIiIiJhIATkRERERERERERETKSAnIiIiIiIiIiJiIgXkRERERERERERETKSAnIiIiIiIiIiIiIkUkBMRERERERERETGRAnIiIiIiIiIiIiIm+n8EoAELlCzLJwAAAABJRU5ErkJggg=="}}},{"cell_type":"code","source":"# Some of the most important hyperparameters for the ALS model are:\n# `factors`: The number of latent factors to use for the underlying model\n# `regularization` (lambda from the formula above): Strength of regularization\n# `iterations`: The number of ALS iterations to use when fitting data","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:24.218359Z","iopub.execute_input":"2022-03-29T07:44:24.21878Z","iopub.status.idle":"2022-03-29T07:44:24.222188Z","shell.execute_reply.started":"2022-03-29T07:44:24.218741Z","shell.execute_reply":"2022-03-29T07:44:24.221052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def validate_ALS(X_train_csr,\n                 X_val_csr,\n                 factors=64,\n                 iterations=15,\n                 regularization=0.01,\n                 show_progress=True,\n                 show_params=True):\n    \"\"\"\n    Define an AlternatingLeastSquares model with the given hyperparameters\n    Then, fit the model and validate the model, computing the MAP@K metric\n    MAP@K details:https://towardsdatascience.com/breaking-down-mean-average-precision-map-ae462f623a52\n    \n    @param `X_train_csr`: CSR matrix used for fitting the model\n    @param `X_val_csr`: CSR matrix used for validating the model\n    @param `factors`: The number of latent factors to compute\n    @param `iterations`: The number of ALS iterations to use when fitting data\n    @param `regularization`: The regularization factor to use\n    @param `show_progress`: Print or suppress the progress of fitting/validating the model\n    @param `show_params`: Print or suppress the current hyperparameters of the model\n    @return: The computed MAP@K metric\n    \"\"\"\n\n    # Define the model with the given hyperparameters\n    als = AlternatingLeastSquares(factors=factors,\n                                  iterations=iterations,\n                                  regularization=regularization,\n                                  use_gpu=True,\n                                  random_state=RANDOM_STATE)\n    \n    # In the previous versions (version < 0.5.0), ALS required a\n    # CSR/COO matrix of type (items x users) for training the model\n    if implicit.__version__ < '0.5.0':\n        X_train_csr = X_train_csr.T\n    \n    # Fit the model with the training data\n    als.fit(X_train_csr, show_progress=show_progress)\n    \n    # A common metric for validating the model is `MAP@K`\n    # We can also try the `Cosine similarity` metric\n    K = 10 # Number of items to test on\n    map_res = mean_average_precision_at_k(als,\n                                          X_train_csr,\n                                          X_val_csr,\n                                          K,\n                                          show_progress=show_progress)\n    \n    if show_params:\n        print(f'Factors: {factors}\\t |             \\\n              Iterations: {iterations}\\t |         \\\n              Regularization: {regularization}\\t | \\\n              MAP@{K}: {map_res}')\n        \n    return map_res","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:24.22339Z","iopub.execute_input":"2022-03-29T07:44:24.224047Z","iopub.status.idle":"2022-03-29T07:44:24.235134Z","shell.execute_reply.started":"2022-03-29T07:44:24.224005Z","shell.execute_reply":"2022-03-29T07:44:24.234365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Learning curves","metadata":{}},{"cell_type":"code","source":"def learning_curves(model, data, show_progress=False):\n    \"\"\"\n    Fit and validate the model using different proportions\n    of the dataset for training and validating respectively\n    \n    @param `model`: ALS model\n    @param `data`: Dataframe with mandatory `t_dat`, `user_id` and `item_id` columns\n    @param `show_progress`:\n    @return: List of MAP@K (floats) results for each values of `val_days`\n    \"\"\"\n    \n    mapk = []\n    for val_days in range(21, 6, -1):\n        X_train_csr, X_val_csr = get_users_items_matrices(data, validation_days=val_days)\n        model.fit(X_train_csr, show_progress=show_progress)\n        curr_mapk = mean_average_precision_at_k(als, X_train_csr, X_val_csr, show_progress=show_progress)\n        mapk.append(curr_mapk)\n        val_set_frac = round((val_days / 30) * 100, 2)\n        print(f'Train: {(100 - val_set_frac):.2f}% | Validation: {val_set_frac:.2f}% | MAP@K: {curr_mapk:.5f}')\n    \n    return mapk","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:24.236621Z","iopub.execute_input":"2022-03-29T07:44:24.237142Z","iopub.status.idle":"2022-03-29T07:44:24.24833Z","shell.execute_reply.started":"2022-03-29T07:44:24.237103Z","shell.execute_reply":"2022-03-29T07:44:24.247649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mapk = learning_curves(als, transactions)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:24.251048Z","iopub.execute_input":"2022-03-29T07:44:24.251255Z","iopub.status.idle":"2022-03-29T07:44:40.895449Z","shell.execute_reply.started":"2022-03-29T07:44:24.251231Z","shell.execute_reply":"2022-03-29T07:44:40.894773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the results\nplt.figure(figsize=(10, 8))\nplt.xticks(range(7, 22))\nplt.plot(range(7, 22), mapk, \"b-\", linewidth=4)\nplt.xlabel(\"Training set days\", fontsize=14)\nplt.ylabel(\"MAP@K\", fontsize=14)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:40.897636Z","iopub.execute_input":"2022-03-29T07:44:40.898151Z","iopub.status.idle":"2022-03-29T07:44:41.210531Z","shell.execute_reply.started":"2022-03-29T07:44:40.89811Z","shell.execute_reply":"2022-03-29T07:44:41.209863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmap_res = validate_ALS(X_train_csr, X_val_csr)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:41.211713Z","iopub.execute_input":"2022-03-29T07:44:41.211964Z","iopub.status.idle":"2022-03-29T07:44:42.219409Z","shell.execute_reply.started":"2022-03-29T07:44:41.211928Z","shell.execute_reply":"2022-03-29T07:44:42.218706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tuning hyperparameters","metadata":{}},{"cell_type":"code","source":"# I will use this class from `sklearn` to generate a `params_search_space`\n# used for computing a manual grid search\nfrom sklearn.model_selection import ParameterGrid\nsearch_space = {\n    'factors': [50, 75, 100, 125, 150, 175, 225, 250],\n    'iterations': [5, 10, 15, 20],\n    'regularization': [0.005, 0.01, 0.05]\n}\nparam_grid = list(ParameterGrid(search_space))","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:42.22074Z","iopub.execute_input":"2022-03-29T07:44:42.221503Z","iopub.status.idle":"2022-03-29T07:44:42.227933Z","shell.execute_reply.started":"2022-03-29T07:44:42.221446Z","shell.execute_reply":"2022-03-29T07:44:42.227177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# len(param_grid) = len(factors) * len(iterations) * len(regularization)\nlen(param_grid)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:42.229283Z","iopub.execute_input":"2022-03-29T07:44:42.229981Z","iopub.status.idle":"2022-03-29T07:44:42.239545Z","shell.execute_reply.started":"2022-03-29T07:44:42.229939Z","shell.execute_reply":"2022-03-29T07:44:42.238775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param_grid[0]","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:42.240879Z","iopub.execute_input":"2022-03-29T07:44:42.241308Z","iopub.status.idle":"2022-03-29T07:44:42.252405Z","shell.execute_reply.started":"2022-03-29T07:44:42.24127Z","shell.execute_reply":"2022-03-29T07:44:42.251559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param_grid[len(param_grid) - 1]","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:42.253668Z","iopub.execute_input":"2022-03-29T07:44:42.253969Z","iopub.status.idle":"2022-03-29T07:44:42.262206Z","shell.execute_reply.started":"2022-03-29T07:44:42.253933Z","shell.execute_reply":"2022-03-29T07:44:42.261363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Perfect, now we can find the best hyperparameters for our ALS model\n# I will not run this in the final version of the notebook, because it takes quite a while\n\ncurr_map = -1\nbest_map = -1\nbest_params = {}\nfor params in param_grid:\n    # Compute the current MAP value\n#     curr_map = validate_ALS(X_train_csr,\n#                             X_val_csr,\n#                             factors=params['factors'],\n#                             iterations=params['iterations'],\n#                             regularization=params['regularization'])\n    \n    # Update `best_*` if found better score\n    if curr_map > best_map:\n        best_map    = curr_map\n        best_params = params \n        print(f'Best score until now: {best_map} | Updating: {best_params}')","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:42.26343Z","iopub.execute_input":"2022-03-29T07:44:42.26412Z","iopub.status.idle":"2022-03-29T07:44:42.273Z","shell.execute_reply.started":"2022-03-29T07:44:42.26408Z","shell.execute_reply":"2022-03-29T07:44:42.272285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# After doing a little research, I've found that a good set of hyperparameters\n# for this model & dataset are: {'factors': 300, 'iterations': 5, 'regularization': 0.01}\nbest_params = {\n    'factors': 300,\n    'iterations': 5,\n    'regularization': 0.01\n}\n\n# We can train the model with those parameters\nbest_model = AlternatingLeastSquares(factors=best_params['factors'],\n                                     iterations=best_params['iterations'],\n                                     regularization=best_params['regularization'],\n                                     use_gpu=True,\n                                     random_state=RANDOM_STATE)\n    \n# Now, we can train the model on the entire training data (last month of transactions)\nX_train_full_csr = users_items_csr(transactions)\nbest_model.fit(X_train_full_csr)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:42.274529Z","iopub.execute_input":"2022-03-29T07:44:42.27544Z","iopub.status.idle":"2022-03-29T07:44:44.266464Z","shell.execute_reply.started":"2022-03-29T07:44:42.275409Z","shell.execute_reply":"2022-03-29T07:44:44.265657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generate submission file","metadata":{}},{"cell_type":"code","source":"# This version of code works well if the `implicit` package is v0.5.0 or greater\n# It's much much faster when passing an array of user_ids instead of a single user_id,\n# because it allows multi-thread processing: https://github.com/benfred/implicit/pull/520\ndef recommend(model, csr_train, batch_size=1000):\n    \"\"\"\n    Recommend items for all customers and generate the submission df\n    \n    @param `model`: Fitted ALS model used for recommendations\n    @param `csr_train`: Matrix used for training the model\n    @param `batch_size`: Size of each batch used for predictions\n    \"\"\"\n    \n    # List of tuples of type (customer_id, [recommended_article_ids])\n    predictions = []\n    \n    # List starting from 0 to len(users_list) with step 1\n    list_ids = np.arange(len(users_list))\n\n    # For each batch\n    for idx in range(0, len(list_ids), batch_size):\n        # Select the current batch of users\n        batch = list_ids[idx : (idx + batch_size)]\n\n        # Recommend to those specific users in the current batch\n        ids, scores = model.recommend(batch, csr_train[batch])\n        \n        # For each user in the batch\n        for curr_user_idx, user_id in enumerate(batch):\n            # Convert from `user_id` to the original `customer_id` format,\n            # using the 1:1 mapping created earlier\n            customer_id = users_list[user_id]\n            \n            # Get the recommended items for this specific user\n            user_items = ids[curr_user_idx]\n            \n            # Convert from `item_id` to the original `article_id` format,\n            # creating a list of `article_ids`\n            article_ids = [items_list[item_id] for item_id in user_items]\n            \n            # Append the tuple (user, recommended_items)\n            predictions.append((customer_id, '0' + ' '.join(article_ids)))\n\n    # Generate the pd df\n    return pd.DataFrame(predictions, columns=['customer_id', 'prediction'])","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:44.26791Z","iopub.execute_input":"2022-03-29T07:44:44.268313Z","iopub.status.idle":"2022-03-29T07:44:44.278513Z","shell.execute_reply.started":"2022-03-29T07:44:44.268273Z","shell.execute_reply":"2022-03-29T07:44:44.277752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npred_df = recommend(best_model, X_train_full_csr)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:44:44.27976Z","iopub.execute_input":"2022-03-29T07:44:44.280127Z","iopub.status.idle":"2022-03-29T07:45:05.51409Z","shell.execute_reply.started":"2022-03-29T07:44:44.280088Z","shell.execute_reply":"2022-03-29T07:45:05.513271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df.to_csv('submission_alg2.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-29T07:45:05.515585Z","iopub.execute_input":"2022-03-29T07:45:05.515854Z","iopub.status.idle":"2022-03-29T07:45:14.991104Z","shell.execute_reply.started":"2022-03-29T07:45:05.515815Z","shell.execute_reply":"2022-03-29T07:45:14.99026Z"},"trusted":true},"execution_count":null,"outputs":[]}]}