{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os \nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom glob import glob\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.optimizers import Adam","metadata":{"execution":{"iopub.status.busy":"2023-07-04T08:42:34.901104Z","iopub.execute_input":"2023-07-04T08:42:34.901573Z","iopub.status.idle":"2023-07-04T08:42:34.908734Z","shell.execute_reply.started":"2023-07-04T08:42:34.90154Z","shell.execute_reply":"2023-07-04T08:42:34.907401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the path to your data\ndata_path = Path('dataset/')","metadata":{"execution":{"iopub.status.busy":"2023-07-04T08:42:42.194582Z","iopub.execute_input":"2023-07-04T08:42:42.195041Z","iopub.status.idle":"2023-07-04T08:42:42.201635Z","shell.execute_reply.started":"2023-07-04T08:42:42.195004Z","shell.execute_reply":"2023-07-04T08:42:42.199856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the size of the chunks you'll load from the JSONL file\nchunksize = 100_000\n\n# Set whether to save the processed chunks to disk as Parquet files\nsave = True\n\n# Load the JSONL file in chunks\nchunks = pd.read_json('train.jsonl', lines=True, chunksize=chunksize)","metadata":{"execution":{"iopub.status.busy":"2023-07-04T08:42:44.541816Z","iopub.execute_input":"2023-07-04T08:42:44.542202Z","iopub.status.idle":"2023-07-04T08:42:44.548824Z","shell.execute_reply.started":"2023-07-04T08:42:44.542173Z","shell.execute_reply":"2023-07-04T08:42:44.547336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.mkdir('train_parquet')","metadata":{"execution":{"iopub.status.busy":"2023-07-04T08:42:48.412927Z","iopub.execute_input":"2023-07-04T08:42:48.41337Z","iopub.status.idle":"2023-07-04T08:42:48.419402Z","shell.execute_reply.started":"2023-07-04T08:42:48.413325Z","shell.execute_reply":"2023-07-04T08:42:48.418449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loop over the chunks\nfor e, chunk in enumerate(tqdm(chunks, total=129)):\n    \n    # Initialize a dictionary to hold the processed events\n    event_dict = {\n        'session': [],\n        'aid': [],\n        'ts': [],\n        'type': [],\n    }\n    \n    # Loop over the sessions and events in the chunk\n    for session, events in zip(chunk['session'].tolist(), chunk['events'].tolist()):\n        \n        # Loop over the individual events\n        for event in events:\n            event_dict['session'].append(session)\n            event_dict['aid'].append(event['aid'])\n            event_dict['ts'].append(event['ts'])\n            event_dict['type'].append(event['type'])\n    \n    # save DataFrame\n    start = str(e*chunksize).zfill(9)\n    end = str(e*chunksize+chunksize).zfill(9)\n    \n    # Convert the event dictionary to a DataFrame\n    event_df = pd.DataFrame(event_dict)\n    \n    # If save is set to True, save the DataFrame to disk as a Parquet file\n    if save == True:\n        \n        # The file name includes the range of indices included in this chunk\n        event_df.to_parquet(f\"train_parquet/{start}_{end}.parquet\")","metadata":{"execution":{"iopub.status.busy":"2023-07-04T08:42:52.442606Z","iopub.execute_input":"2023-07-04T08:42:52.443088Z","iopub.status.idle":"2023-07-04T08:42:53.596475Z","shell.execute_reply.started":"2023-07-04T08:42:52.443051Z","shell.execute_reply":"2023-07-04T08:42:53.59504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Collect sorted list of parquet files in the directory","metadata":{}},{"cell_type":"code","source":"file_paths = sorted(glob('train_parquet/*'))[:20]\ndataframes = []","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Loop over each file","metadata":{}},{"cell_type":"code","source":"for file_path in tqdm(file_paths):\n    dataframes.append(pd.read_parquet(file_path))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Concatenate all the dataframes into a single dataframe","metadata":{}},{"cell_type":"code","source":"dataframe_copy = pd.concat(dataframes).reset_index(drop=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del dataframes","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe_copy","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Deep Learning Approach","metadata":{}},{"cell_type":"markdown","source":"### Exploring the Deep Learning Approach\n\nOur recommendation model involves several key steps, each serving a crucial role in the process.\n\n1. **Loading and Preprocessing Data:** The first step involves loading data from multiple Parquet files into a combined pandas DataFrame. We then create a copy of the DataFrame. After this, we map the unique article identifiers (aid) to integers, enhancing computational efficiency for the later steps.\n\n2. **Splitting the Dataset:** Next, we split the dataset into a training set and a test set based on unique sessions, reserving the last 20% of unique sessions for the test set. The training set and test set are split in such a way that all records associated with a particular session remain together in either the training set or the test set.\n\n3. **Preparing Data for Training:** In this step, we first group the training data by 'session' and 'type', creating lists of article IDs for each group. We do the same for the test data. We then convert these lists into NumPy arrays and apply padding to ensure all arrays have the same length.\n\n4. **Defining the Model:** Here, we define a Sequential model with Keras, starting with an Embedding layer that takes the unique article identifiers as input and outputs a dense 20-dimensional vector. We then add a Bidirectional LSTM layer. LSTMs (Long Short-Term Memory) are a type of Recurrent Neural Network (RNN) that are great for sequence prediction problems because they can store past information. This LSTM layer has 64 units and includes dropout and recurrent dropout for regularization. The final layer is a Dense output layer with a softmax activation function, which outputs a probability distribution over all unique article IDs.\n\n5. **Compiling and Training the Model:** We use the Adam optimizer with a learning rate of 0.01, which is a popular choice as it combines the best properties of the AdaGrad and RMSProp algorithms. The model is compiled with a categorical crossentropy loss function, which is suitable for multiclass classification problems. After compiling, we train the model on the training data.\n\n6. **Generating Recommendations:** For each session in the test set, we use the trained model to predict the next article to be visited. This prediction is based on the sequence of articles visited in the session so far. The output is a probability distribution over all articles, and we select the top 20 articles with the highest probabilities to recommend.\n\n7. **Preparing Submission:** Finally, we map the predicted article IDs back to their original values and prepare a DataFrame for submission, which includes the session, type, and predicted labels.","metadata":{}},{"cell_type":"markdown","source":"Filter out sessions that do not include all types","metadata":{}},{"cell_type":"code","source":"counts = dataframe_copy.groupby('session')['type'].nunique()\nfull_sessions = counts[counts == 3].index\ndataframe_copy = dataframe_copy[dataframe_copy['session'].isin(full_sessions)]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe_copy","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Generate an array of unique IDs equal to the number of unique article ids","metadata":{}},{"cell_type":"code","source":"unique_ids = np.arange(dataframe_copy.aid.nunique())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Shuffle the array to prevent any correlation between new labels and outcome","metadata":{}},{"cell_type":"code","source":"np.random.shuffle(unique_ids)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_ids","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe_copy","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Create a dictionary mapping each unique article id to a new unique integer ID","metadata":{}},{"cell_type":"code","source":"map_aid = {i: j for i, j in zip(dataframe_copy.aid.unique(), unique_ids)}","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Add a new column to the dataframe with the new integer IDs for each article","metadata":{}},{"cell_type":"code","source":"dataframe_copy['aid_id'] = dataframe_copy['aid'].map(map_aid)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe_copy","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Spliting","metadata":{}},{"cell_type":"markdown","source":"Test contains the last 20% of total sessions because we need to evaluate our model's ability to predict aid values for future sessions.","metadata":{}},{"cell_type":"code","source":"# Calculate the number of sessions for the test set (Last 20 % of total sessions)\nnum_sessions = dataframe_copy['session'].nunique()\ntest_sessions = int(num_sessions * 0.2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_sessions","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the session IDs for the test set\ntest_session_ids = dataframe_copy['session'].unique()[-test_sessions:]\n\n# Split the DataFrame based on the session IDs\ntrain_df = dataframe_copy[~dataframe_copy['session'].isin(test_session_ids)]\ntest_df = dataframe_copy[dataframe_copy['session'].isin(test_session_ids)]\n\n# Print the sizes of train and test sets\nprint(\"Train set size:\", len(train_df))\nprint(\"Test set size:\", len(test_df))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Group the dataframe by session, aggregating the aid_ids into lists","metadata":{}},{"cell_type":"code","source":"# Group by 'session' and 'type'\ntrain_dataframe = train_df.groupby(['session', 'type']).agg({'aid_id': lambda x: list(x)})\ntest_dataframe = test_df.groupby(['session', 'type']).agg({'aid_id': lambda x: list(x)})","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataframe","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Filter out any sessions with more than 20 aid_ids","metadata":{}},{"cell_type":"code","source":"train_dataframe = train_dataframe[train_dataframe.aid_id.map(len) <= 20]\ntest_dataframe = test_dataframe[test_dataframe.aid_id.map(len) <= 20]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1- Train","metadata":{}},{"cell_type":"markdown","source":"map function is applied on 'aid_id' column of 'test_dataframe'. \nThis map function applies the len function on each element of 'aid_id' column.\nSo, this will return a list of lengths of each element in 'aid_id'.\n max function will then find the maximum length from this list.\nSo, 'max_length_test' is the maximum length of any 'aid_id' in our dataframe.","metadata":{}},{"cell_type":"code","source":"max_length = max(map(len, train_dataframe.aid_id))\nX = np.asarray([[0]*(max_length-len(xi)) + xi for xi in train_dataframe.aid_id]).astype('int32')\n\nX_train = X[:,:-1]\ny_train_label = X[:, -1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"2- Test","metadata":{}},{"cell_type":"code","source":"max_length_test = max(map(len, test_dataframe.aid_id))\nX_t = np.asarray([[0]*(max_length_test-len(xi)) + xi for xi in test_dataframe.aid_id]).astype('int32')\n\nX_test = X_t[:,:-1]\ny_test_label = X_t[:, -1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# One-hot encode the labels\ny_train = tf.keras.utils.to_categorical(y_train_label, num_classes=dataframe_copy.aid.nunique())\ny_test = tf.keras.utils.to_categorical(y_test_label, num_classes=dataframe_copy.aid.nunique())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del dataframe_copy\ndel train_df\ndel test_df\ndel train_dataframe\ndel test_dataframe\ndel y_train_label\ndel y_test_label","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define the DL model","metadata":{}},{"cell_type":"code","source":"model = tf.keras.models.Sequential()\n\nmodel.add(layers.Embedding(input_dim=dataframe_copy.aid.nunique(), output_dim=20, input_length=X_train.shape[1]))\n\nmodel.add(layers.Bidirectional(layers.LSTM(64, dropout=0.2, recurrent_dropout=0.2)))\n\nmodel.add(layers.Dense(dataframe_copy.aid.nunique(), activation='softmax'))\n\noptimizer = Adam(learning_rate=0.01)\n\nmodel.compile(loss='categorical_crossentropy', optimizer=optimizer, metrics=['accuracy'])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(X_train, y_train, epochs=32, verbose=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Save the model after training","metadata":{}},{"cell_type":"code","source":"model.save('DNN_model.h5')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading the model","metadata":{}},{"cell_type":"code","source":"# Specify the path to the saved model directory\nmodel_path = 'DNN_model.h5'\n\n# Load the model\nmodel = tf.keras.models.load_model(model_path)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Make predictions using the trained model","metadata":{}},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reverse the mapping dictionary\nreverse_map = {v: k for k, v in map_aid.items()}\n\n# Use the model to predict the `aid_id` values\ny_test_pred = model.predict(X_test)\n\n# For each prediction, find the top 20 `aid_id` values\ntop_20_aid_ids = np.argsort(y_test_pred, axis=1)[:, -20:]\n\n# Map the `aid_id` values back to `aid` values\ntop_20_aids = np.vectorize(reverse_map.get)(top_20_aid_ids)\n\n# Note that the result is still in `aid_id` order, which may not be the same as the original `aid` order\n# To sort them back into `aid` order, you could do:\ntop_20_aids_sorted = np.sort(top_20_aids, axis=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a DataFrame to hold the results\nresults = pd.DataFrame({\n    'session_type': [f\"{session}_{event_type}\" for session, event_type in test_dataframe.index],\n    'labels': [' '.join(map(str, aids)) for aids in top_20_aids_sorted]\n})\n\n# Write the results to a CSV file\nresults.to_csv('submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('submission.csv')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Generating Labels for Test Data based on Historical Events ( function from the otto-offical-github )","metadata":{}},{"cell_type":"code","source":"from typing import List, Dict\n\ndef ground_truth(events: List[Dict]):\n    prev_labels = {\"clicks\": [], \"carts\": [], \"orders\": []}\n\n    for event in reversed(events):\n        event[\"labels\"] = {}\n\n        for label in ['clicks', 'carts', 'orders']:\n            if prev_labels[label]:\n                event[\"labels\"][label] = prev_labels[label].copy()\n\n        if event[\"type\"] == \"clicks\":\n            prev_labels['clicks'].insert(0, event[\"aid\"])\n        if event[\"type\"] == \"carts\":\n            prev_labels['carts'].insert(0, event[\"aid\"])\n        elif event[\"type\"] == \"orders\":\n            prev_labels['orders'].insert(0, event[\"aid\"])\n\n    return events[:-1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataframe.reset_index(inplace=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = test_df[test_df['session'].isin(test_dataframe['session'])]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# assuming df is your DataFrame\ntest['aid'] = test['aid_id']\ntest = test.drop('aid_id', axis=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert the DataFrame to a list of dictionaries\ntest_events = test.to_dict('records')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Call the function with the list of event dictionaries\nground_truth_events = ground_truth(test_events)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Write the data to jsonl file\nimport json\n\nwith open('test_labels.jsonl', 'w') as f:\n    for event in ground_truth_events:\n        f.write(json.dumps(event) + '\\n')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\n\nwith open('test_labels.jsonl', 'r') as f:\n    count = 0\n    for line in f:\n        if count < 1:\n            event = json.loads(line)\n            print('--- Event', count+1, '---')\n            print(json.dumps(event, indent=4))  # Print formatted JSON\n            count += 1\n        else:\n            break","metadata":{"_kg_hide-output":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install beartype","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python -m evaluate --test-labels test_labels.jsonl --predictions submission.csv","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# co-visitation matrix","metadata":{}},{"cell_type":"markdown","source":"### Understanding the Item-Item Collaborative Filtering Approach\n\nOur recommendation approach involves several steps, each serving an essential role in the process. \n\n1. **Mapping unique article identifiers:** This is our preprocessing stage where we convert each unique article ID in the dataset into an integer. This conversion is vital for enhanced performance, particularly because we need to generate a matrix, i.e., the co-visitation matrix, that utilizes these IDs.\n\n2. **Splitting the dataset:** Next, we partition our data into a training set and a test set, based on unique sessions. We opt for sessions, rather than individual records, to ensure all records associated with a particular session remain together in either the training or test set.\n\n3. **Creating a co-visitation matrix:** This step embodies the essence of Item-Item collaborative filtering. We construct a co-visitation matrix using the training data, where the entries account for the frequency at which each pair of articles (items) were visited during the same session. Consequently, this matrix encapsulates the \"similarity\" between pairs of items grounded on their co-visitation count. This similarity is then leveraged to generate recommendations.\n\n4. **Generating recommendations:** For each unique session in the test set, we calculate a \"score\" for each article. This score is the sum of the co-visitation counts of that article with all other articles visited during that session, as documented in the co-visitation matrix. Subsequently, we select the top 20 articles boasting the highest scores. These articles are those that are most \"similar\" (i.e., frequently co-visited) to the articles in the current session, and are therefore recommended.\n\n5. **Evaluating the model:** Finally, we assess our model's performance using recall. For each session, the algorithm determines how many of the actual articles visited during that session feature among the top 20 recommended articles. The final performance metric is then derived as the average of these recall values across all sessions.\n\nThis approach stands out as an Item-Item Collaborative Filtering technique, where recommendations are based on the similarity between items. It differs from User-User Collaborative Filtering, where the similarity between users (sessions, in this context) would drive the recommendation process.\n","metadata":{}},{"cell_type":"code","source":"from collections import Counter\nfrom itertools import combinations\nfrom scipy import sparse as sps\nfrom tqdm import tqdm","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the path to your data\ndata_path = Path('dataset/')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_paths = sorted(glob('train_parquet/*'))[:20]\ndataframes = []","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for file_path in tqdm(file_paths):\n    dataframes.append(pd.read_parquet(file_path))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframes = pd.concat(dataframes).reset_index(drop=True)\ndataframes","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe_copy = dataframes.copy()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del dataframes","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe_copy","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Map the unique article ids to integers\nunique_ids = np.arange(dataframe_copy.aid.nunique())\nnp.random.shuffle(unique_ids)\nmap_aid = {i:j for i, j in zip(dataframe_copy.aid.unique(), unique_ids)}","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reverse mapping for later use\nreverse_map_aid = {v: k for k, v in map_aid.items()}","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assign each article an integer id\ndataframe_copy['aid_id'] = dataframe_copy['aid'].map(map_aid)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Splitting data into training and test set\nnum_sessions = dataframe_copy['session'].nunique()  # total number of sessions\ntest_sessions = int(num_sessions * 0.01)  # 1% of total sessions for testing\ntest_session_ids = dataframe_copy['session'].unique()[-test_sessions:]  # session ids for test set","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create training and test dataframes\ntrain_df = dataframe_copy[~dataframe_copy['session'].isin(test_session_ids)]\ntest_df = dataframe_copy[dataframe_copy['session'].isin(test_session_ids)]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a co-visitation matrix for training data\n# Co-visitation is defined as the number of times two articles appear in the same session\nco_visits = Counter()\nfor _, group in train_df.groupby('session'):\n    aids = group['aid_id'].values\n    co_visits.update(combinations(aids, 2)) ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create sparse matrix from the counter\nrows, cols = zip(*co_visits.keys())\ndata = list(co_visits.values())\nco_matrix = sps.coo_matrix((data, (rows, cols)), shape=(len(unique_ids), len(unique_ids)))\nco_matrix_csr = co_matrix.tocsr()  # Convert the matrix to CSR format for efficient arithmetic and matrix operations","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to calculate recall\ndef recall_per_session(predicted_aids, actual_aids):\n    hits = len(set(predicted_aids) & set(actual_aids))\n    total_actual = len(actual_aids)\n    recall = hits / total_actual\n    return recall\n\n# Function to generate recommendations for each session\ndef generate_recommendations(df):\n    recommendations = dict()\n    for (session, type_), group in tqdm(df.groupby(['session', 'type']), \"Generating recommendations\"):\n        aids = group['aid_id'].values\n        scores = np.asarray(co_matrix_csr[aids].sum(axis=0)).squeeze()\n        top_aid_ids = np.argpartition(scores, -20)[-20:]  # Get the top 20 articles\n        recommendations[(session, type_)] = top_aid_ids\n    return recommendations","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate recommendations for test set\ntest_recommendations = generate_recommendations(test_df)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print the top 20 recommendations for each session and type in the test set\nfor (session_id, type_), top20_aids in test_recommendations.items():\n    print(f\"Session ID: {session_id}, Type: {type_}\")\n    print(f\"Recommended aid values: {top20_aids}\")\n    print(\"\\n\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate recall for test set\nrecalls = []\nfor key, actual in test_df.groupby(['session', 'type'])['aid_id']:\n    if key in test_recommendations:\n        predicted = test_recommendations[key]\n        recalls.append(recall_per_session(predicted, actual))\nprint(\"Average recall on test set: \", np.mean(recalls))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Matrix Factorization","metadata":{}},{"cell_type":"markdown","source":"### Understanding the Matrix Factorization Approach\n\nOur recommendation approach entails several critical steps to build and evaluate a Matrix Factorization model for our session-based recommendation task. Here's a breakdown of these steps:\n\n1. **Mapping unique article and session identifiers:** The preprocessing stage involves converting each unique article ID and session ID in the dataset into an integer. This transformation ensures improved performance during the creation of an interaction matrix. This interaction matrix represents the session-article interactions where rows represent sessions and columns represent articles. Entries in this matrix are binary, indicating whether an article was visited during a particular session.\n\n2. **Splitting the dataset:** Next, we partition our data into a training set and a test set, based on unique sessions. We opt for sessions, rather than individual records, to ensure all records associated with a particular session remain together in either the training or test set.\n\n3. **Creating a Session-Article Interaction Matrix:** We create an interaction matrix, where the rows represent unique sessions, and the columns represent unique articles. If a particular article is visited during a specific session, the corresponding entry in the matrix is marked as 1; otherwise, it remains 0. This matrix plays a crucial role in characterizing the interactions between sessions and articles.\n\n4. **Matrix Factorization using Non-negative Matrix Factorization (NMF):** At the heart of our approach is the application of Non-negative Matrix Factorization (NMF) on the interaction matrix. NMF decomposes the interaction matrix into two lower-rank matrices: one representing the 'session-latent factor' relationships and the other representing 'article-latent factor' relationships. The latent factors can be thought of as underlying patterns that explain the observed interactions between sessions and articles.\n\n5. **Generating Recommendations:** For each unique session, we compute a 'score' for each article based on the dot product of the session's latent factor vector and the latent factor vectors of all articles. This score measures the likelihood of the session interacting with each article. We then select the top 20 articles with the highest scores as our recommendations.\n\n6. **Evaluating the Model:** Finally, we assess the performance of our model using recall. For each session, the algorithm calculates how many of the actual articles visited during that session are included in the top 20 recommended articles. The final performance metric is the average of these recall values across all sessions.\n\nMatrix Factorization, unlike Collaborative Filtering, doesn't rely on explicit similarity measures between items or users. Instead, it infers latent factors from the interaction data, which captures the underlying patterns driving the interactions. These inferred patterns are then used to make recommendations.","metadata":{}},{"cell_type":"code","source":"from scipy import sparse as sps\nfrom sklearn.decomposition import NMF\nfrom tqdm import tqdm","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the path to your data\ndata_path = Path('dataset/')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_paths = sorted(glob('train_parquet/*'))[:20]\ndataframes = []","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for file_path in tqdm(file_paths):\n    dataframes.append(pd.read_parquet(file_path))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframes = pd.concat(dataframes).reset_index(drop=True)\ndataframes","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe_copy = dataframes.copy()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe_copy","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Map the unique article ids to integers\nunique_ids = np.arange(dataframe_copy.aid.nunique())\nnp.random.shuffle(unique_ids)\nmap_aid = {i:j for i, j in zip(dataframe_copy.aid.unique(), unique_ids)}","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reverse mapping for later use\nreverse_map_aid = {v: k for k, v in map_aid.items()}\n\n# Assign each article an integer id\ndataframe_copy['aid_id'] = dataframe_copy['aid'].map(map_aid)\n\n# Create a unique index for session & type pair\ndataframe_copy['session_type'] = dataframe_copy['session'].astype(str) + '_' + dataframe_copy['type']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assign each article an integer id\ndataframe_copy['aid_id'] = dataframe_copy['aid'].map(map_aid)\n\n# Map the unique session ids to integers\nunique_sessions = np.arange(dataframe_copy.session.nunique())\nnp.random.shuffle(unique_sessions)\nmap_session = {i:j for i, j in zip(dataframe_copy.session.unique(), unique_sessions)}\n\n# Assign each session an integer id\ndataframe_copy['session_id'] = dataframe_copy['session'].map(map_session)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Splitting data into training and test set\nnum_sessions = dataframe_copy['session'].nunique()  # total number of sessions\ntest_sessions = int(num_sessions * 0.2)  # 20% of total sessions for testing\ntest_session_ids = dataframe_copy['session'].unique()[-test_sessions:]  # session ids for test set\n\n# Create training and test dataframes\ntrain_df = dataframe_copy[~dataframe_copy['session'].isin(test_session_ids)]\ntest_df = dataframe_copy[dataframe_copy['session'].isin(test_session_ids)]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create session-article matrix\nrows, cols = train_df['session_id'], train_df['aid_id']\ndata = np.ones(len(rows))\nmatrix = sps.coo_matrix((data, (rows, cols)))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Perform matrix factorization using Non-negative Matrix Factorization (NMF)\nn_components = 1000 # Number of components to keep\nnmf = NMF(n_components=n_components, init='random', random_state=0)\nW = nmf.fit_transform(session_article_matrix)  # Session-latent factors matrix\nH = nmf.components_  # Article-latent factors matrix","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to calculate recall\ndef recall_per_session(predicted_aids, actual_aids):\n    hits = len(set(predicted_aids) & set(actual_aids))\n    total_actual = len(actual_aids)\n    recall = hits / total_actual\n    return recall\n\n# Function to generate recommendations for each session\ndef generate_recommendations(df):\n    recommendations = dict()\n    for session in tqdm(df['session_id'].unique(), \"Generating recommendations\"):\n        session_vector = W[session]\n        scores = np.dot(session_vector, H)\n        top_aid_ids = np.argpartition(scores, -20)[-20:]  # Get the top 20 articles\n        recommendations[session] = top_aid_ids\n    return recommendations","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate recommendations for each session\nrecommendations = generate_recommendations(test_df)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the session-type for each session_id\nsession_type_df = test_df[['session', 'type', 'session_id']].drop_duplicates()\n\n# Print the top 20 recommendations for each session and type in the test set\nfor _, row in session_type_df.iterrows():\n    if row['session_id'] in recommendations:\n        print(f\"Session ID: {row['session']}, Type: {row['type']}\")\n        print(f\"Recommended aid values: {recommendations[row['session_id']]}\")\n        print(\"\\n\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate recommendations for test set\ntest_recommendations = generate_recommendations(test_df)\n\n# Calculate recall for test set\nrecalls = []\nfor session, group in test_df.groupby('session_id'):\n    actual = group['aid_id'].values\n    if session in test_recommendations:\n        predicted = test_recommendations[session]\n        recalls.append(recall_per_session(predicted, actual))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Average Recall:', np.mean(recalls))","metadata":{},"execution_count":null,"outputs":[]}]}