{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[],"dockerImageVersionId":30146,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install retrying\n!pip install yaml\n!pip install nbformat\n!pip install nbconvert\n!pip install --upgrade typing_extensions\n!pip install recommenders --no-deps","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport sys\nimport numpy as np\nimport zipfile\nfrom tqdm import tqdm\nfrom tempfile import TemporaryDirectory\nimport tensorflow as tf\ntf.get_logger().setLevel('ERROR') # only show error messages\n\nfrom recommenders.models.deeprec.deeprec_utils import download_deeprec_resources \nfrom recommenders.models.newsrec.newsrec_utils import prepare_hparams\nfrom recommenders.models.newsrec.models.naml import NAMLModel\nfrom recommenders.models.newsrec.io.mind_all_iterator import MINDAllIterator\nfrom recommenders.models.newsrec.newsrec_utils import get_mind_data_set\nfrom recommenders.utils.notebook_utils import store_metadata\n\nprint(\"System version: {}\".format(sys.version))\nprint(\"Tensorflow version: {}\".format(tf.__version__))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T19:16:55.748943Z","iopub.execute_input":"2025-03-10T19:16:55.749740Z","iopub.status.idle":"2025-03-10T19:16:57.441203Z","shell.execute_reply.started":"2025-03-10T19:16:55.749701Z","shell.execute_reply":"2025-03-10T19:16:57.440625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"epochs = 5\nseed = 42\nbatch_size = 32\n\n# Options: demo, small, large\nMIND_type = 'small'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T19:17:00.768551Z","iopub.execute_input":"2025-03-10T19:17:00.768821Z","iopub.status.idle":"2025-03-10T19:17:00.772789Z","shell.execute_reply.started":"2025-03-10T19:17:00.768793Z","shell.execute_reply":"2025-03-10T19:17:00.771867Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tmpdir = TemporaryDirectory()\n# data_path = tmpdir.name\ndata_path = \"/kaggle/working/\"\ntrain_news_file = os.path.join(data_path, 'train', r'news.tsv')\ntrain_behaviors_file = os.path.join(data_path, 'train', r'behaviors.tsv')\nvalid_news_file = os.path.join(data_path, 'valid', r'news.tsv')\nvalid_behaviors_file = os.path.join(data_path, 'valid', r'behaviors.tsv')\nwordEmb_file = os.path.join(data_path, \"utils\", \"embedding_all.npy\")\nuserDict_file = os.path.join(data_path, \"utils\", \"uid2index.pkl\")\nwordDict_file = os.path.join(data_path, \"utils\", \"word_dict_all.pkl\")\nvertDict_file = os.path.join(data_path, \"utils\", \"vert_dict.pkl\")\nsubvertDict_file = os.path.join(data_path, \"utils\", \"subvert_dict.pkl\")\nyaml_file = os.path.join(data_path, \"utils\", r'naml.yaml')\n\nmind_url, mind_train_dataset, mind_dev_dataset, mind_utils = get_mind_data_set(MIND_type)\n\nif not os.path.exists(train_news_file):\n    download_deeprec_resources(mind_url, os.path.join(data_path, 'train'), mind_train_dataset)\n    \nif not os.path.exists(valid_news_file):\n    download_deeprec_resources(mind_url, \\\n                               os.path.join(data_path, 'valid'), mind_dev_dataset)\nif not os.path.exists(yaml_file):\n    download_deeprec_resources(r'https://recodatasets.z20.web.core.windows.net/newsrec/', \\\n                               os.path.join(data_path, 'utils'), mind_utils)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T19:17:03.933556Z","iopub.execute_input":"2025-03-10T19:17:03.933822Z","iopub.status.idle":"2025-03-10T19:17:03.942156Z","shell.execute_reply.started":"2025-03-10T19:17:03.933794Z","shell.execute_reply":"2025-03-10T19:17:03.941434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\n\nwith open(\"/kaggle/working/utils/vert_dict.pkl\", \"rb\") as f:\n    vert_dict = pickle.load(f)\n\nwith open(\"/kaggle/working/utils/subvert_dict.pkl\", \"rb\") as f:\n    subvert_dict = pickle.load(f)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T19:17:11.648676Z","iopub.execute_input":"2025-03-10T19:17:11.648936Z","iopub.status.idle":"2025-03-10T19:17:11.653794Z","shell.execute_reply.started":"2025-03-10T19:17:11.648907Z","shell.execute_reply":"2025-03-10T19:17:11.653027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hparams = prepare_hparams(yaml_file, \n                          vert_num=max(vert_dict.values()) + 1,  \n                          subvert_num=max(subvert_dict.values()) + 1,\n                          batch_size=32,\n                          epochs=5,\n                          wordEmb_file=wordEmb_file,\n                          wordDict_file=wordDict_file, \n                          userDict_file=userDict_file,\n                          vertDict_file=vertDict_file, \n                          subvertDict_file=subvertDict_file,\n                          )\n\n\niterator = MINDAllIterator","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T19:17:13.563447Z","iopub.execute_input":"2025-03-10T19:17:13.564148Z","iopub.status.idle":"2025-03-10T19:17:13.572189Z","shell.execute_reply.started":"2025-03-10T19:17:13.564114Z","shell.execute_reply":"2025-03-10T19:17:13.571519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = NAMLModel(hparams, iterator, seed=seed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T19:17:25.833745Z","iopub.execute_input":"2025-03-10T19:17:25.834504Z","iopub.status.idle":"2025-03-10T19:17:27.967552Z","shell.execute_reply.started":"2025-03-10T19:17:25.834450Z","shell.execute_reply":"2025-03-10T19:17:27.966785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nmodel.fit(train_news_file, train_behaviors_file, valid_news_file, valid_behaviors_file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T19:17:19.958968Z","iopub.status.idle":"2025-03-10T19:17:19.959339Z","shell.execute_reply.started":"2025-03-10T19:17:19.959139Z","shell.execute_reply":"2025-03-10T19:17:19.959159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#model_path = os.path.join(data_path, \"model\")\nmodel_path = \"/kaggle/working/\"\nos.makedirs(model_path, exist_ok=True)\n\nmodel.model.save_weights(os.path.join(model_path, \"naml_ckpt\"))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nres_syn = model.run_eval(valid_news_file, valid_behaviors_file)\nprint(res_syn)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"store_metadata(\"group_auc\", res_syn['group_auc'])\nstore_metadata(\"mean_mrr\", res_syn['mean_mrr'])\nstore_metadata(\"ndcg@5\", res_syn['ndcg@5'])\nstore_metadata(\"ndcg@10\", res_syn['ndcg@10'])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_path = \"/kaggle/working/\"\nmodel.model.load_weights(os.path.join(model_path, \"naml_ckpt\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T19:18:07.263477Z","iopub.execute_input":"2025-03-10T19:18:07.263776Z","iopub.status.idle":"2025-03-10T19:18:07.776976Z","shell.execute_reply.started":"2025-03-10T19:18:07.263749Z","shell.execute_reply":"2025-03-10T19:18:07.776154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scorer = model._build_graph()[1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T19:18:56.880990Z","iopub.execute_input":"2025-03-10T19:18:56.881519Z","iopub.status.idle":"2025-03-10T19:18:59.191394Z","shell.execute_reply.started":"2025-03-10T19:18:56.881468Z","shell.execute_reply":"2025-03-10T19:18:59.190849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport random\nimport pickle\nimport numpy as np\nimport pandas as pd\nimport tensorflow.keras as keras\n\n# Define hyperparameters\nclass HParams:\n    his_size = 50\n    title_size = 30\n    body_size = 50\n\nhparams = HParams()\n\n# Define data paths (update these paths to your local directories)\ndata_path = \"/kaggle/working/\"\nwordDict_file = os.path.join(data_path, \"utils\", \"word_dict_all.pkl\")\nvertDict_file = os.path.join(data_path, \"utils\", \"vert_dict.pkl\")\nsubvertDict_file = os.path.join(data_path, \"utils\", \"subvert_dict.pkl\")\nnews_file = os.path.join(data_path, 'valid', 'news.tsv')\nbehaviors_file = os.path.join(data_path, 'valid', 'behaviors.tsv')\n\n# Load dictionaries\nwith open(wordDict_file, \"rb\") as f:\n    word_dict = pickle.load(f)\nwith open(vertDict_file, \"rb\") as f:\n    vert_dict = pickle.load(f)\nwith open(subvertDict_file, \"rb\") as f:\n    subvert_dict = pickle.load(f)\n\n# Load news and behaviors data\nnews_df = pd.read_csv(news_file, sep=\"\\t\", header=None,\n                      names=[\"news_id\", \"category\", \"subcategory\", \"title\", \"abstract\",\n                             \"url\", \"title_entities\", \"abstract_entities\"])\nbehaviors_df = pd.read_csv(behaviors_file, sep=\"\\t\", header=None,\n                           names=[\"impression_id\", \"user_id\", \"timestamp\", \"history\", \"impressions\"])\n\n# Tokenization function\ndef tokenize_text(text, max_length):\n    tokens = str(text).lower().split()[:max_length]\n    token_ids = [word_dict.get(token, 0) for token in tokens]  # Default to 0 if word not found\n    return token_ids + [0] * (max_length - len(token_ids))  # Pad to max_length\n\n# Convert news articles into a dictionary with NAML-ready inputs\ndef convert_news(news_df):\n    news_dict = {}\n    for _, row in news_df.iterrows():\n        news_dict[row[\"news_id\"]] = {\n            \"title\": tokenize_text(row[\"title\"], hparams.title_size),\n            \"body\": tokenize_text(row[\"abstract\"], hparams.body_size),\n            \"category\": vert_dict.get(row[\"category\"], 0),\n            \"subcategory\": subvert_dict.get(row[\"subcategory\"], 0),\n        }\n    return news_dict\n\nnews_dict = convert_news(news_df)\n\n# Select a random user from the behaviors\nsample_user = random.choice(behaviors_df[\"user_id\"].dropna().unique())\nuser_data = behaviors_df[behaviors_df[\"user_id\"] == sample_user]\n\n# Get the user's clicked history (last his_size news items)\nif not user_data.empty and isinstance(user_data[\"history\"].values[0], str):\n    user_clicked_news = user_data[\"history\"].values[0].split()[-hparams.his_size:]\nelse:\n    user_clicked_news = []\n\n# Build clicked news batches\nclicked_title_batch = [news_dict[n][\"title\"] for n in user_clicked_news if n in news_dict]\nclicked_body_batch = [news_dict[n][\"body\"] for n in user_clicked_news if n in news_dict]\nclicked_vert_batch = [[news_dict[n][\"category\"]] for n in user_clicked_news if n in news_dict]\nclicked_subvert_batch = [[news_dict[n][\"subcategory\"]] for n in user_clicked_news if n in news_dict]\n\n# Pad to fixed history size if needed\nwhile len(clicked_title_batch) < hparams.his_size:\n    clicked_title_batch.append([0] * hparams.title_size)\n    clicked_body_batch.append([0] * hparams.body_size)\n    clicked_vert_batch.append([0])\n    clicked_subvert_batch.append([0])\n\n# Convert and reshape to numpy arrays\nclicked_title_batch = np.array(clicked_title_batch).reshape(1, hparams.his_size, hparams.title_size)\nclicked_body_batch = np.array(clicked_body_batch).reshape(1, hparams.his_size, hparams.body_size)\nclicked_vert_batch = np.array(clicked_vert_batch).reshape(1, hparams.his_size, 1)\nclicked_subvert_batch = np.array(clicked_subvert_batch).reshape(1, hparams.his_size, 1)\n\n# Debug: Print user history details\nprint(\"User ID:\", sample_user)\nprint(\"Clicked Title Batch shape:\", clicked_title_batch.shape)\nprint(\"Clicked Title Batch first row sample:\", clicked_title_batch[0, 0, :])\nprint(\"Sum of first clicked title vector:\", np.sum(clicked_title_batch[0, 0, :]))\nprint(\"Clicked Body Batch shape:\", clicked_body_batch.shape)\nprint(\"Clicked Body Batch first row sample:\", clicked_body_batch[0, 0, :])\nprint(\"Sum of first clicked body vector:\", np.sum(clicked_body_batch[0, 0, :]))\nprint(\"Clicked Vert Batch shape:\", clicked_vert_batch.shape)\nprint(\"Clicked Vert Batch first sample:\", clicked_vert_batch[0, 0, :])\nprint(\"Clicked Subvert Batch shape:\", clicked_subvert_batch.shape)\nprint(\"Clicked Subvert Batch first sample:\", clicked_subvert_batch[0, 0, :])\n\n# Select a random candidate news article from the news dictionary\ncandidate_news_id = random.choice(list(news_dict.keys()))\ncandidate = news_dict[candidate_news_id]\nprint(f\"\\nRandom candidate news ID: {candidate_news_id}\")\n\n# Prepare candidate features\ncandidate_title_batch = np.array(candidate[\"title\"]).reshape(1, 1, hparams.title_size)\ncandidate_body_batch = np.array(candidate[\"body\"]).reshape(1, 1, hparams.body_size)\ncandidate_vert_batch = np.array([[candidate[\"category\"]]]).reshape(1, 1, 1)\ncandidate_subvert_batch = np.array([[candidate[\"subcategory\"]]]).reshape(1, 1, 1)\n\n# Debug: Print candidate features details\nprint(\"Candidate Title Batch shape:\", candidate_title_batch.shape)\nprint(\"Candidate Title Batch sample:\", candidate_title_batch[0, 0, :])\nprint(\"Sum of candidate title vector:\", np.sum(candidate_title_batch[0, 0, :]))\nprint(\"Candidate Body Batch shape:\", candidate_body_batch.shape)\nprint(\"Candidate Body Batch sample:\", candidate_body_batch[0, 0, :])\nprint(\"Sum of candidate body vector:\", np.sum(candidate_body_batch[0, 0, :]))\nprint(\"Candidate Vert Batch shape:\", candidate_vert_batch.shape)\nprint(\"Candidate Vert Batch sample:\", candidate_vert_batch[0, 0, :])\nprint(\"Candidate Subvert Batch shape:\", candidate_subvert_batch.shape)\nprint(\"Candidate Subvert Batch sample:\", candidate_subvert_batch[0, 0, :])\n\n# Print scorer model summary to inspect the layers and architecture\nprint(\"\\nScorer Model Summary:\")\nscorer.summary()\n\n# Run the scorer model (assumes 'scorer' is already loaded from your trained NAML model)\nscore = scorer.predict([\n    clicked_title_batch,\n    clicked_body_batch,\n    clicked_vert_batch,\n    clicked_subvert_batch,\n    candidate_title_batch,\n    candidate_body_batch,\n    candidate_vert_batch,\n    candidate_subvert_batch\n])\n\n# Debug: Print raw model output\nprint(\"\\nRaw score from scorer model:\", score)\nprint(f\"Predicted Click Probability for news '{candidate_news_id}': {score.flatten()[0]:.4f}\")\n\n# Optional: If you want to inspect the pre-sigmoid (raw dot product) values,\n# you could try to create an intermediate model that outputs the penultimate layer.\n# Note: This depends on your model architecture and might require adjustments.\ntry:\n    # Assuming the last layer is the sigmoid activation and the one before is the raw dot product:\n    pre_sigmoid_model = keras.Model(inputs=scorer.input, outputs=scorer.layers[-2].output)\n    raw_dot_product = pre_sigmoid_model.predict([\n        clicked_title_batch,\n        clicked_body_batch,\n        clicked_vert_batch,\n        clicked_subvert_batch,\n        candidate_title_batch,\n        candidate_body_batch,\n        candidate_vert_batch,\n        candidate_subvert_batch\n    ])\n    print(\"Raw dot product (pre-sigmoid):\", raw_dot_product)\nexcept Exception as e:\n    print(\"Could not extract pre-sigmoid output. Error:\", e)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T19:23:56.263184Z","iopub.execute_input":"2025-03-10T19:23:56.263455Z","iopub.status.idle":"2025-03-10T19:24:00.882137Z","shell.execute_reply.started":"2025-03-10T19:23:56.263424Z","shell.execute_reply":"2025-03-10T19:24:00.881374Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# RETRAIN","metadata":{}},{"cell_type":"code","source":"import os\nimport sys\nimport numpy as np\nimport zipfile\nfrom tqdm import tqdm\nfrom tempfile import TemporaryDirectory\nimport tensorflow as tf\ntf.get_logger().setLevel('ERROR') # only show error messages\n\nfrom recommenders.models.deeprec.deeprec_utils import download_deeprec_resources \nfrom recommenders.models.newsrec.newsrec_utils import prepare_hparams\nfrom recommenders.models.newsrec.models.naml import NAMLModel\nfrom recommenders.models.newsrec.io.mind_all_iterator import MINDAllIterator\nfrom recommenders.models.newsrec.newsrec_utils import get_mind_data_set\nfrom recommenders.utils.notebook_utils import store_metadata\n\nprint(\"System version: {}\".format(sys.version))\nprint(\"Tensorflow version: {}\".format(tf.__version__))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T19:41:49.501275Z","iopub.execute_input":"2025-03-10T19:41:49.501498Z","iopub.status.idle":"2025-03-10T19:41:51.131317Z","shell.execute_reply.started":"2025-03-10T19:41:49.501470Z","shell.execute_reply":"2025-03-10T19:41:51.130582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"epochs = 5\nseed = 42\nbatch_size = 32\n\n# Options: demo, small, large\nMIND_type = 'small'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T19:41:51.132207Z","iopub.execute_input":"2025-03-10T19:41:51.132395Z","iopub.status.idle":"2025-03-10T19:41:51.136243Z","shell.execute_reply.started":"2025-03-10T19:41:51.132372Z","shell.execute_reply":"2025-03-10T19:41:51.135332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tmpdir = TemporaryDirectory()\ndata_path = tmpdir.name\n\ntrain_news_file = os.path.join(data_path, 'train', r'news.tsv')\ntrain_behaviors_file = os.path.join(data_path, 'train', r'behaviors.tsv')\nvalid_news_file = os.path.join(data_path, 'valid', r'news.tsv')\nvalid_behaviors_file = os.path.join(data_path, 'valid', r'behaviors.tsv')\nwordEmb_file = os.path.join(data_path, \"utils\", \"embedding_all.npy\")\nuserDict_file = os.path.join(data_path, \"utils\", \"uid2index.pkl\")\nwordDict_file = os.path.join(data_path, \"utils\", \"word_dict_all.pkl\")\nvertDict_file = os.path.join(data_path, \"utils\", \"vert_dict.pkl\")\nsubvertDict_file = os.path.join(data_path, \"utils\", \"subvert_dict.pkl\")\nyaml_file = os.path.join(data_path, \"utils\", r'naml.yaml')\n\nmind_url, mind_train_dataset, mind_dev_dataset, mind_utils = get_mind_data_set(MIND_type)\n\nif not os.path.exists(train_news_file):\n    download_deeprec_resources(mind_url, os.path.join(data_path, 'train'), mind_train_dataset)\n    \nif not os.path.exists(valid_news_file):\n    download_deeprec_resources(mind_url, \\\n                               os.path.join(data_path, 'valid'), mind_dev_dataset)\nif not os.path.exists(yaml_file):\n    download_deeprec_resources(r'https://recodatasets.z20.web.core.windows.net/newsrec/', \\\n                               os.path.join(data_path, 'utils'), mind_utils)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T19:41:51.138023Z","iopub.execute_input":"2025-03-10T19:41:51.138346Z","iopub.status.idle":"2025-03-10T19:42:07.262571Z","shell.execute_reply.started":"2025-03-10T19:41:51.138304Z","shell.execute_reply":"2025-03-10T19:42:07.262023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hparams = prepare_hparams(yaml_file, \n                          wordEmb_file=wordEmb_file,\n                          wordDict_file=wordDict_file, \n                          userDict_file=userDict_file,\n                          vertDict_file=vertDict_file, \n                          subvertDict_file=subvertDict_file,\n                          batch_size=batch_size,\n                          epochs=epochs)\nprint(hparams)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T19:42:07.263679Z","iopub.execute_input":"2025-03-10T19:42:07.264228Z","iopub.status.idle":"2025-03-10T19:42:07.272591Z","shell.execute_reply.started":"2025-03-10T19:42:07.264187Z","shell.execute_reply":"2025-03-10T19:42:07.271877Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"iterator = MINDAllIterator\nmodel = NAMLModel(hparams, iterator, seed=seed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T19:42:07.273502Z","iopub.execute_input":"2025-03-10T19:42:07.273671Z","iopub.status.idle":"2025-03-10T19:42:12.109676Z","shell.execute_reply.started":"2025-03-10T19:42:07.273652Z","shell.execute_reply":"2025-03-10T19:42:12.108939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nmodel.fit(train_news_file, train_behaviors_file, valid_news_file, valid_behaviors_file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T19:42:12.110593Z","iopub.execute_input":"2025-03-10T19:42:12.110761Z","iopub.status.idle":"2025-03-10T22:53:01.140797Z","shell.execute_reply.started":"2025-03-10T19:42:12.110740Z","shell.execute_reply":"2025-03-10T22:53:01.140126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_path = os.path.join(\"/kaggle/working/\", \"model\")\nos.makedirs(model_path, exist_ok=True)\n\nmodel.model.save_weights(os.path.join(model_path, \"naml_ckpt\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T23:13:02.345724Z","iopub.execute_input":"2025-03-10T23:13:02.346511Z","iopub.status.idle":"2025-03-10T23:13:02.688717Z","shell.execute_reply.started":"2025-03-10T23:13:02.346476Z","shell.execute_reply":"2025-03-10T23:13:02.687956Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nres_syn = model.run_eval(valid_news_file, valid_behaviors_file)\nprint(res_syn)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:53:01.801993Z","iopub.execute_input":"2025-03-10T22:53:01.802677Z","iopub.status.idle":"2025-03-10T22:58:46.747271Z","shell.execute_reply.started":"2025-03-10T22:53:01.802637Z","shell.execute_reply":"2025-03-10T22:58:46.746579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Record results for tests - ignore this cell\nstore_metadata(\"group_auc\", res_syn['group_auc'])\nstore_metadata(\"mean_mrr\", res_syn['mean_mrr'])\nstore_metadata(\"ndcg@5\", res_syn['ndcg@5'])\nstore_metadata(\"ndcg@10\", res_syn['ndcg@10'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T22:58:46.751228Z","iopub.execute_input":"2025-03-10T22:58:46.751431Z","iopub.status.idle":"2025-03-10T22:58:46.759311Z","shell.execute_reply.started":"2025-03-10T22:58:46.751407Z","shell.execute_reply":"2025-03-10T22:58:46.758612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scorer = model._build_graph()[1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T23:15:20.519480Z","iopub.execute_input":"2025-03-10T23:15:20.520066Z","iopub.status.idle":"2025-03-10T23:15:23.071630Z","shell.execute_reply.started":"2025-03-10T23:15:20.520032Z","shell.execute_reply":"2025-03-10T23:15:23.070864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport random\nimport pickle\nimport numpy as np\nimport pandas as pd\nimport tensorflow.keras as keras\n\n# Define hyperparameters\nclass HParams:\n    his_size = 50\n    title_size = 30\n    body_size = 50\n\nhparams = HParams()\n\n# Define data paths (update these paths to your local directories)\ndata_path = \"/kaggle/working/\"\nwordDict_file = os.path.join(data_path, \"utils\", \"word_dict_all.pkl\")\nvertDict_file = os.path.join(data_path, \"utils\", \"vert_dict.pkl\")\nsubvertDict_file = os.path.join(data_path, \"utils\", \"subvert_dict.pkl\")\nnews_file = os.path.join(data_path, 'valid', 'news.tsv')\nbehaviors_file = os.path.join(data_path, 'valid', 'behaviors.tsv')\n\n# Load dictionaries\nwith open(wordDict_file, \"rb\") as f:\n    word_dict = pickle.load(f)\nwith open(vertDict_file, \"rb\") as f:\n    vert_dict = pickle.load(f)\nwith open(subvertDict_file, \"rb\") as f:\n    subvert_dict = pickle.load(f)\n\n# Load news and behaviors data\nnews_df = pd.read_csv(news_file, sep=\"\\t\", header=None,\n                      names=[\"news_id\", \"category\", \"subcategory\", \"title\", \"abstract\",\n                             \"url\", \"title_entities\", \"abstract_entities\"])\nbehaviors_df = pd.read_csv(behaviors_file, sep=\"\\t\", header=None,\n                           names=[\"impression_id\", \"user_id\", \"timestamp\", \"history\", \"impressions\"])\n\n# Tokenization function\ndef tokenize_text(text, max_length):\n    tokens = str(text).lower().split()[:max_length]\n    token_ids = [word_dict.get(token, 0) for token in tokens]  # Default to 0 if word not found\n    return token_ids + [0] * (max_length - len(token_ids))  # Pad to max_length\n\n# Convert news articles into a dictionary with NAML-ready inputs\ndef convert_news(news_df):\n    news_dict = {}\n    for _, row in news_df.iterrows():\n        news_dict[row[\"news_id\"]] = {\n            \"title\": tokenize_text(row[\"title\"], hparams.title_size),\n            \"body\": tokenize_text(row[\"abstract\"], hparams.body_size),\n            \"category\": vert_dict.get(row[\"category\"], 0),\n            \"subcategory\": subvert_dict.get(row[\"subcategory\"], 0),\n        }\n    return news_dict\n\nnews_dict = convert_news(news_df)\n\n# Select a random user from the behaviors\nsample_user = random.choice(behaviors_df[\"user_id\"].dropna().unique())\nuser_data = behaviors_df[behaviors_df[\"user_id\"] == sample_user]\n\n# Get the user's clicked history (last his_size news items)\nif not user_data.empty and isinstance(user_data[\"history\"].values[0], str):\n    user_clicked_news = user_data[\"history\"].values[0].split()[-hparams.his_size:]\nelse:\n    user_clicked_news = []\n\n# Build clicked news batches\nclicked_title_batch = [news_dict[n][\"title\"] for n in user_clicked_news if n in news_dict]\nclicked_body_batch = [news_dict[n][\"body\"] for n in user_clicked_news if n in news_dict]\nclicked_vert_batch = [[news_dict[n][\"category\"]] for n in user_clicked_news if n in news_dict]\nclicked_subvert_batch = [[news_dict[n][\"subcategory\"]] for n in user_clicked_news if n in news_dict]\n\n# Pad to fixed history size if needed\nwhile len(clicked_title_batch) < hparams.his_size:\n    clicked_title_batch.append([0] * hparams.title_size)\n    clicked_body_batch.append([0] * hparams.body_size)\n    clicked_vert_batch.append([0])\n    clicked_subvert_batch.append([0])\n\n# Convert and reshape to numpy arrays\nclicked_title_batch = np.array(clicked_title_batch).reshape(1, hparams.his_size, hparams.title_size)\nclicked_body_batch = np.array(clicked_body_batch).reshape(1, hparams.his_size, hparams.body_size)\nclicked_vert_batch = np.array(clicked_vert_batch).reshape(1, hparams.his_size, 1)\nclicked_subvert_batch = np.array(clicked_subvert_batch).reshape(1, hparams.his_size, 1)\n\n# Debug: Print user history details\nprint(\"User ID:\", sample_user)\nprint(\"Clicked Title Batch shape:\", clicked_title_batch.shape)\nprint(\"Clicked Title Batch first row sample:\", clicked_title_batch[0, 0, :])\nprint(\"Sum of first clicked title vector:\", np.sum(clicked_title_batch[0, 0, :]))\nprint(\"Clicked Body Batch shape:\", clicked_body_batch.shape)\nprint(\"Clicked Body Batch first row sample:\", clicked_body_batch[0, 0, :])\nprint(\"Sum of first clicked body vector:\", np.sum(clicked_body_batch[0, 0, :]))\nprint(\"Clicked Vert Batch shape:\", clicked_vert_batch.shape)\nprint(\"Clicked Vert Batch first sample:\", clicked_vert_batch[0, 0, :])\nprint(\"Clicked Subvert Batch shape:\", clicked_subvert_batch.shape)\nprint(\"Clicked Subvert Batch first sample:\", clicked_subvert_batch[0, 0, :])\n\n# Select a random candidate news article from the news dictionary\ncandidate_news_id = random.choice(list(news_dict.keys()))\ncandidate = news_dict[candidate_news_id]\nprint(f\"\\nRandom candidate news ID: {candidate_news_id}\")\n\n# Prepare candidate features\ncandidate_title_batch = np.array(candidate[\"title\"]).reshape(1, 1, hparams.title_size)\ncandidate_body_batch = np.array(candidate[\"body\"]).reshape(1, 1, hparams.body_size)\ncandidate_vert_batch = np.array([[candidate[\"category\"]]]).reshape(1, 1, 1)\ncandidate_subvert_batch = np.array([[candidate[\"subcategory\"]]]).reshape(1, 1, 1)\n\n# Debug: Print candidate features details\nprint(\"Candidate Title Batch shape:\", candidate_title_batch.shape)\nprint(\"Candidate Title Batch sample:\", candidate_title_batch[0, 0, :])\nprint(\"Sum of candidate title vector:\", np.sum(candidate_title_batch[0, 0, :]))\nprint(\"Candidate Body Batch shape:\", candidate_body_batch.shape)\nprint(\"Candidate Body Batch sample:\", candidate_body_batch[0, 0, :])\nprint(\"Sum of candidate body vector:\", np.sum(candidate_body_batch[0, 0, :]))\nprint(\"Candidate Vert Batch shape:\", candidate_vert_batch.shape)\nprint(\"Candidate Vert Batch sample:\", candidate_vert_batch[0, 0, :])\nprint(\"Candidate Subvert Batch shape:\", candidate_subvert_batch.shape)\nprint(\"Candidate Subvert Batch sample:\", candidate_subvert_batch[0, 0, :])\n\n# Print scorer model summary to inspect the layers and architecture\nprint(\"\\nScorer Model Summary:\")\nscorer.summary()\n\n# Run the scorer model (assumes 'scorer' is already loaded from your trained NAML model)\nscore = scorer.predict([\n    clicked_title_batch,\n    clicked_body_batch,\n    clicked_vert_batch,\n    clicked_subvert_batch,\n    candidate_title_batch,\n    candidate_body_batch,\n    candidate_vert_batch,\n    candidate_subvert_batch\n])\n\n# Debug: Print raw model output\nprint(\"\\nRaw score from scorer model:\", score)\nprint(f\"Predicted Click Probability for news '{candidate_news_id}': {score.flatten()[0]:.4f}\")\n\n# Optional: If you want to inspect the pre-sigmoid (raw dot product) values,\n# you could try to create an intermediate model that outputs the penultimate layer.\n# Note: This depends on your model architecture and might require adjustments.\ntry:\n    # Assuming the last layer is the sigmoid activation and the one before is the raw dot product:\n    pre_sigmoid_model = keras.Model(inputs=scorer.input, outputs=scorer.layers[-2].output)\n    raw_dot_product = pre_sigmoid_model.predict([\n        clicked_title_batch,\n        clicked_body_batch,\n        clicked_vert_batch,\n        clicked_subvert_batch,\n        candidate_title_batch,\n        candidate_body_batch,\n        candidate_vert_batch,\n        candidate_subvert_batch\n    ])\n    print(\"Raw dot product (pre-sigmoid):\", raw_dot_product)\nexcept Exception as e:\n    print(\"Could not extract pre-sigmoid output. Error:\", e)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T23:15:25.283102Z","iopub.execute_input":"2025-03-10T23:15:25.283355Z","iopub.status.idle":"2025-03-10T23:15:31.408859Z","shell.execute_reply.started":"2025-03-10T23:15:25.283327Z","shell.execute_reply":"2025-03-10T23:15:31.408019Z"}},"outputs":[],"execution_count":null}]}