{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Install & Import","metadata":{}},{"cell_type":"code","source":"!pip install recbole","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport gc","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import logging\nfrom logging import getLogger\nfrom recbole.config import Config\nfrom recbole.data import create_dataset, data_preparation\nfrom recbole.model.sequential_recommender import GRU4RecF, FDSA, BERT4Rec, GRU4Rec#, SASRecF\nfrom recbole.trainer import Trainer\nfrom recbole.utils import init_seed, init_logger\n\nimport random\n\nimport torch\nfrom torch import nn\n\nfrom recbole.model.abstract_recommender import SequentialRecommender\nfrom recbole.model.layers import FeedForward\n# from recbole.model.layers import FeatureSeqEmbLayer\n\nfrom recbole.utils import FeatureType\nimport copy\nimport math\n\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as fn\nfrom torch.nn.init import normal_\n\nfrom recbole.utils import FeatureType, FeatureSource\nimport torch.nn.functional as F\nfrom recbole.data.interaction import Interaction","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Create atomic files for Recbole training","metadata":{}},{"cell_type":"markdown","source":"These datasets are all publicly available on kaggle. ","metadata":{}},{"cell_type":"code","source":"!mkdir /kaggle/working/hm_atomic_interation_with_item_feature\n# inter = pd.read_csv('../input/hm-atomic-interation-with-item-feature/hm_atomic_interation_with_item_feature.inter', sep='\\t')\n\ninter = pd.read_csv('../input/reduced-inter/recbox_data_post2020.inter', sep='\\t')\n# inter = inter[inter['timestamp:float'] > 1589620000 ]# 1595620000\ninter.to_csv('/kaggle/working/hm_atomic_interation_with_item_feature/hm_atomic_interation_with_item_feature.inter', index=False, sep='\\t')\ndel inter\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# item = pd.read_csv('../input/bertembedding/out_bert_embed.csv')\n# item = pd.read_csv('../input/tfidf-embedding/out_2.csv')\nitem = pd.read_csv('../input/feature-bert-embed/bert_embed_feature.csv')\nitem = item.rename(columns={'article_id':'item_id:token', 'embed': 'item_emb:float_seq'})\nprint(item.head())\nprint(item.shape)\nitem.to_csv('/kaggle/working/hm_atomic_interation_with_item_feature/hm_atomic_interation_with_item_feature.item', index=False, sep='\\t')\ndel item\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create and train Recbole model","metadata":{}},{"cell_type":"markdown","source":"## Model training\n\nThis part trains the model","metadata":{}},{"cell_type":"code","source":"parameter_dict = {\n    'data_path': '/kaggle/working',\n    'USER_ID_FIELD': 'user_id',\n    'ITEM_ID_FIELD': 'item_id',\n    'TIME_FIELD': 'timestamp',\n    'user_inter_num_interval': \"[40,inf)\",\n    'item_inter_num_interval': \"[40,inf)\",\n    'load_col': {'inter': ['user_id', 'item_id', 'timestamp'],\n                  'item': ['item_id', 'item_emb']\n             },\n    'selected_features': ['item_emb'],\n    'neg_sampling': None,\n    'epochs': 1,\n#     'train_batch_size': 256,\n    'n_layers': 2,\n    'n_heads': 2,\n    'hidden_size': 64,\n    'inner_size': 256,\n    'hidden_dropout_prob': 0.5,\n    'attn_dropout_prob': 0.5,\n    'hidden_act': 'gelu',\n    'layer_norm_eps': 1e-12,\n    'initializer_range': 0.02,\n    'mask_ratio': 0.2,\n    'loss_type': 'CE',\n    'learning_rate': 0.002,\n    'pooling_mode': 'sum',\n    'eval_args': {\n        'split': {'RS': [10, 0, 0]},\n        'group_by': 'user',\n        'order': 'TO',\n        'mode': 'full'}\n}\n\nconfig = Config(model=\"BERT4Rec\", dataset='hm_atomic_interation_with_item_feature', config_dict=parameter_dict)\n\n# init random seed\ninit_seed(config['seed'], config['reproducibility'])\n\n# logger initialization\ninit_logger(config)\nlogger = getLogger()\n# Create handlers\nc_handler = logging.StreamHandler()\nc_handler.setLevel(logging.INFO)\nlogger.addHandler(c_handler)\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = create_dataset(config)\nlogger.info(dataset)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data, valid_data, test_data = data_preparation(config, dataset)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # model loading and initialization\nmodel = BERT4Rec(config, train_data.dataset).to(config['device'])\nlogger.info(model)\n\n# trainer loading and initialization\ntrainer = Trainer(config, model)\n\n# model training\nbest_valid_score, best_valid_result = trainer.fit(train_data)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The following commented code chunk was used for loading trained models.","metadata":{}},{"cell_type":"code","source":"# model_file = \"../input/onehot-bert-m/BERT4RecF-Apr-18-2022_02-18-35.pth\"\n# checkpoint = torch.load(model_file)\n# config = checkpoint['config']\n# init_seed(config['seed'], config['reproducibility'])\n# init_logger(config)\n# logger = getLogger()\n# logger.info(config)\n# model = BERT4RecF(config, train_data.dataset).to(config['device'])\n# model.load_state_dict(checkpoint['state_dict'])\n# model.load_other_parameter(checkpoint.get('other_parameter'))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Combine models\n\nThis part makes predictions and fills out the \"cold-start\" ones with 12 most frequence items.","metadata":{}},{"cell_type":"code","source":"from recbole.utils.case_study import full_sort_topk\nfrom recbole.quick_start.quick_start import load_data_and_model\n# config, model, dataset, train_data, valid_data, test_data = load_data_and_model(\n#     model_file='/kaggle/working/saved/SASRecF-Apr-05-2022_20-56-46.pth',\n# )\nexternal_user_ids = dataset.id2token(\n    dataset.uid_field, list(range(dataset.user_num)))[1:]#fist element in array is 'PAD'(default of Recbole) ->remove it ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom recbole.data.interaction import Interaction\n\ndef add_last_item(old_interaction, last_item_id, max_len=50):\n    new_seq_items = old_interaction['item_id_list'][-1]\n    if old_interaction['item_length'][-1].item() < max_len:\n        new_seq_items[old_interaction['item_length'][-1].item()] = last_item_id\n    else:\n        new_seq_items = torch.roll(new_seq_items, -1)\n        new_seq_items[-1] = last_item_id\n    return new_seq_items.view(1, len(new_seq_items))\n\ndef predict_for_all_item(external_user_id, dataset, model):\n    model.eval()\n    with torch.no_grad():\n        uid_series = dataset.token2id(dataset.uid_field, [external_user_id])\n        index = np.isin(dataset.inter_feat[dataset.uid_field].numpy(), uid_series)\n        input_interaction = dataset[index]\n        test = {\n            'item_id_list': add_last_item(input_interaction, \n                                          input_interaction['item_id'][-1].item(), model.max_seq_length),\n            'item_length': torch.tensor(\n                [input_interaction['item_length'][-1].item() + 1\n                 if input_interaction['item_length'][-1].item() < model.max_seq_length else model.max_seq_length])\n        }\n        new_inter = Interaction(test)\n        new_inter = new_inter.to(config['device'])\n        new_scores, attention = model.full_sort_predict(new_inter)\n        new_scores = new_scores.view(-1, test_data.dataset.item_num)\n        new_scores[:, 0] = -np.inf  # set scores of [pad] to -inf\n    return torch.topk(new_scores, 12)[1], attention","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"topk_items = []\nfor external_user_id in external_user_ids[112:]:\n    topk_iid_list, attention = predict_for_all_item(external_user_id, dataset, model)\n    last_topk_iid_list = topk_iid_list[-1]\n    external_item_list = dataset.id2token(dataset.iid_field, last_topk_iid_list.cpu()).tolist()\n    topk_items.append(external_item_list)\nprint(len(topk_items))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"external_item_str = [' '.join(x) for x in topk_items]\nresult = pd.DataFrame(external_user_ids, columns=['customer_id'])\nresult['prediction'] = external_item_str\nresult.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del external_item_str\ndel topk_items\ndel external_user_ids\ndel train_data\ndel valid_data\ndel test_data\ndel model\ndel Trainer\ndel logger\ndel dataset\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reference = pd.read_csv('../input/uid-reference/reference.csv')\nreference.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result.customer_id = result.customer_id.astype('int64')\nresult.dtypes","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_result = pd.merge(result, reference, how='left', left_on='customer_id', right_on='new_id', indicator=False, suffixes=(\"_x\", \"\")).drop(columns=['customer_id_x', 'new_id'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_result = new_result[['customer_id', 'prediction']]\nnew_result.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit_df = pd.read_csv('../input/cold-start/submission.csv')\nsubmit_df = pd.merge(submit_df, new_result, on='customer_id', how='outer')\nsubmit_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit_df = submit_df.fillna(-1)\nsubmit_df['prediction'] = submit_df.apply(\n    lambda x: x['prediction_y'] if x['prediction_y'] != -1 else x['prediction_x'], axis=1)\nsubmit_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit_df = submit_df.drop(columns=['prediction_y', 'prediction_x'])\nsubmit_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}