{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install polars\n!pip install recbole","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-27T19:24:09.845448Z","iopub.execute_input":"2022-12-27T19:24:09.846063Z","iopub.status.idle":"2022-12-27T19:24:40.268954Z","shell.execute_reply.started":"2022-12-27T19:24:09.845928Z","shell.execute_reply":"2022-12-27T19:24:40.267207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import logging\nfrom logging import getLogger\nfrom recbole.config import Config\nfrom recbole.data import create_dataset, data_preparation\nfrom recbole.model.sequential_recommender import GRU4Rec\nfrom recbole.trainer import Trainer\nfrom recbole.utils import init_seed, init_logger\nfrom recbole.utils.case_study import full_sort_topk\nfrom typing import List, Tuple\nimport numpy as np\nimport pandas as pd\nfrom collections import defaultdict\nimport torch\nfrom pydantic import BaseModel\nfrom recbole.data import create_dataset\nfrom recbole.data.dataset.sequential_dataset import SequentialDataset\nimport polars as pl\nfrom recbole.data.interaction import Interaction\nfrom recbole.model.sequential_recommender.sine import SINE\nfrom recbole.utils import get_model, init_seed\n\n\nclass ItemHistory(BaseModel):\n    sequence: List[str]\n    topk: int\n\nclass RecommendedItems(BaseModel):\n    score_list: List[float]\n    item_list: List[str]\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-12-27T19:24:40.271624Z","iopub.execute_input":"2022-12-27T19:24:40.273186Z","iopub.status.idle":"2022-12-27T19:24:43.801451Z","shell.execute_reply.started":"2022-12-27T19:24:40.273121Z","shell.execute_reply":"2022-12-27T19:24:43.800240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parameter_dict = {\n    'data_path': '/kaggle/input/otto-prepared-data',\n    'USER_ID_FIELD': 'session',\n    'ITEM_ID_FIELD': 'aid',\n    'TIME_FIELD': 'ts',\n    'user_inter_num_interval': \"[5,Inf)\",\n    'item_inter_num_interval': \"[5,Inf)\",\n    'load_col': {'inter': ['session', 'aid', 'ts']},\n    'train_neg_sample_args': None,\n    'epochs': 10,\n    'stopping_step':3,\n    'eval_batch_size': 1024,\n    'MAX_ITEM_LIST_LENGTH': 20,\n    'eval_args': {\n        'split': {'RS': [9, 1, 0]},\n        'group_by': 'user',\n        'order': 'TO',\n        'mode': 'full'}\n}\n","metadata":{"execution":{"iopub.status.busy":"2022-12-27T19:24:43.802811Z","iopub.execute_input":"2022-12-27T19:24:43.804258Z","iopub.status.idle":"2022-12-27T19:24:43.813612Z","shell.execute_reply.started":"2022-12-27T19:24:43.804216Z","shell.execute_reply":"2022-12-27T19:24:43.812305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config = Config(model='GRU4Rec', dataset='recbox_data', config_dict=parameter_dict)\ninit_logger(config)\nlogger = getLogger()\n\nc_handler = logging.StreamHandler()\nc_handler.setLevel(logging.INFO)\nlogger.addHandler(c_handler)\n\nlogger.info(config)","metadata":{"execution":{"iopub.status.busy":"2022-12-27T19:24:43.816589Z","iopub.execute_input":"2022-12-27T19:24:43.817673Z","iopub.status.idle":"2022-12-27T19:24:44.315406Z","shell.execute_reply.started":"2022-12-27T19:24:43.817618Z","shell.execute_reply":"2022-12-27T19:24:44.314720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = create_dataset(config)\ntrain_data, valid_data, test_data = data_preparation(config, dataset)\nmodel = GRU4Rec(config, train_data.dataset).to(config['device'])\ntrainer = Trainer(config, model)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-27T19:24:44.316514Z","iopub.execute_input":"2022-12-27T19:24:44.317222Z","iopub.status.idle":"2022-12-27T19:30:47.934097Z","shell.execute_reply.started":"2022-12-27T19:24:44.317184Z","shell.execute_reply":"2022-12-27T19:30:47.932666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_valid_score, best_valid_result = trainer.fit(train_data, valid_data)","metadata":{"execution":{"iopub.status.busy":"2022-12-27T19:30:47.936464Z","iopub.execute_input":"2022-12-27T19:30:47.936935Z","iopub.status.idle":"2022-12-27T19:34:44.757918Z","shell.execute_reply.started":"2022-12-27T19:30:47.936894Z","shell.execute_reply":"2022-12-27T19:34:44.756415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_data, valid_data, test_data\n","metadata":{"execution":{"iopub.status.busy":"2022-12-27T19:01:30.489196Z","iopub.execute_input":"2022-12-27T19:01:30.489628Z","iopub.status.idle":"2022-12-27T19:01:30.498342Z","shell.execute_reply.started":"2022-12-27T19:01:30.489593Z","shell.execute_reply":"2022-12-27T19:01:30.497199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pred_user_to_item(item_history: ItemHistory):\n    item_history_dict = item_history.dict()\n    item_sequence = item_history_dict[\"sequence\"]\n    item_length = len(item_sequence)\n    pad_length = MAX_ITEM  # pre-defined by recbole\n\n    padded_item_sequence = torch.nn.functional.pad(\n        torch.tensor(dataset.token2id(dataset.iid_field, item_sequence)),\n        (0, pad_length - item_length),\n        \"constant\",\n        0,\n    )\n\n    input_interaction = Interaction(\n        {\n            \"aid_list\": padded_item_sequence.reshape(1, -1),\n            \"item_length\": torch.tensor([item_length]),\n        }\n    )\n    scores = model.full_sort_predict(input_interaction.to(model.device))\n    scores = scores.view(-1, dataset.item_num)\n    scores[:, 0] = -np.inf  # pad item score -> -inf\n    topk_score, topk_iid_list = torch.topk(scores, item_history_dict[\"topk\"])\n\n    predicted_score_list = topk_score.tolist()[0]\n    predicted_item_list = dataset.id2token(\n        dataset.iid_field, topk_iid_list.tolist()\n    ).tolist()\n\n    recommended_items = {\n        \"score_list\": predicted_score_list,\n        \"item_list\": predicted_item_list,\n    }\n    return recommended_items\n","metadata":{"execution":{"iopub.status.busy":"2022-12-27T19:01:34.426039Z","iopub.execute_input":"2022-12-27T19:01:34.426485Z","iopub.status.idle":"2022-12-27T19:01:34.437779Z","shell.execute_reply.started":"2022-12-27T19:01:34.426446Z","shell.execute_reply":"2022-12-27T19:01:34.436351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pl.read_parquet('../input/otto-full-optimized-memory-footprint/test.parquet')\nsession_types = ['clicks', 'carts', 'orders']\ntest_session_AIDs = test.to_pandas().reset_index(drop=True).groupby('session')['aid'].apply(list)\ntest_session_types = test.to_pandas().reset_index(drop=True).groupby('session')['type'].apply(list)\ndel test\n","metadata":{"execution":{"iopub.status.busy":"2022-12-27T19:04:14.005834Z","iopub.execute_input":"2022-12-27T19:04:14.006235Z","iopub.status.idle":"2022-12-27T19:05:32.826328Z","shell.execute_reply.started":"2022-12-27T19:04:14.006205Z","shell.execute_reply":"2022-12-27T19:05:32.824862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = []\n\ntype_weight_multipliers = {0: 1, 1: 6, 2: 3}\nfor AIDs, types in zip(test_session_AIDs, test_session_types):\n    if len(AIDs) >= 20:\n        weights=np.logspace(0.1,1,len(AIDs),base=2, endpoint=True)-1\n        aids_temp=defaultdict(lambda: 0)\n        for aid,w,t in zip(AIDs,weights,types): \n            aids_temp[aid]+= w * type_weight_multipliers[t]\n            \n        sorted_aids=[k for k, v in sorted(aids_temp.items(), key=lambda item: -item[1])]\n        labels.append(sorted_aids[:20])\n    else:\n        AIDs = list(dict.fromkeys(AIDs))\n        item = ItemHistory(sequence=AIDs, topk=20)\n        try:\n            nns = [ int(v) for v in pred_user_to_item(item)['item_list']]\n        except:\n            nns = []\n\n        for word in nns:\n            if len(AIDs) == 20:\n                break\n            if int(word) not in AIDs:\n                AIDs.append(word)\n\n        labels.append(AIDs[:20])\n","metadata":{"execution":{"iopub.status.busy":"2022-12-27T19:05:32.828786Z","iopub.execute_input":"2022-12-27T19:05:32.829133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_as_strings = [' '.join([str(l) for l in lls]) for lls in labels]\npredictions = pd.DataFrame(data={'session_type': test_session_AIDs.index, 'labels': labels_as_strings})\nlabels_as_strings = [' '.join([str(l) for l in lls]) for lls in labels]\n\npredictions = pd.DataFrame(data={'session_type': test_session_AIDs.index, 'labels': labels_as_strings})\n\nprediction_dfs = []\n\nfor st in session_types:\n    modified_predictions = predictions.copy()\n    modified_predictions.session_type = modified_predictions.session_type.astype('str') + f'_{st}'\n    prediction_dfs.append(modified_predictions)\n\nsubmission = pd.concat(prediction_dfs).reset_index(drop=True)\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}