{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":4436180,"sourceType":"datasetVersion","datasetId":2597726},{"sourceId":7053627,"sourceType":"datasetVersion","datasetId":4059719}],"dockerImageVersionId":30301,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Code NN","metadata":{}},{"cell_type":"code","source":"!pip install polars","metadata":{"execution":{"iopub.status.busy":"2023-11-26T00:54:05.710063Z","iopub.execute_input":"2023-11-26T00:54:05.711156Z","iopub.status.idle":"2023-11-26T00:54:20.023945Z","shell.execute_reply.started":"2023-11-26T00:54:05.711030Z","shell.execute_reply":"2023-11-26T00:54:20.022840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.auto import tqdm\nimport pandas as pd\nimport numpy as np\nimport torch\nfrom torch import nn\nimport glob\nfrom argparse import ArgumentParser\nimport os\nimport polars as pl","metadata":{"execution":{"iopub.status.busy":"2023-11-26T00:54:20.026123Z","iopub.execute_input":"2023-11-26T00:54:20.026456Z","iopub.status.idle":"2023-11-26T00:54:21.878789Z","shell.execute_reply.started":"2023-11-26T00:54:20.026424Z","shell.execute_reply":"2023-11-26T00:54:21.877684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"version = \"version_2\"\nclass CFG:\n    train_path = \"/kaggle/input/otto-chunk-data-inparquet-format/train_parquet/000000000_000100000.parquet\"\n    val_path = \"/kaggle/input/otto-chunk-data-inparquet-format/test_parquet/000000000_000100000.parquet\"\n    output_path = \"./models\"\n    output_fn = f\"aid_h_emb_{version}.pkl\"\n    output_model_fn = f\"model_emb_{version}.pkl\"\n    n_epoch = 10 \n    n_length = 10  # input length\n    n_feat = 2  # num of features except for aid\n    temperature = 0.05\n    batch_size = 2000 #2000 * max([torch.cuda.device_count(), 1])\n    n_topk = 50 #200 # on inference, how many topk samples are saved\n    emb_dim =  128 #500\n    time_emb_dim = 32 #50\n    n_label = 10  # label length\n    lr = 0.001 * max([torch.cuda.device_count(), 1])\n    n_aid = 1855612\n    h_min = 460918\n    h_max = 461757 + 3\n    d_min = 19204\n    d_max = 19239 + 3\n    n_neg_coef = 30 #30 # n_neg_coef * batch_size is the number of negative aids \n    neg_topk = 500 #1000 # the number of negative aids which is used for train\n    run_test = False","metadata":{"execution":{"iopub.status.busy":"2023-11-26T04:44:24.872361Z","iopub.execute_input":"2023-11-26T04:44:24.872752Z","iopub.status.idle":"2023-11-26T04:44:24.881110Z","shell.execute_reply.started":"2023-11-26T04:44:24.872722Z","shell.execute_reply":"2023-11-26T04:44:24.880126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if CFG.run_test:\n    CFG.n_epoch = 1\n\n# columns used as inputs\nCFG.x_cols  = [f\"aid_pre{i}\" for i in range(CFG.n_length)] \nCFG.x_cols += [f\"h_pre{i}\" for i in range(CFG.n_length)]\nCFG.x_cols += [f\"weight_pre{i}\" for i in range(CFG.n_length)] # type of aid\n\n# columns used as targets\nCFG.y_cols  = [f\"aid_label{i}\" for i in range(CFG.n_label)]\nCFG.y_cols += [f\"weight_label{i}\" for i in range(CFG.n_label)] # type of aid","metadata":{"execution":{"iopub.status.busy":"2023-11-26T04:44:27.936903Z","iopub.execute_input":"2023-11-26T04:44:27.937554Z","iopub.status.idle":"2023-11-26T04:44:27.944709Z","shell.execute_reply.started":"2023-11-26T04:44:27.937518Z","shell.execute_reply":"2023-11-26T04:44:27.943624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(CFG.x_cols)","metadata":{"execution":{"iopub.status.busy":"2023-11-26T04:44:28.245910Z","iopub.execute_input":"2023-11-26T04:44:28.246846Z","iopub.status.idle":"2023-11-26T04:44:28.251272Z","shell.execute_reply.started":"2023-11-26T04:44:28.246805Z","shell.execute_reply":"2023-11-26T04:44:28.250373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(CFG.y_cols)","metadata":{"execution":{"iopub.status.busy":"2023-11-26T04:44:28.788120Z","iopub.execute_input":"2023-11-26T04:44:28.789024Z","iopub.status.idle":"2023-11-26T04:44:28.794024Z","shell.execute_reply.started":"2023-11-26T04:44:28.788988Z","shell.execute_reply":"2023-11-26T04:44:28.792978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def metric_smoothing(mets, mets_smooth):\n    \"\"\"\n    utility for displaying metrics during training\n    Receive the value of the metric at that time,\n    returns the value of metric after smoothing\n    \"\"\"\n    assert(len(mets) == len(mets_smooth))\n    metric_smooth_coef = 0.99\n    met_str = []\n    for k, (key, value) in enumerate(mets_smooth.items()):\n        try:\n            v = mets[k].item()\n        except:\n            v = mets[k]\n        if mets_smooth[key] is None:\n            mets_smooth[key] = v\n        else:\n            mets_smooth[key] = mets_smooth[key] * metric_smooth_coef + v * (1 - metric_smooth_coef)\n        met_str.append(\"{} : {:.4f} \".format(key, mets_smooth[key]))\n    met_str = \", \".join(met_str)\n    return met_str, mets_smooth\n\ndef preprocess_df(df, with_label):\n    \"\"\"\n    Format data to suit nn input\n    output is Dataframe with CFG.x_cols + CFG.y_cols\n    \"\"\"\n    df = pl.from_pandas(df)\n    df = df.sort([\"session\", \"ts\"])\n    n_length = CFG.n_length\n    n_label = CFG.n_label\n    n_aid = CFG.n_aid\n    assert(n_aid > df[\"aid\"].max()+3)\n\n    # mapping : type string -> weight int\n    weights = {\"clicks\" : 1, \"carts\" : 3, \"orders\" : 6}\n    mapper = pl.DataFrame({\n        \"type\" : list(weights.keys()),\n        \"weight\": list(weights.values())\n    })\n    df = df.join(mapper, on = \"type\")\n\n    # hour information from ts\n    df = df.with_columns((pl.col(\"ts\")//3600//1000).alias(\"h\"))\n    df = df.with_columns((pl.col(\"h\") - CFG.h_min).alias(\"h\"))\n\n    # shift features to get past values\n    exprs = []\n    for i in range(0,n_length):\n        exprs += [\n            pl.col(\"aid\").shift(i).alias(f\"aid_pre{i}\"),\n            pl.col(\"h\").shift(i).alias(f\"h_pre{i}\"),\n            pl.col(\"weight\").shift(i).alias(f\"weight_pre{i}\"),\n            pl.col(\"session\").shift(i).alias(f\"session_pre{i}\"),\n        ]\n    df = df.with_columns(exprs)\n    def get_replace(name, rep, condition):\n        return pl.when(\n            condition).then(\n            pl.lit(rep)\n        ).otherwise(pl.col(name)).alias(name)\n\n    # If the session id is different before and after shifting, replace features to pre defined non informative values\n    # aid -> n_aid - 1\n    # h -> CFG.h_max-CFG.h_min+3\n    # weight -> 0\n    conditions = [pl.col(\"session\") != pl.col(f\"session_pre{i}\") for i in range(n_length)]\n    df = df.with_columns(\n        [get_replace(f\"aid_pre{i}\", n_aid - 1, conditions[i]) for i in range(n_length)] + \\\n        [get_replace(f\"h_pre{i}\", CFG.h_max-CFG.h_min+3, conditions[i]) for i in range(n_length)] + \\\n        [get_replace(f\"weight_pre{i}\", 0, conditions[i]) for i in range(n_length)]\n    )\n\n    # for training\n    if with_label:\n        # get aids and weights in future as labels\n        for i in range(0, n_label):\n            exprs = [\n                pl.col([\"aid\"]).shift(-1-i).alias(f\"aid_label{i}\"),     \n                pl.col([\"weight\"]).shift(-1-i).alias(f\"weight_label{i}\")\n            ]\n            # if session is disagreement, replace labels with predefined values\n            condition = pl.col(\"session\") != pl.col(\"session\").shift(-1-i)\n            exprs_rep = [\n                get_replace(f\"aid_label{i}\", n_aid - 1, condition).fill_null(n_aid - 1),\n                get_replace(f\"weight_label{i}\", 0, condition).fill_null(0),\n            ]\n            df = df.with_columns(exprs).with_columns(exprs_rep)\n\n        # \"aid_label0 != n_aid -1\" indicates that there is at least one label\n        condition = pl.col(\"aid_label0\") != n_aid - 1\n        return torch.LongTensor(df.filter(condition)[CFG.x_cols + CFG.y_cols].to_numpy())\n    return df.to_pandas()\n\ndef train(model, opt, xy):\n    \"\"\"\n    train nn model for one epoch\n\n    model : pytorch model\n    opt : optimizer of the model\n    xy : pytorch tensor containing inputs and labels\n    \"\"\"\n    mets_smooth = {\"loss\" : None, \"acc\" : None}\n    model.train()\n    batch_size = CFG.batch_size\n    batch_indices = np.array_split(np.random.permutation(len(xy)), len(xy)//batch_size + 1)\n    i = 0\n    mets_smooth_acc_rec = [0]\n    train_end_flag = False\n    for indices in tqdm(batch_indices):  # training loop\n        xy_batch = xy[indices]\n        outs = model(xy_batch)\n        outs = [torch.mean(l) for l in outs]\n        loss, acc = outs\n        model.zero_grad()\n        loss.backward()\n        opt.step()\n\n        i+=1\n        mets_str, mets_smooth = metric_smoothing([loss,\n                                                  acc,\n                                                 ], mets_smooth)\n        if i%100 == 0:\n            print(i, mets_str)\n            if CFG.run_test & (i > 1000):\n                break\n        if i%2000 == 0:\n            # if accuracy dosent improve, stop training or reduce learning rate\n            if mets_smooth[\"acc\"] < mets_smooth_acc_rec[-1]:\n                train_end_flag = opt.step_scheduler()\n                if train_end_flag:\n                    break\n            mets_smooth_acc_rec.append(mets_smooth[\"acc\"])            \n            print(\"acc improve {} -> {}\".format(mets_smooth_acc_rec[-2], mets_smooth_acc_rec[-1]))\n    return model, train_end_flag\n\ndef inference(model, is_inference_all):\n    \"\"\"\n    inference with a given model\n    \"\"\"\n\n    print(\"start inference\")\n    model = model.eval()\n    n_topk = CFG.n_topk\n    if is_inference_all:\n        path = CFG.val_path\n        print(\"inference on all data\")\n    else:\n        path = glob.glob(CFG.val_path + \"/*\")[0]\n        print(\"inference on partial data\")\n    df_inf = pd.read_parquet(path)\n    df_inf = preprocess_df(df_inf, with_label=False)\n    df_inf = df_inf.groupby(\"session\").last() # get last row of each session\n\n    x_cols = CFG.x_cols\n    sessions = np.array(list(df_inf.index))\n    x_tensor = torch.LongTensor(df_inf[x_cols].values)\n\n    ## inference\n    batch_size = 50\n    weights = {\"clicks\" : 1, \"carts\" : 3, \"orders\" : 6}\n    aid_cands = torch.arange(CFG.n_aid, device = device)\n    # prepare aid embedding in advance\n    y_em = model.do_emb(aid_cands).detach()\n    y_em = y_em/torch.norm(y_em, dim=1, keepdim = True)\n    batch_indices = np.array_split(np.random.permutation(len(x_tensor)), len(x_tensor)//batch_size + 1)\n    for label, weight in weights.items(): # predict for each type\n        label_type = (torch.zeros(batch_size*2, device = device) + weight).long()\n        recs = []\n        for indices in tqdm(batch_indices):\n            x_batch = x_tensor[indices].to(device)\n            # session embedding for the label_type\n            x_em = model.calc_query_emb(x_batch, label_type[:len(x_batch)])\n            x_em = x_em.detach()\n            x_em = x_em/torch.norm(x_em, dim=1, keepdim = True)\n\n            # cosine similarity between session embedding and aid embedding\n            cossim = torch.matmul(x_em, y_em.T)\n\n            # get topk aids and scores\n            scores, topk = torch.topk(cossim, n_topk, axis = 1)\n            scores = scores.cpu().numpy()\n            scores = (scores * 1000).astype(int)\n            topk = topk.cpu().numpy()    \n            ses = sessions[indices]\n            recs.append([ses, topk, scores])\n\n            if CFG.run_test:\n                if len(recs) > 5:\n                    break\n\n        # concat inference results\n        sessions_save = np.hstack([r[0] for r in recs])\n        topks = np.vstack([r[1] for r in recs])\n        scores = np.vstack([r[2] for r in recs])\n\n        # save inference results\n        lis = []\n        dic = {}\n        lis = np.zeros([len(sessions_save), n_topk*2], dtype = np.int32)\n        for i in tqdm(range(len(sessions_save))):\n             # first n_topk elements are topk aids, and last n_topk elements are cosine similarity score\n            lis[i] = list(topks[i]) + list(scores[i])\n            # mapping dict session_id to index\n            dic[sessions_save[i]] = i\n        arr = np.array(lis).astype(np.uint32)\n        # save\n        os.system(f\"mkdir -p {CFG.output_path}\")\n        dst = CFG.output_path + CFG.output_fn.replace(\"emb\", f\"emb_{label}\")\n        # save matrix containing topk_aids and cosine similarities, and mapper of session to idx\n        pd.to_pickle([arr, dic], dst)\n        print(\"save! : \", dst)\n        import gc; gc.collect()\n\n\nclass Encoder(torch.nn.Module):\n    \"\"\"\n    nn model \n    \"\"\"\n    def __init__(self):\n        super(Encoder, self).__init__()\n        # embedding of aids\n        self.emb = torch.nn.Embedding(CFG.n_aid, CFG.emb_dim, padding_idx=CFG.n_aid-1)\n        # embedding of time\n        self.emb_h = torch.nn.Embedding(CFG.h_max-CFG.h_min+4, CFG.time_emb_dim, padding_idx=CFG.h_max-CFG.h_min+3)\n\n        inp_dim = CFG.emb_dim * CFG.n_length + (CFG.n_feat + 1 + CFG.time_emb_dim) * CFG.n_length\n        self.mlp = torch.nn.Sequential(\n            torch.nn.BatchNorm1d(inp_dim),\n            torch.nn.Linear(inp_dim,inp_dim),\n            torch.nn.GELU(),\n            torch.nn.BatchNorm1d(inp_dim),\n            torch.nn.Linear(inp_dim,inp_dim//2),\n            torch.nn.GELU(),\n            torch.nn.BatchNorm1d(inp_dim//2),\n            torch.nn.Linear(inp_dim//2,CFG.emb_dim)\n        )\n        nn.init.uniform_(self.emb.weight, -1.0, 1.0)\n        nn.init.uniform_(self.emb_h.weight, -1.0, 1.0)\n        \n        # embedding of type\n        self.emb_label_type = torch.nn.Embedding(7, CFG.emb_dim, padding_idx=0)\n        \n        \n    def forward(self, xy_batch):\n        \"\"\"\n        receives batch of x and y,\n        returns loss and accuracy\n        \"\"\"\n\n        # split data to x, y, and label_type\n        x_batch = xy_batch[:,:-CFG.n_label*2]\n        y_batch = xy_batch[:,-CFG.n_label*2:-CFG.n_label]\n        label_type = xy_batch[:,-CFG.n_label:]\n\n        # sampling negative aids\n        neg_aids = torch.randperm(CFG.n_aid, device = xy_batch.get_device())[:CFG.n_neg_coef * len(xy_batch)]\n\n        # calculate session embedding\n        x_em = self._calc_query_emb(x_batch) # (batch_size, CFG.emb_dim)\n        label_type_em = self.emb_label_type(label_type) # (batch_size, CFG.n_label, CFG.emb_dim)\n        x_em = x_em.unsqueeze(1) + label_type_em # (batch_size, CFG.n_label, CFG.emb_dim)\n\n        # calculate target aid embeddings\n        y_em = self.do_emb(y_batch) # (batch_size, CFG.n_label, CFG.emb_dim)\n        neg_em = self.do_emb(neg_aids) # (n_negative_sample, CFG.emb_dim)\n\n        # normalize emebddings\n        x_em = x_em/torch.norm(x_em, dim=2, keepdim = True)\n        y_em = y_em/torch.norm(y_em, dim=2, keepdim = True)\n        neg_em = neg_em/torch.norm(neg_em, dim=1, keepdim = True)\n\n        # cosine similarity with positive aids\n        cossim_pos_all = torch.sum(x_em * y_em, dim = 2) # (batch_size, CFG.n_label)\n        y_mask = y_batch.eq(CFG.n_aid - 1)\n        cossim_pos_min = torch.min(cossim_pos_all + y_mask, dim = 1)[0] # cosine similarity of most difficult positive sample\n        cossim_pos_mean = torch.sum(cossim_pos_all * label_type, dim = 1) / torch.sum(label_type, dim = 1)\n        cossim_pos = (cossim_pos_mean + cossim_pos_min) / 2 # (batch_size, )\n\n        # cosine similarity with negative aids\n        cossim_neg = torch.matmul(x_em[:,0], neg_em.T) # (batch_size, n_negative_sample)\n        cossim_neg = torch.topk(cossim_neg, CFG.neg_topk, dim = 1)[0] # get difficult negative samples\n\n        # calculate loss and accuracy\n        cossim = torch.cat([cossim_pos.unsqueeze(1), cossim_neg], dim = 1) / CFG.temperature\n        p = torch.softmax(cossim,dim=1)\n        loss = - torch.mean(torch.log(p[:,0])) # first sample in each batch is positive\n        acc = torch.mean((torch.max(p,dim=1)[1] == 0).to(torch.float))\n        return loss, acc\n\n        \n    def _calc_query_emb(self, x_batch):\n        \"\"\"\n        calculate session embedding\n        \"\"\"\n        x_batch = x_batch.reshape(-1, CFG.n_feat + 1, CFG.n_length).transpose(1,2) ## (batch_size, n_length, n_feat)\n\n        # embedding from aids\n        em = self.emb(x_batch[:,:,0])\n\n        # other features like time in a day and type of aid\n        x_other = torch.cat([\n            torch.sin(x_batch[:,:,1:2] % 24 / 24 * np.pi * 2),\n            torch.cos(x_batch[:,:,1:2] % 24 / 24 * np.pi * 2),\n            x_batch[:,:,2:],\n        ], dim=2)\n\n        # embedding from hour\n        emh = self.emb_h(x_batch[:,:,1]) + \\\n              self.emb_h(torch.clip(x_batch[:,:,1]+1, max = CFG.h_max-CFG.h_min+3)) + \\\n              self.emb_h(torch.clip(x_batch[:,:,1]+2, max = CFG.h_max-CFG.h_min+3))\n\n        # concat all embeddings\n        x = torch.cat([em, x_other, emh], dim = 2)\n\n        # masking where no history aid\n        mask = ~x_batch[:,:,0].eq(CFG.n_aid-1).unsqueeze(2)\n        x = x * mask\n        x = x.reshape(len(x), -1) # (batch_size, CFG.n_length * sum_dim_embeddings_and_features)\n        \n        # mlp to create session embedding\n        return self.mlp(x)  ## (batch_size, CFG.emb_dim)\n    \n    def calc_query_emb(self, x_batch, label_type):\n        \"\"\"\n        calculate session embedding to target type\n        calculate session embedding by self._calc_query_emb, and add target type embedding\n        \"\"\"\n        return self._calc_query_emb(x_batch) + self.emb_label_type(label_type)\n    \n    def do_emb(self, x):\n        return self.emb(x)\n    \nclass MyOptimizer():\n    \"\"\"\n    optimizer + lr scheduler\n    \"\"\"\n    def __init__(self, model):\n        params = []\n        params_emb = []\n        for name, param in model.named_parameters():\n            if param.requires_grad:\n                if name in [\"module.emb.weight\", \"module.emb_x.weight\"]:\n                    params_emb.append(param)\n                else:\n                    params.append(param)\n                    \n        self.opt = torch.optim.AdamW(params, lr = CFG.lr)\n        self.opt_em = torch.optim.AdamW(params_emb, lr = CFG.lr)\n        self.scheduler = torch.optim.lr_scheduler.StepLR(self.opt, step_size=1, gamma=0.5)\n        self.scheduler_em = torch.optim.lr_scheduler.StepLR(self.opt_em, step_size=1, gamma=0.5)\n        self.count = 0\n        \n    def step(self):\n        self.opt.step()\n        self.opt_em.step()\n        \n    def step_scheduler(self):\n        self.scheduler.step()\n        self.scheduler_em.step()\n        self.count += 1\n        print(\"lr half\", self.count)\n        if self.count > 4:\n            return True\n        else:\n            return False\n\n\ndef load_train():\n    df = pd.read_parquet(CFG.train_path)\n    df = df[df[\"ts\"] > (df[\"ts\"].max() - (3600 * 24 * 14 * 1000))]\n    if CFG.run_test:\n        df = df[:10000000]\n    return df\n\ndef load_test():\n    df = pd.read_parquet(CFG.val_path)\n    if CFG.run_test:\n        df = df[:10000000]\n    return df\n","metadata":{"execution":{"iopub.status.busy":"2023-11-26T04:44:29.819046Z","iopub.execute_input":"2023-11-26T04:44:29.819452Z","iopub.status.idle":"2023-11-26T04:44:29.918411Z","shell.execute_reply.started":"2023-11-26T04:44:29.819416Z","shell.execute_reply":"2023-11-26T04:44:29.917565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load data","metadata":{}},{"cell_type":"code","source":"class args:\n    pretrain = \"\" # if not empty, load model\n    is_skip_train = False #if true, skip training and conduct only inference\n    is_inference_all = False #if true, inference on all data. if false inference on partial data for local validation\n    \n\nif args.pretrain == \"at_train\":\n    # load latest model \n    model_path = \"\" #CFG.output_path + \"/last_\"+CFG.output_model_fn\nelif args.pretrain == \"at_test\":\n    # load latest model from training directory\n    model_path = os.path.join(CFG.output_path,CFG.output_model_fn) \nelif args.pretrain == \"\":\n    model_path = None\n\nif not args.is_skip_train:\n    xy = torch.cat([preprocess_df(load_train(), with_label=True), \n                    preprocess_df(load_test(), with_label=True)], dim = 0)\n    print(xy.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-26T04:44:31.901595Z","iopub.execute_input":"2023-11-26T04:44:31.901983Z","iopub.status.idle":"2023-11-26T04:44:38.241903Z","shell.execute_reply.started":"2023-11-26T04:44:31.901950Z","shell.execute_reply":"2023-11-26T04:44:38.240789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### create model","metadata":{}},{"cell_type":"code","source":"# create model\nif model_path is None:\n    model = Encoder()\nelse:\n    print(\"load model from \", model_path)\n    model = pd.read_pickle(model_path).module\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n # make parallel\nmodel = torch.nn.DataParallel(model)\ntorch.backends.cudnn.benchmark = True\nmodel.do_emb = model.module.do_emb\nmodel.calc_query_emb = model.module.calc_query_emb\nmodel = model.to(device)","metadata":{"execution":{"iopub.status.busy":"2023-11-26T04:44:38.243561Z","iopub.execute_input":"2023-11-26T04:44:38.243875Z","iopub.status.idle":"2023-11-26T04:44:42.480329Z","shell.execute_reply.started":"2023-11-26T04:44:38.243845Z","shell.execute_reply":"2023-11-26T04:44:42.479393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-11-26T12:09:54.283992Z","iopub.execute_input":"2023-11-26T12:09:54.284994Z","iopub.status.idle":"2023-11-26T12:09:54.305626Z","shell.execute_reply.started":"2023-11-26T12:09:54.284935Z","shell.execute_reply":"2023-11-26T12:09:54.304168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### train","metadata":{}},{"cell_type":"code","source":"# if not args.is_skip_train:\n#     opt = MyOptimizer(model)\n#     os.system(f\"mkdir -p {CFG.output_path}\")\n#     print(\"start training\")\n#     for i in range(CFG.n_epoch):\n#         model, train_end_flag = train(model, opt, xy)\n#         if train_end_flag:\n#             break\n#     pd.to_pickle(model.cpu(), CFG.output_path + \"/\" + \"last\"+\"_\"+CFG.output_model_fn)\n#     model = model.to(device)\n\n# inference(model, args.is_inference_all)","metadata":{"execution":{"iopub.status.busy":"2023-11-25T09:28:53.374140Z","iopub.execute_input":"2023-11-25T09:28:53.374494Z","iopub.status.idle":"2023-11-25T09:28:53.379643Z","shell.execute_reply.started":"2023-11-25T09:28:53.374464Z","shell.execute_reply":"2023-11-25T09:28:53.378418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# load data từng phần và train","metadata":{}},{"cell_type":"code","source":"import os\n\nopt = MyOptimizer(model)\nos.system(f\"mkdir -p {CFG.output_path}\")\n\nlist_data = os.listdir('/kaggle/input/otto-chunk-data-inparquet-format/train_parquet')\nfor i in range(61, len(list_data)):\n    if i %\n    file_name = list_data[i]\n    if not args.is_skip_train:\n        CFG.train_path = os.path.join('/kaggle/input/otto-chunk-data-inparquet-format/train_parquet',file_name)\n        xy = preprocess_df(load_train(), with_label=True)\n        print(f\"Start training on part {i} : {file_name}\")\n        \n        for i in range(CFG.n_epoch):\n            model, train_end_flag = train(model, opt, xy)\n            if train_end_flag:\n                break\n    if i % 30 == 0:\n        pd.to_pickle(model.cpu(), CFG.output_path + \"/\" + f\"last_{i}\"+\"_\"+CFG.output_model_fn)\n        model = model.to(device)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.to_pickle(model.cpu(), CFG.output_path + \"/\" + f\"last_61\"+\"_\"+CFG.output_model_fn )\nmodel = model.to(device)","metadata":{"execution":{"iopub.status.busy":"2023-11-26T12:13:22.651505Z","iopub.execute_input":"2023-11-26T12:13:22.652599Z","iopub.status.idle":"2023-11-26T12:13:26.287279Z","shell.execute_reply.started":"2023-11-26T12:13:22.652556Z","shell.execute_reply":"2023-11-26T12:13:26.286150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(list_data))","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:03:52.428485Z","iopub.execute_input":"2023-11-25T17:03:52.429418Z","iopub.status.idle":"2023-11-25T17:03:52.435114Z","shell.execute_reply.started":"2023-11-25T17:03:52.429370Z","shell.execute_reply":"2023-11-25T17:03:52.434116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(list_data)):\n    if list_data[i] == '000800000_000900000.parquet':\n        print(i)","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:04:41.674105Z","iopub.execute_input":"2023-11-25T17:04:41.674902Z","iopub.status.idle":"2023-11-25T17:04:41.680470Z","shell.execute_reply.started":"2023-11-25T17:04:41.674866Z","shell.execute_reply":"2023-11-25T17:04:41.679449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# explore preprocess_df","metadata":{}},{"cell_type":"code","source":"df = load_test()","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.898830Z","iopub.status.idle":"2023-11-25T17:00:55.899200Z","shell.execute_reply.started":"2023-11-25T17:00:55.899009Z","shell.execute_reply":"2023-11-25T17:00:55.899025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.901168Z","iopub.status.idle":"2023-11-25T17:00:55.901543Z","shell.execute_reply.started":"2023-11-25T17:00:55.901363Z","shell.execute_reply":"2023-11-25T17:00:55.901380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pl.from_pandas(df) ","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.902908Z","iopub.status.idle":"2023-11-25T17:00:55.903309Z","shell.execute_reply.started":"2023-11-25T17:00:55.903087Z","shell.execute_reply":"2023-11-25T17:00:55.903104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sort theo session -> time\ndf = df.sort([\"session\", \"ts\"])\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.904209Z","iopub.status.idle":"2023-11-25T17:00:55.904568Z","shell.execute_reply.started":"2023-11-25T17:00:55.904391Z","shell.execute_reply":"2023-11-25T17:00:55.904408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_length = CFG.n_length\nn_label = CFG.n_label\nn_aid = CFG.n_aid\nprint(n_length,n_label,n_aid)\nassert(n_aid > df[\"aid\"].max()+3)","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.905636Z","iopub.status.idle":"2023-11-25T17:00:55.906003Z","shell.execute_reply.started":"2023-11-25T17:00:55.905823Z","shell.execute_reply":"2023-11-25T17:00:55.905840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" # mapping : type string -> weight int\nweights = {\"clicks\" : 1, \"carts\" : 3, \"orders\" : 6}\nmapper = pl.DataFrame({\n    \"type\" : list(weights.keys()),\n    \"weight\": list(weights.values())\n})\ndf = df.join(mapper, on = \"type\")\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.907806Z","iopub.status.idle":"2023-11-25T17:00:55.908205Z","shell.execute_reply.started":"2023-11-25T17:00:55.907990Z","shell.execute_reply":"2023-11-25T17:00:55.908008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# hour information from ts\ndf = df.with_columns((pl.col(\"ts\")//3600//1000).alias(\"h\"))\ndf = df.with_columns((pl.col(\"h\") - CFG.h_min).alias(\"h\"))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.909673Z","iopub.status.idle":"2023-11-25T17:00:55.910057Z","shell.execute_reply.started":"2023-11-25T17:00:55.909857Z","shell.execute_reply":"2023-11-25T17:00:55.909874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shift features to get past values\nexprs = [] # list of expression\nfor i in range(0,n_length):\n    exprs += [\n        pl.col(\"aid\").shift(i).alias(f\"aid_pre{i}\"),\n        pl.col(\"h\").shift(i).alias(f\"h_pre{i}\"),\n        pl.col(\"weight\").shift(i).alias(f\"weight_pre{i}\"),\n        pl.col(\"session\").shift(i).alias(f\"session_pre{i}\"),\n    ]\ndf = df.with_columns(exprs)\ndf.tail()","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.911840Z","iopub.status.idle":"2023-11-25T17:00:55.912239Z","shell.execute_reply.started":"2023-11-25T17:00:55.912025Z","shell.execute_reply":"2023-11-25T17:00:55.912042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"exprs[10]","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.913766Z","iopub.status.idle":"2023-11-25T17:00:55.914129Z","shell.execute_reply.started":"2023-11-25T17:00:55.913949Z","shell.execute_reply":"2023-11-25T17:00:55.913966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_replace(name, rep, condition):\n        return pl.when(\n            condition).then(\n            pl.lit(rep)\n        ).otherwise(pl.col(name)).alias(name)","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.915675Z","iopub.status.idle":"2023-11-25T17:00:55.916014Z","shell.execute_reply.started":"2023-11-25T17:00:55.915844Z","shell.execute_reply":"2023-11-25T17:00:55.915860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### If the session id is different before and after shifting, replace features to pre defined non informative values\n# aid -> n_aid - 1\n# h -> CFG.h_max-CFG.h_min+3\n# weight -> 0\nconditions = [pl.col(\"session\") != pl.col(f\"session_pre{i}\") for i in range(n_length)]\ndf = df.with_columns(\n    [get_replace(f\"aid_pre{i}\", n_aid - 1, conditions[i]) for i in range(n_length)] + \\\n    [get_replace(f\"h_pre{i}\", CFG.h_max-CFG.h_min+3, conditions[i]) for i in range(n_length)] + \\\n    [get_replace(f\"weight_pre{i}\", 0, conditions[i]) for i in range(n_length)]\n)\ndf.tail()","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.917376Z","iopub.status.idle":"2023-11-25T17:00:55.917734Z","shell.execute_reply.started":"2023-11-25T17:00:55.917558Z","shell.execute_reply":"2023-11-25T17:00:55.917575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for training\nwith_label = True\nif with_label:\n    # get aids and weights in future as labels\n    for i in range(0, n_label): #lấy n_label event tiếp theo trong session đó làm label\n        exprs = [\n            pl.col([\"aid\"]).shift(-1-i).alias(f\"aid_label{i}\"),     \n            pl.col([\"weight\"]).shift(-1-i).alias(f\"weight_label{i}\")\n        ]\n        # if session is disagreement, replace labels with predefined values\n        condition = pl.col(\"session\") != pl.col(\"session\").shift(-1-i)\n        exprs_rep = [\n            get_replace(f\"aid_label{i}\", n_aid - 1, condition).fill_null(n_aid - 1),\n            get_replace(f\"weight_label{i}\", 0, condition).fill_null(0),\n        ]\n        df = df.with_columns(exprs).with_columns(exprs_rep)\n\n    # \"aid_label0 != n_aid -1\" indicates that there is at least one label\n    condition = pl.col(\"aid_label0\") != n_aid - 1\n    #return torch.LongTensor(df.filter(condition)[CFG.x_cols + CFG.y_cols].to_numpy())\n    res = torch.LongTensor(df.filter(condition)[CFG.x_cols + CFG.y_cols].to_numpy())","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.918827Z","iopub.status.idle":"2023-11-25T17:00:55.919201Z","shell.execute_reply.started":"2023-11-25T17:00:55.918991Z","shell.execute_reply":"2023-11-25T17:00:55.919006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(res.shape)\nprint(df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.920270Z","iopub.status.idle":"2023-11-25T17:00:55.920607Z","shell.execute_reply.started":"2023-11-25T17:00:55.920434Z","shell.execute_reply":"2023-11-25T17:00:55.920450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[0]","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.922263Z","iopub.status.idle":"2023-11-25T17:00:55.922605Z","shell.execute_reply.started":"2023-11-25T17:00:55.922435Z","shell.execute_reply":"2023-11-25T17:00:55.922451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(CFG.x_cols)\nprint(len(CFG.x_cols))\nprint(CFG.y_cols)\nprint(len(CFG.y_cols))","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.924188Z","iopub.status.idle":"2023-11-25T17:00:55.924532Z","shell.execute_reply.started":"2023-11-25T17:00:55.924361Z","shell.execute_reply":"2023-11-25T17:00:55.924377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(CFG.n_label)","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.925702Z","iopub.status.idle":"2023-11-25T17:00:55.926035Z","shell.execute_reply.started":"2023-11-25T17:00:55.925868Z","shell.execute_reply":"2023-11-25T17:00:55.925883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\n\n# Define the size of the vocabulary and the dimensionality of embeddings\nvocab_size = 10000  # Example vocabulary size\nembedding_dim = 300  # Example embedding dimension\n\n# Create an Embedding layer\nembedding_layer = nn.Embedding(vocab_size, embedding_dim)\n\n# Input indices (representing words or tokens)\ninput_indices = torch.LongTensor([1, 5, 9, 3])  # Example input indices\n\n# Get embeddings for the input indices\nembeddings = embedding_layer(input_indices)\n\nprint(embeddings)\n# This will print the embeddings corresponding to the input indices\n","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.927673Z","iopub.status.idle":"2023-11-25T17:00:55.928044Z","shell.execute_reply.started":"2023-11-25T17:00:55.927860Z","shell.execute_reply":"2023-11-25T17:00:55.927878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = torch.randn(4,100).unsqueeze(1)","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.929313Z","iopub.status.idle":"2023-11-25T17:00:55.929814Z","shell.execute_reply.started":"2023-11-25T17:00:55.929547Z","shell.execute_reply":"2023-11-25T17:00:55.929571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"b = torch.randn(4,10,100)","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.931995Z","iopub.status.idle":"2023-11-25T17:00:55.932503Z","shell.execute_reply.started":"2023-11-25T17:00:55.932240Z","shell.execute_reply":"2023-11-25T17:00:55.932265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"c = a+b","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.933796Z","iopub.status.idle":"2023-11-25T17:00:55.934326Z","shell.execute_reply.started":"2023-11-25T17:00:55.934036Z","shell.execute_reply":"2023-11-25T17:00:55.934062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"c.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-25T17:00:55.935871Z","iopub.status.idle":"2023-11-25T17:00:55.936378Z","shell.execute_reply.started":"2023-11-25T17:00:55.936102Z","shell.execute_reply":"2023-11-25T17:00:55.936125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}