{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q ../input/faiss-163/faiss_gpu-1.6.3-cp37-cp37m-manylinux2010_x86_64.whl\n# !pip install -q ../input/unidecode/Unidecode-1.3.4-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2022-07-05T11:58:19.268196Z","iopub.execute_input":"2022-07-05T11:58:19.268523Z","iopub.status.idle":"2022-07-05T11:58:50.613919Z","shell.execute_reply.started":"2022-07-05T11:58:19.268433Z","shell.execute_reply":"2022-07-05T11:58:50.612652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport cv2\nimport math\nimport copy\nimport time\nimport random\n\nimport cudf\nimport cupy\nimport pandas as pd\nimport numpy as np\nimport xgboost as xgb\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\n# Pytorch Imports\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.cuda import amp\n\n# For Transformer Models\nimport transformers\nfrom transformers import AutoTokenizer, AutoModel, AdamW, AutoConfig\n\nfrom cuml.neighbors import NearestNeighbors\nfrom cuml.feature_extraction.text import TfidfVectorizer\nfrom cuml.metrics import pairwise_distances\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import GroupKFold, KFold, StratifiedKFold\n\n# from unidecode import unidecode\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T11:58:50.616682Z","iopub.execute_input":"2022-07-05T11:58:50.617317Z","iopub.status.idle":"2022-07-05T11:59:02.878979Z","shell.execute_reply.started":"2022-07-05T11:58:50.617261Z","shell.execute_reply":"2022-07-05T11:59:02.878180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    debug = False\n    one_fold = True\n\n    seed = 42\n    n_candidates = 3 if debug else 50\n    n_splits = 4\n\n    #reuse_dir = \"../input/v-embed-bert-0614/\"\n    reuse_dir1 = \"../input/xgb-roberta-large-add-feat-0618/\" # No42\n    #reuse_dir2 = \"../input/xgb-exp030-all/\" # No47\n    #reuse_dir3 = \"../input/exp031-xgb/\" # No48\n    reuse_dir2 = \"../input/exp032-xgb/\" # No52\n    reuse_dir3 = \"../input/xgb-paraphrasemultilingualmpnetbase/\" # No67\n    reuse_dir4 = \"../input/xgb-rembert/\" # No70\n    \n    # Training config\n    train_batch_size = 128\n    valid_batch_size = 128\n    epochs = 10\n    lr = 5e-5\n    n_accumulate = 1\n    max_grad_norm = 1000\n    weight_decay = 1e-6\n\n    device = torch.device('cuda')\n\n    # Model config\n\n    max_length = 64\n\n    # Metric loss and its params\n    loss_module = 'arcface'\n    s = 30.0\n    m = 0.5 \n    ls_eps = 0.0\n    easy_margin = False\n\n    # Model parameters\n    # n_classes = df[\"point_of_interest\"].nunique()\n    n_classes = 739972\n    # n_classes = 587049\n    pooling = 'clf'\n    use_fc = False\n    dropout = 0.0\n    # fc_dim = 384","metadata":{"execution":{"iopub.status.busy":"2022-07-05T11:59:02.880353Z","iopub.execute_input":"2022-07-05T11:59:02.880729Z","iopub.status.idle":"2022-07-05T11:59:02.894247Z","shell.execute_reply.started":"2022-07-05T11:59:02.880689Z","shell.execute_reply":"2022-07-05T11:59:02.893472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\n\ndef save_pickle(data, file_name):\n    with open(f\"{file_name}.pickle\", \"wb\") as f:\n        pickle.dump(data, f)\n\n\ndef load_pickle(file_name):\n    with open(f\"{file_name}.pickle\", \"rb\") as f:\n        d = pickle.load(f)\n    return d","metadata":{"execution":{"iopub.status.busy":"2022-07-05T11:59:02.898187Z","iopub.execute_input":"2022-07-05T11:59:02.898477Z","iopub.status.idle":"2022-07-05T11:59:02.951881Z","shell.execute_reply.started":"2022-07-05T11:59:02.898449Z","shell.execute_reply":"2022-07-05T11:59:02.950468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef set_seed(seed=42):\n    '''Sets the seed of the entire notebook so results are the same every time we run.\n    This is for REPRODUCIBILITY.'''\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    # When running on the CuDNN backend, two further options must be set\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    # Set a fixed value for the hash seed\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    \nset_seed(CFG.seed)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T11:59:02.955873Z","iopub.execute_input":"2022-07-05T11:59:02.956356Z","iopub.status.idle":"2022-07-05T11:59:02.970051Z","shell.execute_reply.started":"2022-07-05T11:59:02.956306Z","shell.execute_reply":"2022-07-05T11:59:02.969157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ArcMarginProduct(nn.Module):\n    r\"\"\"Implement of large margin arc distance: :\n        Args:\n            in_features: size of each input sample\n            out_features: size of each output sample\n            s: norm of input feature\n            m: margin\n            cos(theta + m)\n        \"\"\"\n    def __init__(self, in_features, out_features, s=30.0, \n                 m=0.50, easy_margin=False, ls_eps=0.0):\n        super(ArcMarginProduct, self).__init__()\n        self.in_features = in_features\n        self.out_features = out_features\n        self.s = s\n        self.m = m\n        self.ls_eps = ls_eps  # label smoothing\n        self.weight = nn.Parameter(torch.FloatTensor(out_features, in_features))\n        nn.init.xavier_uniform_(self.weight)\n\n        self.easy_margin = easy_margin\n        self.cos_m = math.cos(m)\n        self.sin_m = math.sin(m)\n        self.th = math.cos(math.pi - m)\n        self.mm = math.sin(math.pi - m) * m\n\n    def forward(self, input, label):\n        # --------------------------- cos(theta) & phi(theta) ---------------------\n        cosine = F.linear(F.normalize(input), F.normalize(self.weight))\n        sine = torch.sqrt(1.0 - torch.pow(cosine, 2))\n        phi = cosine * self.cos_m - sine * self.sin_m\n        if self.easy_margin:\n            phi = torch.where(cosine > 0, phi, cosine)\n        else:\n            phi = torch.where(cosine > self.th, phi, cosine - self.mm)\n        # --------------------------- convert label to one-hot ---------------------\n        # one_hot = torch.zeros(cosine.size(), requires_grad=True, device='cuda')\n        one_hot = torch.zeros(cosine.size(), device=CFG.device)\n        one_hot.scatter_(1, label.view(-1, 1).long(), 1)\n        if self.ls_eps > 0:\n            one_hot = (1 - self.ls_eps) * one_hot + self.ls_eps / self.out_features\n        # -------------torch.where(out_i = {x_i if condition_i else y_i) ------------\n        output = (one_hot * phi) + ((1.0 - one_hot) * cosine)\n        output *= self.s\n\n        return output\n\n\nclass FourSquareDataset(Dataset):\n    def __init__(self, df, tokenizer, max_length):\n        self.fulltext = df['fulltext'].values\n        self.latitudes = df['latitude'].values\n        self.longitudes = df['longitude'].values\n        self.coord_x = df['coord_x'].values\n        self.coord_y = df['coord_y'].values\n        self.coord_z = df['coord_z'].values\n        self.labels = df['point_of_interest'].values\n        self.tokenizer = tokenizer\n        self.max_length = max_length\n        \n    def __len__(self):\n        return len(self.fulltext)\n    \n    def __getitem__(self, index):\n        fulltext = self.fulltext[index]\n        latitude = self.latitudes[index]\n        longitude = self.longitudes[index]\n        label = self.labels[index]\n        coord_x = self.coord_x[index]\n        coord_y = self.coord_y[index]\n        coord_z = self.coord_z[index]\n        \n        inputs = self.tokenizer(\n            fulltext,\n            truncation=True,\n            add_special_tokens=True,\n            max_length=self.max_length,\n            padding='max_length',\n            return_tensors=\"pt\"\n        )\n\n        return {\n            'ids': inputs['input_ids'][0],\n            'mask': inputs['attention_mask'][0],\n            'latitude': torch.tensor(latitude, dtype=torch.float),\n            'longitude': torch.tensor(longitude, dtype=torch.float),\n            'coord_x': torch.tensor(coord_x),\n            'coord_y': torch.tensor(coord_y),\n            'coord_z': torch.tensor(coord_z),\n            'label': torch.tensor(label, dtype=torch.long)\n        }\n    \n\nclass FSNet(nn.Module):\n    def __init__(self, model_name, fc_dim):\n        super(FSNet, self).__init__()\n        self.config = AutoConfig.from_pretrained(model_name)\n        self.bert_model = AutoModel.from_pretrained(model_name, config=self.config)\n        # self.embedding = nn.Linear(self.config.hidden_size + 2, embedding_size)\n\n        self.fc = nn.Linear(self.bert_model.config.hidden_size, fc_dim)\n        self.bn = nn.BatchNorm1d(fc_dim)\n        self._init_params()\n\n        self.margin = ArcMarginProduct(\n            fc_dim,\n            CFG.n_classes,\n            s=CFG.s, \n            m=CFG.m, \n            easy_margin=CFG.easy_margin,\n            ls_eps=CFG.ls_eps\n        )\n\n    def _init_params(self):\n        nn.init.xavier_normal_(self.fc.weight)\n        nn.init.constant_(self.fc.bias, 0)\n        nn.init.constant_(self.bn.weight, 1)\n        nn.init.constant_(self.bn.bias, 0)\n\n    def forward(self, ids, mask, lat, lon, coord_x, coord_y, coord_z, labels):\n        feature = self.extract_feature(ids, mask, lat, lon, coord_x, coord_y, coord_z)\n        output = self.margin(feature, labels)\n\n        return output\n    \n    def extract_feature(self, input_ids, attention_mask, lat, lon, coord_x, coord_y, coord_z):\n        x = self.bert_model(input_ids=input_ids, attention_mask=attention_mask)\n        x = x.last_hidden_state.mean(dim=1)\n\n        x = self.fc(x)\n        x = self.bn(x)\n\n        return x\n    \n\nclass FSMultiModalNet(nn.Module):\n    def __init__(self, model_name, fc_dim, num_features=3):\n        super(FSMultiModalNet, self).__init__()\n        self.config = AutoConfig.from_pretrained(model_name)\n        self.bert_model = AutoModel.from_pretrained(model_name, config=self.config)\n        # self.embedding = nn.Linear(self.config.hidden_size + 2, embedding_size)\n\n        self.fc = nn.Linear(self.bert_model.config.hidden_size + num_features, fc_dim)\n        self.bn = nn.BatchNorm1d(fc_dim)\n        self._init_params()\n\n        self.margin = ArcMarginProduct(\n            fc_dim,\n            CFG.n_classes,\n            s=CFG.s, \n            m=CFG.m, \n            easy_margin=CFG.easy_margin,\n            ls_eps=CFG.ls_eps\n        )\n\n    def _init_params(self):\n        nn.init.xavier_normal_(self.fc.weight)\n        nn.init.constant_(self.fc.bias, 0)\n        nn.init.constant_(self.bn.weight, 1)\n        nn.init.constant_(self.bn.bias, 0)\n\n    def forward(self, ids, mask, lat, lon, coord_x, coord_y, coord_z, labels):\n        feature = self.extract_feature(ids, mask, lat, lon, coord_x, coord_y, coord_z)\n        output = self.margin(feature, labels)\n\n        return output\n    \n    def extract_feature(self, input_ids, attention_mask, lat, lon, coord_x, coord_y, coord_z):\n        x = self.bert_model(input_ids=input_ids, attention_mask=attention_mask)\n        x = torch.sum(x.last_hidden_state * attention_mask.unsqueeze(-1), dim=1) / attention_mask.sum(dim=1, keepdims=True)\n\n        # x = torch.cat([x, lat.view(-1, 1), lon.view(-1, 1)], axis=1)\n        x = torch.cat([x, coord_x.view(-1, 1), coord_y.view(-1, 1), coord_z.view(-1, 1)], axis=1)\n\n        x = self.fc(x)\n        x = self.bn(x)\n\n        return x\n\n        \n\n# model_text = FSNet(CFG.model_name, 320)\n# model_text.to(CFG.device)\n# model_text.load_state_dict(torch.load(\"../input/fs-distilbert-base-multilingual-cased/bert_epoch10.bin\"))\n\n#model_bert_multi = FSMultiModalNet(CFGBertMulti.model_name, 320)\n#model_bert_multi.to(CFG.device);\n#model_bert_multi.load_state_dict(torch.load(\"../input/fsq-heartkilla/exp36-xlm-roberta-large_epoch28.bin\"))\n\n# model_bert = FSMultiModalNet(CFGBert.model_name, 384)\n# model_bert.to(CFG.device);\n# model_bert.load_state_dict(torch.load(\"../input/fs-multimodal/distilbert-base-uncased_epoch20.bin\"))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T11:59:02.974380Z","iopub.execute_input":"2022-07-05T11:59:02.974608Z","iopub.status.idle":"2022-07-05T11:59:03.021515Z","shell.execute_reply.started":"2022-07-05T11:59:02.974582Z","shell.execute_reply":"2022-07-05T11:59:03.020354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = cudf.read_csv(\"../input/fsq-fillna/train_filled.csv\")\nif CFG.debug:\n    kf = GroupKFold(n_splits=11)\n    for fold, (_, valid_index) in enumerate(kf.split(df.to_pandas(), df['id'].to_pandas(), df['point_of_interest'].to_pandas())):\n        if fold == 5:  # pick only fold 5 for debug\n            df = df.iloc[valid_index].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T11:59:03.025706Z","iopub.execute_input":"2022-07-05T11:59:03.026706Z","iopub.status.idle":"2022-07-05T11:59:14.356709Z","shell.execute_reply.started":"2022-07-05T11:59:03.026666Z","shell.execute_reply":"2022-07-05T11:59:14.355870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in [\"name\", \"address\", \"city\", \"state\", \"zip\", \"country\", \"url\", \"phone\", \"categories\"]:\n    df[col] = df[col].fillna(\"\")\n\ndf[\"fulltext\"] = (\n    df[\"name\"] + \" \" + df[\"address\"] + \" \" + df[\"city\"] + \" \" + df[\"state\"] + \" \"  + df[\"country\"] + \" \" + df[\"categories\"]\n).to_pandas().replace(r'\\s+', ' ', regex=True)\n\n# preprocess of string\n# df[\"fulltext\"] = df[\"fulltext\"].str.lower()  # lowercase\n# df[\"fulltext\"] = df[\"fulltext\"].str.replace(r'[^\\w\\s]+', '')  # remove punctuation\n\n# Standardization of coordinates.\n# https://datascience.stackexchange.com/questions/13567/ways-to-deal-with-longitude-latitude-feature\ndf[\"coord_x\"] = cupy.cos(df[\"latitude\"]) * cupy.cos(df[\"longitude\"])\ndf[\"coord_y\"] = cupy.cos(df[\"latitude\"]) * cupy.sin(df[\"longitude\"])\ndf[\"coord_z\"] = cupy.sin(df[\"latitude\"])\n                       \nprint(df.shape)\ndf.head()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-05T11:59:14.358214Z","iopub.execute_input":"2022-07-05T11:59:14.358510Z","iopub.status.idle":"2022-07-05T11:59:27.205684Z","shell.execute_reply.started":"2022-07-05T11:59:14.358469Z","shell.execute_reply":"2022-07-05T11:59:27.204905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoder = LabelEncoder()\ndf['point_of_interest'] = encoder.fit_transform(df['point_of_interest'].to_array())","metadata":{"execution":{"iopub.status.busy":"2022-07-05T11:59:27.207175Z","iopub.execute_input":"2022-07-05T11:59:27.207467Z","iopub.status.idle":"2022-07-05T11:59:31.623329Z","shell.execute_reply.started":"2022-07-05T11:59:27.207426Z","shell.execute_reply":"2022-07-05T11:59:31.622444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Embeding","metadata":{}},{"cell_type":"code","source":"V_embed_bert1 = load_pickle(f\"{CFG.reuse_dir1}/V_embed_bert\")\nprint(V_embed_bert1.shape)\n\nV_embed_bert2 = load_pickle(f\"{CFG.reuse_dir2}/V_embed_bert\")\nprint(V_embed_bert2.shape)\n\nV_embed_bert3 = load_pickle(f\"{CFG.reuse_dir3}/V_embed_bert\")\nprint(V_embed_bert3.shape)\n\nV_embed_bert4 = load_pickle(f\"{CFG.reuse_dir4}/V_embed_bert\")\nprint(V_embed_bert4.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T11:59:31.627172Z","iopub.execute_input":"2022-07-05T11:59:31.627577Z","iopub.status.idle":"2022-07-05T12:00:32.678396Z","shell.execute_reply.started":"2022-07-05T11:59:31.627538Z","shell.execute_reply":"2022-07-05T12:00:32.676412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# l2 normalization and concat\nV_embed_concat = cupy.concatenate([\n    V_embed_bert1,\n    V_embed_bert2,\n    V_embed_bert3,\n    V_embed_bert4,\n], axis=1)\n\n\ndel V_embed_bert1,V_embed_bert2,V_embed_bert3,V_embed_bert4\ngc.collect()\ntorch.cuda.empty_cache()\n\n\n#V_embed_concat.shape\nV_embed_concat = V_embed_concat / cupy.linalg.norm(V_embed_concat, ord=2, axis=1, keepdims=True)\n\nprint(V_embed_concat.shape)\n#del dataset, loader, model_bert_multi","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:00:32.679808Z","iopub.execute_input":"2022-07-05T12:00:32.680128Z","iopub.status.idle":"2022-07-05T12:00:33.901529Z","shell.execute_reply.started":"2022-07-05T12:00:32.680085Z","shell.execute_reply":"2022-07-05T12:00:33.900586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generate Candidates","metadata":{}},{"cell_type":"code","source":"# Create candidate index country by country\nimport faiss\n\ndef gen_candidate_ranks(df, V_embed, n_candidates, no_country=False):\n    \n    if no_country:\n        res = faiss.StandardGpuResources()\n        index = faiss.IndexFlatIP(V_embed.shape[1])\n        index = faiss.index_cpu_to_gpu(res, 0, index)\n        index.add(V_embed)\n        D, I = index.search(V_embed, CFG.n_candidates)\n    else:\n        D = np.zeros((len(df), n_candidates), dtype=np.float32)\n        I = np.zeros((len(df), n_candidates), dtype=np.int32)\n        for country, coutry_df in tqdm(df[[\"country\"]].to_pandas().groupby(\"country\")):\n            country_df = coutry_df.reset_index()\n            country_V_embed = V_embed[country_df[\"index\"].values, :]    \n            res = faiss.StandardGpuResources()\n            index = faiss.IndexFlatIP(country_V_embed.shape[1])\n            index = faiss.index_cpu_to_gpu(res, 0, index)\n            index.add(country_V_embed)\n            country_embed_d, country_embed_i = index.search(country_V_embed, n_candidates)\n            country_embed_i = np.where(country_embed_i == -1, 0, country_embed_i)\n            for i in range(n_candidates):\n                D[country_df[\"index\"].values, i] = country_embed_d[:, i]\n                I[country_df[\"index\"].values, i] = country_df.loc[country_embed_i[:, i], \"index\"].values\n            del country_df, res, index, country_embed_d, country_embed_i  # ここで del するとなんとか動く\n            torch.cuda.empty_cache()\n    D = np.clip(D, 0, 1)\n    return D, I","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:00:33.903056Z","iopub.execute_input":"2022-07-05T12:00:33.903556Z","iopub.status.idle":"2022-07-05T12:00:34.070362Z","shell.execute_reply.started":"2022-07-05T12:00:33.903514Z","shell.execute_reply":"2022-07-05T12:00:34.069185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"D_concat, I_concat = gen_candidate_ranks(df, cupy.asnumpy(V_embed_concat), CFG.n_candidates, no_country=True)\nnp.save(\"D_concat.npy\", D_concat)\nnp.save(\"I_concat.npy\", I_concat)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:00:34.072457Z","iopub.execute_input":"2022-07-05T12:00:34.072879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I_spatial = gen_candidate_rank_spatial(df, 10)\n# np.save(\"I_spatial.npy\", I_spatial)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DBA/QE weighted","metadata":{}},{"cell_type":"code","source":"# https://www.kaggle.com/code/lyakaap/2nd-place-solution/notebook\ndef query_expansion(V, D, I, alpha=3, k=2):\n    weights = cupy.array(np.expand_dims(D[:, :k] ** alpha, axis=-1).astype(np.float32))\n    chunk_size = 100_000\n    for i in range(0, len(df), chunk_size):  # chunk に分けてやらないと `cudaErrorMemoryAllocation out of memory` になる…\n        V[i:i+chunk_size] = (V[I[i:i+chunk_size, :k]] * weights[i:i+chunk_size]).sum(axis=1)\n    return V\n\n\nV_embed_concat = query_expansion(V_embed_concat, D_concat, I_concat)\nV_embed_concat /= np.linalg.norm(V_embed_concat, 2, axis=1, keepdims=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del D_concat,I_concat\ngc.collect()\ntorch.cuda.empty_cache()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"D_concat, I_concat = gen_candidate_ranks(df, cupy.asnumpy(V_embed_concat), CFG.n_candidates, no_country=True)\nnp.save(\"D_concat_dbaqe.npy\", D_concat)\nnp.save(\"I_concat_dbaqe.npy\", I_concat)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Spatial candidates","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import BallTree\n\n\ndef gen_candidate_rank_spatial(df, n_candidates):\n    \"\"\"lat/lon から曲面上の距離 (haversine) の近い順に候補点を作成する\n    \n    Returns:\n        I (np.array): I[s_index][r] -> s_index の r 番目に近い地点の index\n    \"\"\"\n    for column in df[[\"latitude\", \"longitude\"]]:\n        rad = np.deg2rad(df[column].values)\n        df[f'{column}_rad'] = rad\n\n    I = np.full((len(df), n_candidates), -1, dtype=np.int32)\n    for country, country_df in tqdm(df[[\"country\", \"latitude_rad\", \"longitude_rad\"]].to_pandas().groupby(\"country\")):\n        clip_n_candidates = min(len(country_df), n_candidates)\n        country_df = country_df.reset_index()\n        ball = BallTree(country_df[[\"latitude_rad\", \"longitude_rad\"]].values, metric='haversine')\n        _, indices = ball.query(\n            country_df[[\"latitude_rad\", \"longitude_rad\"]].values, \n            k = clip_n_candidates\n        )\n        indices = np.concatenate(\n            [indices, np.zeros((len(indices), n_candidates - clip_n_candidates), dtype=np.int32)], axis=1\n        )\n        for i in range(n_candidates):\n            I[country_df[\"index\"].values, i] = country_df.loc[indices[:, i], \"index\"].values\n\n    return I\n\nI_spatial = gen_candidate_rank_spatial(df, 10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# tfdfi","metadata":{}},{"cell_type":"code","source":"tfidf = TfidfVectorizer(stop_words='english')\nV_name = tfidf.fit_transform(df[\"name\"])\nV_name.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfidf = TfidfVectorizer(stop_words='english')\ndf[\"full_address\"] = df[\"address\"] + \", \" + df[\"city\"] + \", \" + df[\"state\"] + \", \"  + df[\"country\"]\nV_full_address = tfidf.fit_transform(df[\"full_address\"])\nV_full_address.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfidf = TfidfVectorizer(stop_words='english')\nV_cat = tfidf.fit_transform(df[\"categories\"].fillna(\"nocategory\"))\nV_cat.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train & Test","metadata":{}},{"cell_type":"code","source":"R = 6371.0  # radius of the earth in km\n\ndef manhattan(lat1, long1, lat2, long2):\n    return np.abs(lat2 - lat1) + np.abs(long2 - long1)\n\n\ndef haversine_np(lon1, lat1, lon2, lat2):\n    \"\"\"https://www.kaggle.com/code/justfor/speedup-haversine/script\n    \n    Calculate the great circle distance between two points\n    on the earth (specified in decimal degrees)\n\n    All args must be of equal length.    \n\n    \"\"\"\n    lon1, lat1, lon2, lat2 = map(np.radians, [lon1, lat1, lon2, lat2])\n\n    dlon = lon2 - lon1\n    dlat = lat2 - lat1\n\n    a = np.sin(dlat/2.0)**2 + np.cos(lat1) * np.cos(lat2) * np.sin(dlon/2.0)**2\n\n    c = 2 * np.arcsin(np.sqrt(a))\n    km = R * c\n    \n    return km\n\n\ndef create_features(df, i, indices):\n    \n    prev_i = max(i-1, 0)\n    next_i = min(i+1, indices.shape[1] - 1)\n    \n    prev_cand_index = indices[:, prev_i]\n    next_cand_index = indices[:, next_i]\n    \n    cand_index = indices[:, i]\n    \n    lon1 = df[\"longitude\"].to_pandas().to_numpy()\n    lat1 = df[\"latitude\"].to_pandas().to_numpy()\n    lon2 = df[\"longitude\"][cand_index].to_pandas().to_numpy()\n    lat2 = df[\"latitude\"][cand_index].to_pandas().to_numpy()\n    #df[\"diff_lon\"] = lon1 - lon2\n    #df[\"diff_lat\"] = lat1 - lat2\n    df[\"diff_lon\"] = np.abs(lon1 - lon2)\n    df[\"diff_lat\"] = np.abs(lat1 - lat2)\n    \n    df[\"lonlat_eucdist\"] =  (df['diff_lon'] ** 2 + df['diff_lat'] ** 2) ** 0.5\n    df[\"lonlat_manhattan\"] = manhattan(lat1, lon2, lat2, lon2)\n    df[\"lonlat_haversine_dist\"] = haversine_np(lon1, lat1, lon2, lat2)\n    \n    df[\"name_cossim\"] = V_name.multiply(V_name[cand_index]).sum(axis=1).ravel()\n    df[\"full_address_cossim\"] = V_full_address.multiply(V_full_address[cand_index]).sum(axis=1).ravel()\n    df[\"cat_cossim\"] = V_cat.multiply(V_cat[cand_index]).sum(axis=1).ravel()\n    \n    df[\"cand_hit_count_02\"] = df[\"hit_count_02\"][cand_index].to_pandas().to_numpy()\n    df[\"cand_hit_count_03\"] = df[\"hit_count_03\"][cand_index].to_pandas().to_numpy()\n    df[\"cand_hit_count_04\"] = df[\"hit_count_04\"][cand_index].to_pandas().to_numpy()\n    df[\"cand_hit_count_05\"] = df[\"hit_count_05\"][cand_index].to_pandas().to_numpy()\n    df[\"cand_hit_count_sum\"] = df[\"hit_count_sum\"][cand_index].to_pandas().to_numpy()\n    \n    df[\"hit_count_02_min\"] = df[[\"hit_count_02\", \"cand_hit_count_02\"]].min(axis=1)\n    df[\"hit_count_03_min\"] = df[[\"hit_count_03\", \"cand_hit_count_03\"]].min(axis=1)\n    df[\"hit_count_04_min\"] = df[[\"hit_count_04\", \"cand_hit_count_04\"]].min(axis=1)\n    df[\"hit_count_05_min\"] = df[[\"hit_count_05\", \"cand_hit_count_05\"]].min(axis=1)\n    df[\"hit_count_sum_min\"] = df[[\"hit_count_sum\", \"cand_hit_count_sum\"]].min(axis=1)\n    \n    df[\"hit_count_02_max\"] = df[[\"hit_count_02\", \"cand_hit_count_02\"]].max(axis=1)\n    df[\"hit_count_03_max\"] = df[[\"hit_count_03\", \"cand_hit_count_03\"]].max(axis=1)\n    df[\"hit_count_04_max\"] = df[[\"hit_count_04\", \"cand_hit_count_04\"]].max(axis=1)\n    df[\"hit_count_05_max\"] = df[[\"hit_count_05\", \"cand_hit_count_05\"]].max(axis=1)\n    df[\"hit_count_sum_max\"] = df[[\"hit_count_sum\", \"cand_hit_count_sum\"]].max(axis=1)\n    \n    cossim = []\n    eucdist = []\n    \n    eucdist1=[]\n    eucdist2=[]\n    eucdist3=[]\n    eucdist4=[]\n    eucdist5=[]\n    \n    chunk_size = 100_000\n    for i in range(0, len(df), chunk_size):  # chunk に分けてやらないと `cudaErrorMemoryAllocation out of memory` になる…\n        cossim.append(cupy.multiply(V_embed_concat[i:i+chunk_size], V_embed_concat[cand_index[i:i+chunk_size]]).sum(axis=1))\n        eucdist.append(cupy.sqrt(((V_embed_concat[i:i+chunk_size] - V_embed_concat[cand_index[i:i+chunk_size]]) ** 2).sum(axis=1)))\n        \n        eucdist1.append(cupy.sqrt(((V_embed_concat[prev_cand_index[i:i+chunk_size]] - V_embed_concat[cand_index[i:i+chunk_size]]) ** 2).sum(axis=1)))\n        eucdist2.append(cupy.sqrt(((V_embed_concat[next_cand_index[i:i+chunk_size]] - V_embed_concat[cand_index[i:i+chunk_size]]) ** 2).sum(axis=1)))\n        eucdist3.append(cupy.sqrt(((V_embed_concat[i:i+chunk_size]                  - V_embed_concat[prev_cand_index[i:i+chunk_size]]) ** 2).sum(axis=1)))\n        eucdist4.append(cupy.sqrt(((V_embed_concat[i:i+chunk_size]                  - V_embed_concat[next_cand_index[i:i+chunk_size]]) ** 2).sum(axis=1)))\n        eucdist5.append(cupy.sqrt(((V_embed_concat[prev_cand_index[i:i+chunk_size]] - V_embed_concat[next_cand_index[i:i+chunk_size]]) ** 2).sum(axis=1)))\n\n    df[\"embed_cossim\"] = cupy.concatenate(cossim)\n    df[\"embed_eucdist\"] = cupy.concatenate(eucdist)\n    \n    df[\"embed_eucdist1\"] = cupy.concatenate(eucdist1)\n    df[\"embed_eucdist2\"] = cupy.concatenate(eucdist2)\n    df[\"embed_eucdist3\"] = cupy.concatenate(eucdist3)\n    df[\"embed_eucdist4\"] = cupy.concatenate(eucdist4)\n    df[\"embed_eucdist5\"] = cupy.concatenate(eucdist5)\n    \n    df[\"d0_d1\"] = df[\"embed_eucdist\"] - df[\"embed_eucdist1\"]\n    df[\"d0_d2\"] = df[\"embed_eucdist\"] - df[\"embed_eucdist2\"]\n    df[\"d0_d3\"] = df[\"embed_eucdist\"] - df[\"embed_eucdist3\"]\n    df[\"d0_d4\"] = df[\"embed_eucdist\"] - df[\"embed_eucdist4\"]\n    df[\"d0_d5\"] = df[\"embed_eucdist\"] - df[\"embed_eucdist5\"]\n    \n    for col in [\"id\", \"name\", \"address\", \"city\", \"state\", \"zip\", \"country\", \"url\", \"phone\", \"categories\", \"full_address\"]:\n        df[f\"{col}_edit_dist\"] = df[col].str.edit_distance(df[col][cand_index])\n        df[f\"norm_{col}_edit_dist\"] = df[f\"{col}_edit_dist\"] / df[col].str.len()\n        df[f\"norm_{col}_edit_dist\"] = df[f\"norm_{col}_edit_dist\"].replace([np.inf, -np.inf], 0)\n    \n    features = [\n        \"diff_lon\",\n        \"diff_lat\",\n        \"lonlat_eucdist\",\n        \"lonlat_manhattan\",\n        \"lonlat_haversine_dist\",\n        \"name_cossim\",\n        \"full_address_cossim\",\n        \"cat_cossim\",\n        \"embed_cossim\",\n        \"embed_eucdist\",\n        \n        \"embed_eucdist1\",\n        \"embed_eucdist2\",\n        \"embed_eucdist3\",\n        \"embed_eucdist4\",\n        \"embed_eucdist5\",\n        \"d0_d1\",\n        \"d0_d2\",\n        \"d0_d3\",\n        \"d0_d4\",\n        \"d0_d5\",\n\n        \"hit_count_02\",\n        \"hit_count_03\",\n        \"hit_count_04\",\n        \"hit_count_05\",\n        \"hit_count_sum\",\n        \n        \"hit_count_02_min\",\n        \"hit_count_03_min\",\n        \"hit_count_04_min\",\n        \"hit_count_05_min\",\n        \"hit_count_sum_min\",\n        \"hit_count_02_max\",\n        \"hit_count_03_max\",\n        \"hit_count_04_max\",\n        \"hit_count_05_max\",\n        \"hit_count_sum_max\"\n    ]\n    \n    for col in [\"id\", \"name\", \"address\", \"city\", \"state\", \"zip\", \"country\", \"url\", \"phone\", \"categories\", \"full_address\"]:\n        features.append(f\"{col}_edit_dist\")\n        features.append(f\"norm_{col}_edit_dist\")\n    \n    return df, features","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/code/columbia2131/foursquare-iou-metrics\ndef get_id2poi(input_df: pd.DataFrame) -> dict:\n    return dict(zip(input_df['id'], input_df['point_of_interest']))\n\n\ndef get_poi2ids(input_df: pd.DataFrame) -> dict:\n    return input_df.groupby('point_of_interest')['id'].apply(set).to_dict()\n\nid2poi = get_id2poi(df.to_pandas())\npoi2ids = get_poi2ids(df.to_pandas())\n\n\ndef id2target_size(id_str: str):\n    return len(poi2ids[id2poi[id_str]])\n\n\ndef get_score(input_df: pd.DataFrame):\n    scores = []\n    for id_str, matches in zip(input_df['id'].to_numpy(), input_df['matches'].to_numpy()):\n        targets = poi2ids[id2poi[id_str]]\n        preds = set(matches.split())\n        score = len((targets & preds)) / len((targets | preds))\n        scores.append(score)\n    scores = np.array(scores)\n    return scores.mean()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"THRESH = 0.4","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import defaultdict\n\n\ndef train(I, n_candidates, params, cand_type):\n    \n    link = defaultdict(lambda:defaultdict(lambda: 10**10))\n    \n    models = {}\n    kf = GroupKFold(n_splits=CFG.n_splits)\n    for i, (trn_idx, val_idx) in enumerate(kf.split(df.to_pandas(), df[\"point_of_interest\"].to_pandas(), df[\"point_of_interest\"].to_pandas())):\n        df.loc[val_idx, \"fold\"] = i\n\n    folds = df[\"fold\"].to_pandas().to_numpy()\n    df[\"fold\"].value_counts()\n\n    folds = df[\"fold\"].to_pandas().to_numpy()\n\n    # 0で初期化しておく\n    df[\"hit_count_02\"] = 0\n    df[\"hit_count_03\"] = 0\n    df[\"hit_count_04\"] = 0\n    df[\"hit_count_05\"] = 0\n    df[\"hit_count_sum\"] = 0\n    \n    df[\"cand_hit_count_02\"] = 0\n    df[\"cand_hit_count_03\"] = 0\n    df[\"cand_hit_count_04\"] = 0\n    df[\"cand_hit_count_05\"] = 0\n    df[\"cand_hit_count_sum\"] = 0\n    \n    oof = []\n    # training\n    for i in range(I.shape[1]):\n        tmp_df = df.copy()\n        tmp_df[\"target\"] = tmp_df[\"point_of_interest\"] == tmp_df[\"point_of_interest\"].to_pandas().values[I[:, i]]\n        tmp_df[\"match_id\"] = tmp_df[\"id\"].to_pandas().values[I[:, i]]\n        \n        tmp_df[\"id_target_size\"] = tmp_df.id.to_pandas().apply(id2target_size)\n        tmp_df[\"match_id_target_size\"] = tmp_df.match_id.to_pandas().apply(id2target_size)\n        tmp_df[\"sample_weight\"] = (1 / tmp_df.id_target_size) + (1 / tmp_df.match_id_target_size)\n\n        tmp_df, features = create_features(tmp_df, i, I)\n\n        print(f\"Candidate rank {i} target mean = {tmp_df['target'].mean()}\")\n\n        for fold in range(CFG.n_splits):\n            if CFG.one_fold and fold != 0:\n                continue\n\n            print(f\"== fold {fold} ==\")\n            train_idx = folds != fold\n            test_idx = folds == fold\n\n            X_train, y_train, train_weights = (\n                tmp_df.loc[train_idx, features],\n                tmp_df.loc[train_idx, \"target\"].astype(int),\n                tmp_df.loc[train_idx, \"sample_weight\"],\n            )\n            X_test, y_test = (\n                tmp_df.loc[test_idx, features],\n                tmp_df.loc[test_idx, \"target\"].astype(int),\n            )\n            X_all, y_all = (\n                tmp_df[features],\n                tmp_df[\"target\"].astype(int)\n            )\n\n            _oof = tmp_df.loc[test_idx, [\"id\", \"point_of_interest\", \"match_id\"]].to_pandas()\n\n            dtrain = xgb.DMatrix(X_train, y_train, weight=train_weights)\n            dtest = xgb.DMatrix(X_test, y_test)\n            dall = xgb.DMatrix(X_all, y_all)\n\n            xgb_model = xgb.train(\n                params=params,\n                dtrain=dtrain,\n                num_boost_round=5_000,\n                evals=[(dtrain, \"train\"), (dtest, \"test\")],\n                early_stopping_rounds=100,\n                verbose_eval=500,\n            )\n            xgb_model.save_model(f\"fs_xgb_model_{cand_type}_candidate{i}_fold{fold}.json\")\n            models[i] = xgb_model\n            _oof[\"pred\"] = xgb_model.predict(dtest, ntree_limit=xgb_model.best_iteration)\n\n            oof.append(_oof.query(\"pred >= @THRESH\"))\n            all_pred = xgb_model.predict(dall, ntree_limit=xgb_model.best_iteration)\n            \n            ids = _oof.id.values\n            match_ids = _oof.match_id.values\n            pred_lis = _oof.pred.values\n            for i in range(len(_oof)):\n                dist = 1 - pred_lis[i]\n\n                if dist > 0.9:\n                    continue\n                current_min = min([link[ids[i]][match_ids[i]], link[match_ids[i]][ids[i]], dist])\n                link[match_ids[i]][ids[i]] = current_min\n                link[ids[i]][match_ids[i]] = current_min\n\n        #hit_count \n        df[\"hit_count_02\"] += (all_pred > 0.2)\n        df[\"hit_count_03\"] += (all_pred > 0.3)\n        df[\"hit_count_04\"] += (all_pred > 0.4)\n        df[\"hit_count_05\"] += (all_pred > 0.5)\n        df[\"hit_count_sum\"] += all_pred\n\n        print()\n        \n        gc.collect()\n        torch.cuda.empty_cache()\n\n    oof = pd.concat(oof, axis=0).reset_index(drop=True)\n    print(\"Done training.\")\n\n    return models, oof, link\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\ntorch.cuda.empty_cache()\n\nmodels_embed, oof_embed, link_embed = train(I_concat, CFG.n_candidates, params={\n    \"objective\": \"binary:logistic\",\n    \"tree_method\": \"gpu_hist\",\n    \"verbosity\": 0,\n    \"learning_rate\": 0.05,\n    \"max_depth\": 8,\n    \"min_child_weight\": 0,\n    \"lambda\": 1.0,\n    \"alpha\": 0.1,\n}, cand_type=\"embed\")\n\nmodels_spatial, oof_spatial, link_spatial = train(I_spatial, CFG.n_candidates, params={\n    \"objective\": \"binary:logistic\",\n    \"tree_method\": \"gpu_hist\",\n    \"verbosity\": 0,\n    \"learning_rate\": 0.05,\n    \"max_depth\": 8,\n    \"min_child_weight\": 0,\n    \"lambda\": 1.0,\n    \"alpha\": 0.1,\n}, cand_type=\"spatial\")\n\noof = pd.concat([oof_embed, oof_spatial], axis=0)\n\n# del df\n# gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = oof_embed.groupby(\"id\")[\"match_id\"].apply(list).reset_index()\ntmp[\"matches\"] = tmp[\"match_id\"].apply(lambda x: \" \".join(set(x)))\nprint(f\"Score {get_score(tmp)}, THRESH {THRESH}\")\n\ntmp = oof_spatial.groupby(\"id\")[\"match_id\"].apply(list).reset_index()\ntmp[\"matches\"] = tmp[\"match_id\"].apply(lambda x: \" \".join(set(x)))\nprint(f\"Score {get_score(tmp)}, THRESH {THRESH}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = oof.groupby(\"id\")[\"match_id\"].apply(list).reset_index()\ntmp[\"matches\"] = tmp[\"match_id\"].apply(lambda x: \" \".join(set(x)))\nprint(f\"Score {get_score(tmp)}, THRESH {THRESH}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature = []\nimportance = []\nfor m in models_embed.values():\n    fimp = m.get_score(importance_type=\"gain\")\n    feature.extend(fimp.keys())\n    importance.extend(fimp.values())\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\ndata = pd.DataFrame({\"feature\": feature, \"importance\": importance})\n\nplt.figure(figsize=(4, 12))\nsns.barplot(\n    x=\"importance\",\n    y=\"feature\",\n    data=data,\n    order=data.groupby(\"feature\").importance.mean().sort_values(ascending=False).index.values,\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from heapq import heappush, heappop\n# from collections import defaultdict\n\n# def dks(link,start,MAX_COST):\n#     visited = set()\n#     hq = []\n#     heappush(hq, (0,start))\n\n#     while hq:\n#         shortest, now = heappop(hq)\n        \n#         if (shortest > MAX_COST) :\n#             break\n        \n#         if now in visited:\n#             continue\n#         visited.add(now)\n            \n#         for nxt,next_cost in link[now].items():\n#             if nxt in visited:\n#                 continue\n#             heappush(hq, (shortest + next_cost, nxt))\n            \n#     return list(visited)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# id_lis = []\n# matches = []\n# for source_id in tqdm(set(oof.id)):\n#     ret=dks(link,source_id,THRESH)\n#     matches.append(ret)\n    \n#     id_lis.append(source_id)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dks_df = pd.DataFrame()\n# dks_df[\"id\"] = id_lis\n# dks_df[\"match_id\"] = matches\n# dks_df[\"matches\"] = dks_df[\"match_id\"].apply(lambda x: \" \".join(set(x)))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dks_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(f\"Score {get_score(dks_df)}, THRESH {THRESH}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for THRESH in np.arange(0.07,0.7,0.03):\n#     id_lis = []\n#     matches = []\n#     for source_id in set(oof.id):\n#         ret=dks(link,source_id,THRESH)\n#         matches.append(ret)\n\n#         id_lis.append(source_id)\n        \n#     dks_df = pd.DataFrame()\n#     dks_df[\"id\"] = id_lis\n#     dks_df[\"match_id\"] = matches\n#     dks_df[\"matches\"] = dks_df[\"match_id\"].apply(lambda x: \" \".join(set(x)))\n    \n#     print(f\"Score {get_score(dks_df)}, THRESH {THRESH}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}