{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q ../input/faiss-163/faiss_gpu-1.6.3-cp37-cp37m-manylinux2010_x86_64.whl\n# !pip install -q ../input/unidecode/Unidecode-1.3.4-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:02:48.192461Z","iopub.execute_input":"2022-07-05T14:02:48.193003Z","iopub.status.idle":"2022-07-05T14:03:20.824229Z","shell.execute_reply.started":"2022-07-05T14:02:48.192904Z","shell.execute_reply":"2022-07-05T14:03:20.822978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport cv2\nimport math\nimport copy\nimport time\nimport random\n\nimport cudf\nimport cupy\nimport pandas as pd\nimport numpy as np\nimport xgboost as xgb\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\n# Pytorch Imports\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.cuda import amp\n\n# For Transformer Models\nimport transformers\nfrom transformers import AutoTokenizer, AutoModel, AdamW, AutoConfig\n\nfrom cuml.neighbors import NearestNeighbors\nfrom cuml.feature_extraction.text import TfidfVectorizer\nfrom cuml.metrics import pairwise_distances\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import GroupKFold, KFold, StratifiedKFold\n\nfrom unidecode import unidecode\n\n\nclass CFG:\n    debug = False\n    one_fold = True\n\n    seed = 42\n    n_candidates = 3 if debug else 50\n    n_spatial_candidates = 10\n    n_splits = 4\n\n    # Training config\n    train_batch_size = 128\n    valid_batch_size = 128\n    epochs = 10\n    lr = 5e-5\n    n_accumulate = 1\n    max_grad_norm = 1000\n    weight_decay = 1e-6\n\n    device = torch.device('cuda')\n\n    # Model config\n\n    max_length = 64\n\n    # Metric loss and its params\n    loss_module = 'arcface'\n    s = 30.0\n    m = 0.5 \n    ls_eps = 0.0\n    easy_margin = False\n\n    # Model parameters\n    # n_classes = df[\"point_of_interest\"].nunique()\n    n_classes = 739972\n    # n_classes = 587049\n    pooling = 'clf'\n    use_fc = False\n    dropout = 0.0\n    # fc_dim = 384\n    \n\nclass CFG_Model1:\n    model_name = \"../input/xlmrobertalarge/\"\n    weight = \"../input/exp36xlmrobertalarge-epoch28/exp36-xlm-roberta-large_epoch28.bin\"\n    tokenizer = transformers.AutoTokenizer.from_pretrained(model_name)\n\nclass CFG_Model2:\n    model_name = \"../input/sentence-transformers/LaBSE/0_Transformer/\"\n    weight = \"../input/exp032-bert/LaBSE_epoch40.bin\"\n    tokenizer = transformers.AutoTokenizer.from_pretrained(model_name)\n    \nclass CFG_Model3:\n    model_name = \"../input/paraphrasemultilingualmpnetbasev2/paraphrase-multilingual-mpnet-base-v2/\"\n    weight = \"../input/exp036-paraphrase-multilingual-mpnet-base-v2/paraphrase-multilingual-mpnet-base-v2_epoch35.bin\"\n    tokenizer = transformers.AutoTokenizer.from_pretrained(model_name)\n    \nclass CFG_Model4:\n    model_name = \"../input/rembert\"\n    weight = \"../input/fsq-heartkilla/rembert_epoch40.bin\"\n    tokenizer = transformers.AutoTokenizer.from_pretrained(model_name)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-05T14:03:20.826781Z","iopub.execute_input":"2022-07-05T14:03:20.827055Z","iopub.status.idle":"2022-07-05T14:03:40.956943Z","shell.execute_reply.started":"2022-07-05T14:03:20.827021Z","shell.execute_reply":"2022-07-05T14:03:40.955871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = cudf.read_csv(\"../input/foursquare-location-matching/test.csv\", dtype={\n    \"id\": str, \"name\": str, \"latitude\": float, \"longitude\": float, \"address\": str, \n    \"city\": str, \"state\": str, \"zip\": str, \"country\": str, \"url\": str, \"phone\": str, \"categories\": str\n})\n\n# 欠損値埋め済み train.csv を利用する\ndf_train = cudf.read_csv(\"../input/fsq-fillna/train_filled.csv\", dtype={\n    \"id\": str, \"name\": str, \"latitude\": float, \"longitude\": float, \"address\": str, \n    \"city\": str, \"state\": str, \"zip\": str, \"country\": str, \"url\": str, \"phone\": str, \"categories\": str, \"point_of_interest\": str\n}).drop(columns=[\"point_of_interest\"])\n\ndf_all = cudf.concat([df, df_train], axis=0).reset_index(drop=True)\n\n# # increase data size for debug mode\n# if df.shape[0] == 5:\n# #     df = df.repeat(1_000_000).reset_index(drop=True)\n#     df = df.repeat(1_0000).reset_index(drop=True)\n\n# df_all.shape, df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:03:40.958887Z","iopub.execute_input":"2022-07-05T14:03:40.959235Z","iopub.status.idle":"2022-07-05T14:03:53.699421Z","shell.execute_reply.started":"2022-07-05T14:03:40.959149Z","shell.execute_reply":"2022-07-05T14:03:53.698492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import BallTree\n\n# Creates new columns converting coordinate degrees to radians.\nfor column in [\"latitude\", \"longitude\"]:\n    rad = np.deg2rad(df_all[column].to_pandas().values)\n    df_all[f'{column}_rad'] = rad\n    rad = np.deg2rad(df[column].to_pandas().values)\n    df[f'{column}_rad'] = rad\n\n\nk = 11\nball = BallTree(df_all[[\"latitude_rad\", \"longitude_rad\"]].to_pandas().values, metric='haversine')\nD_latlon, I_latlon = ball.query(df[[\"latitude_rad\", \"longitude_rad\"]].to_pandas().values, k=k)   ","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:03:53.701852Z","iopub.execute_input":"2022-07-05T14:03:53.702157Z","iopub.status.idle":"2022-07-05T14:03:56.295043Z","shell.execute_reply.started":"2022-07-05T14:03:53.702113Z","shell.execute_reply":"2022-07-05T14:03:56.293786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# latlon で近い順に住所関連の NaN を埋めていく\nfill_columns = [\"city\", \"state\", \"country\"]\nprint(\"Before fillna\")\nprint(df[fill_columns].isnull().sum() / df.shape[0])\nfor i in range(1, 6):  # 自分以外の5点を取る\n    nearest_index = I_latlon[:, i]\n    for c in fill_columns:\n        df[c] = df[c].fillna(df_all.loc[nearest_index, c].reset_index(drop=True))\n    print(f\"Iter {i}\")\n    print(df[fill_columns].isnull().sum() / df.shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:03:56.296774Z","iopub.execute_input":"2022-07-05T14:03:56.297107Z","iopub.status.idle":"2022-07-05T14:03:56.541640Z","shell.execute_reply.started":"2022-07-05T14:03:56.297061Z","shell.execute_reply":"2022-07-05T14:03:56.540850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in [\"name\", \"address\", \"city\", \"state\", \"zip\", \"country\", \"url\", \"phone\", \"categories\"]:\n    df[col] = df[col].fillna(\"\")\n\ndf[\"fulltext\"] = (\n    df[\"name\"] + \" \" + df[\"address\"] + \" \" + df[\"city\"] + \" \" + df[\"state\"] + \" \"  + df[\"country\"] + \" \" + df[\"categories\"]\n).to_pandas().replace(r'\\s+', ' ', regex=True)\n\n# preprocess of string\n# df[\"fulltext\"] = df[\"fulltext\"].str.lower()  # lowercase\n# df[\"fulltext\"] = df[\"fulltext\"].str.replace(r'[^\\w\\s]+', '')  # remove punctuation\n\n# Standardization of coordinates.\n# https://datascience.stackexchange.com/questions/13567/ways-to-deal-with-longitude-latitude-feature\ndf[\"coord_x\"] = cupy.cos(df[\"latitude\"]) * cupy.cos(df[\"longitude\"])\ndf[\"coord_y\"] = cupy.cos(df[\"latitude\"]) * cupy.sin(df[\"longitude\"])\ndf[\"coord_z\"] = cupy.sin(df[\"latitude\"])\n                       \nprint(df.shape)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:03:56.543418Z","iopub.execute_input":"2022-07-05T14:03:56.543740Z","iopub.status.idle":"2022-07-05T14:03:57.899699Z","shell.execute_reply.started":"2022-07-05T14:03:56.543699Z","shell.execute_reply":"2022-07-05T14:03:57.898831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed=42):\n    '''Sets the seed of the entire notebook so results are the same every time we run.\n    This is for REPRODUCIBILITY.'''\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    # When running on the CuDNN backend, two further options must be set\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    # Set a fixed value for the hash seed\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    \nset_seed(CFG.seed)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:03:57.901446Z","iopub.execute_input":"2022-07-05T14:03:57.902054Z","iopub.status.idle":"2022-07-05T14:03:57.915945Z","shell.execute_reply.started":"2022-07-05T14:03:57.902007Z","shell.execute_reply":"2022-07-05T14:03:57.914670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load ArcFace model","metadata":{}},{"cell_type":"code","source":"class FourSquareDataset(Dataset):\n    def __init__(self, df, tokenizer, max_length):\n        self.fulltext = df['fulltext'].values\n        self.latitudes = df['latitude'].values\n        self.longitudes = df['longitude'].values\n        self.coord_x = df['coord_x'].values\n        self.coord_y = df['coord_y'].values\n        self.coord_z = df['coord_z'].values\n        self.tokenizer = tokenizer\n        self.max_length = max_length\n        \n    def __len__(self):\n        return len(self.fulltext)\n    \n    def __getitem__(self, index):\n        fulltext = self.fulltext[index]\n        latitude = self.latitudes[index]\n        longitude = self.longitudes[index]\n        coord_x = self.coord_x[index]\n        coord_y = self.coord_y[index]\n        coord_z = self.coord_z[index]\n        \n        inputs = self.tokenizer(\n            fulltext,\n            truncation=True,\n            add_special_tokens=True,\n            max_length=self.max_length,\n            padding='max_length',\n            return_tensors=\"pt\"\n        )\n\n        return {\n            'ids': inputs['input_ids'][0],\n            'mask': inputs['attention_mask'][0],\n            'latitude': torch.tensor(latitude, dtype=torch.float),\n            'longitude': torch.tensor(longitude, dtype=torch.float),\n            'coord_x': torch.tensor(coord_x),\n            'coord_y': torch.tensor(coord_y),\n            'coord_z': torch.tensor(coord_z),\n        }","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:03:57.917988Z","iopub.execute_input":"2022-07-05T14:03:57.918811Z","iopub.status.idle":"2022-07-05T14:03:57.938526Z","shell.execute_reply.started":"2022-07-05T14:03:57.918656Z","shell.execute_reply":"2022-07-05T14:03:57.936869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ArcMarginProduct(nn.Module):\n    r\"\"\"Implement of large margin arc distance: :\n        Args:\n            in_features: size of each input sample\n            out_features: size of each output sample\n            s: norm of input feature\n            m: margin\n            cos(theta + m)\n        \"\"\"\n    def __init__(self, in_features, out_features, s=30.0, \n                 m=0.50, easy_margin=False, ls_eps=0.0):\n        super(ArcMarginProduct, self).__init__()\n        self.in_features = in_features\n        self.out_features = out_features\n        self.s = s\n        self.m = m\n        self.ls_eps = ls_eps  # label smoothing\n        self.weight = nn.Parameter(torch.FloatTensor(out_features, in_features))\n        nn.init.xavier_uniform_(self.weight)\n\n        self.easy_margin = easy_margin\n        self.cos_m = math.cos(m)\n        self.sin_m = math.sin(m)\n        self.th = math.cos(math.pi - m)\n        self.mm = math.sin(math.pi - m) * m\n\n    def forward(self, input, label):\n        # --------------------------- cos(theta) & phi(theta) ---------------------\n        cosine = F.linear(F.normalize(input), F.normalize(self.weight))\n        sine = torch.sqrt(1.0 - torch.pow(cosine, 2))\n        phi = cosine * self.cos_m - sine * self.sin_m\n        if self.easy_margin:\n            phi = torch.where(cosine > 0, phi, cosine)\n        else:\n            phi = torch.where(cosine > self.th, phi, cosine - self.mm)\n        # --------------------------- convert label to one-hot ---------------------\n        # one_hot = torch.zeros(cosine.size(), requires_grad=True, device='cuda')\n        one_hot = torch.zeros(cosine.size(), device=CFG.device)\n        one_hot.scatter_(1, label.view(-1, 1).long(), 1)\n        if self.ls_eps > 0:\n            one_hot = (1 - self.ls_eps) * one_hot + self.ls_eps / self.out_features\n        # -------------torch.where(out_i = {x_i if condition_i else y_i) ------------\n        output = (one_hot * phi) + ((1.0 - one_hot) * cosine)\n        output *= self.s\n\n        return output\n\nclass FSMultiModalNet(nn.Module):\n    def __init__(self, model_name, fc_dim, num_features=3):\n        super(FSMultiModalNet, self).__init__()\n        self.config = AutoConfig.from_pretrained(model_name)\n        self.bert_model = AutoModel.from_pretrained(model_name, config=self.config)\n        # self.embedding = nn.Linear(self.config.hidden_size + 2, embedding_size)\n\n        self.fc = nn.Linear(self.bert_model.config.hidden_size + num_features, fc_dim)\n        self.bn = nn.BatchNorm1d(fc_dim)\n        self._init_params()\n\n        self.margin = ArcMarginProduct(\n            fc_dim,\n            CFG.n_classes,\n            s=CFG.s, \n            m=CFG.m, \n            easy_margin=CFG.easy_margin,\n            ls_eps=CFG.ls_eps\n        )\n\n    def _init_params(self):\n        nn.init.xavier_normal_(self.fc.weight)\n        nn.init.constant_(self.fc.bias, 0)\n        nn.init.constant_(self.bn.weight, 1)\n        nn.init.constant_(self.bn.bias, 0)\n\n    def forward(self, ids, mask, lat, lon, coord_x, coord_y, coord_z, labels):\n        feature = self.extract_feature(ids, mask, lat, lon, coord_x, coord_y, coord_z)\n        output = self.margin(feature, labels)\n\n        return output\n    \n    def extract_feature(self, input_ids, attention_mask, lat, lon, coord_x, coord_y, coord_z):\n        x = self.bert_model(input_ids=input_ids, attention_mask=attention_mask)\n        x = torch.sum(x.last_hidden_state * attention_mask.unsqueeze(-1), dim=1) / attention_mask.sum(dim=1, keepdims=True)\n\n        # x = torch.cat([x, lat.view(-1, 1), lon.view(-1, 1)], axis=1)\n        x = torch.cat([x, coord_x.view(-1, 1), coord_y.view(-1, 1), coord_z.view(-1, 1)], axis=1)\n\n        x = self.fc(x)\n        x = self.bn(x)\n\n        return x","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:03:57.941093Z","iopub.execute_input":"2022-07-05T14:03:57.942210Z","iopub.status.idle":"2022-07-05T14:03:57.974802Z","shell.execute_reply.started":"2022-07-05T14:03:57.942142Z","shell.execute_reply":"2022-07-05T14:03:57.973380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_embed(model,tokenizer):\n    # NN embeddings, Multilingual \n    dataset = FourSquareDataset(df.to_pandas(), tokenizer=tokenizer, max_length=CFG.max_length)\n    loader = DataLoader(dataset, batch_size=512, num_workers=6, shuffle=False, pin_memory=True)\n\n    embeds = []\n    with torch.no_grad():\n        for data in tqdm(loader): \n            ids = data['ids'].to(CFG.device, dtype=torch.long)\n            mask = data['mask'].to(CFG.device, dtype=torch.long)\n\n            latitude = data['latitude'].to(CFG.device, dtype=torch.float)\n            longitude = data['longitude'].to(CFG.device, dtype=torch.float)\n            coord_x = data['coord_x'].to(CFG.device, dtype=torch.float)\n            coord_y = data['coord_y'].to(CFG.device, dtype=torch.float)\n            coord_z = data['coord_z'].to(CFG.device, dtype=torch.float)\n            # labels = data['label'].to(CFG.device, dtype=torch.long)\n\n            emb = model.extract_feature(ids, mask, latitude, longitude, coord_x, coord_y, coord_z)\n            embeds.append(emb.detach().cpu().numpy())\n\n    V_embed_bert = cupy.array(np.concatenate(embeds))\n    V_embed_bert = V_embed_bert / cupy.linalg.norm(V_embed_bert, ord=2, axis=1, keepdims=True)\n    \n    return V_embed_bert\n\n\ndef get_embed_model(model_name,fcdim, state_dict,tokenizer):\n\n    model = FSMultiModalNet(model_name, fcdim)\n    model.to(CFG.device);\n    model.load_state_dict(torch.load(state_dict))\n    \n    embed = get_embed(model,tokenizer)\n    \n    del model\n    gc.collect()\n    torch.cuda.empty_cache()\n    \n    return embed","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:03:57.980514Z","iopub.execute_input":"2022-07-05T14:03:57.981496Z","iopub.status.idle":"2022-07-05T14:03:57.999011Z","shell.execute_reply.started":"2022-07-05T14:03:57.981422Z","shell.execute_reply":"2022-07-05T14:03:57.997854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embed1 = get_embed_model(CFG_Model1.model_name, 320, CFG_Model1.weight, CFG_Model1.tokenizer)\ngc.collect()\ntorch.cuda.empty_cache()\n\nembed2 = get_embed_model(CFG_Model2.model_name, 320, CFG_Model2.weight, CFG_Model2.tokenizer)\ngc.collect()\ntorch.cuda.empty_cache()\n\nembed3 = get_embed_model(CFG_Model3.model_name, 320, CFG_Model3.weight, CFG_Model3.tokenizer)\ngc.collect()\ntorch.cuda.empty_cache()\n\nembed4 = get_embed_model(CFG_Model4.model_name, 320, CFG_Model4.weight, CFG_Model4.tokenizer)\ngc.collect()\ntorch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:03:58.000992Z","iopub.execute_input":"2022-07-05T14:03:58.001998Z","iopub.status.idle":"2022-07-05T14:07:50.931341Z","shell.execute_reply.started":"2022-07-05T14:03:58.001944Z","shell.execute_reply":"2022-07-05T14:07:50.930370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# l2 normalization and concat\nV_embed_concat = cupy.concatenate([\n    embed1,\n    embed2,\n    embed3,\n    embed4,\n], axis=1)\n\ndel embed1,embed2,embed3,embed4\ngc.collect()\ntorch.cuda.empty_cache()\n\nV_embed_concat = V_embed_concat / cupy.linalg.norm(V_embed_concat, ord=2, axis=1, keepdims=True)\n\ngc.collect()\ntorch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:07:50.933222Z","iopub.execute_input":"2022-07-05T14:07:50.934807Z","iopub.status.idle":"2022-07-05T14:07:51.973161Z","shell.execute_reply.started":"2022-07-05T14:07:50.934750Z","shell.execute_reply":"2022-07-05T14:07:51.972298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generate candidate rank","metadata":{}},{"cell_type":"code","source":"# Create candidate index country by country\nimport faiss\n\ndef gen_candidate_ranks(df, V_embed, n_candidates, no_country=False):\n    \n    if no_country:\n        res = faiss.StandardGpuResources()\n        index = faiss.IndexFlatIP(V_embed.shape[1])\n        index = faiss.index_cpu_to_gpu(res, 0, index)\n        index.add(V_embed)\n        D, I = index.search(V_embed, CFG.n_candidates)\n        I = np.where(I == -1, 0, I)\n    else:\n        D = np.zeros((len(df), n_candidates), dtype=np.float32)\n        I = np.zeros((len(df), n_candidates), dtype=np.int32)\n        for country, coutry_df in tqdm(df[[\"country\"]].to_pandas().groupby(\"country\")):\n            country_df = coutry_df.reset_index()\n            country_V_embed = V_embed[country_df[\"index\"].values, :]    \n            res = faiss.StandardGpuResources()\n            index = faiss.IndexFlatIP(country_V_embed.shape[1])\n            index = faiss.index_cpu_to_gpu(res, 0, index)\n            index.add(country_V_embed)\n            country_embed_d, country_embed_i = index.search(country_V_embed, n_candidates)\n            country_embed_i = np.where(country_embed_i == -1, 0, country_embed_i)\n            for i in range(n_candidates):\n                D[country_df[\"index\"].values, i] = country_embed_d[:, i]\n                I[country_df[\"index\"].values, i] = country_df.loc[country_embed_i[:, i], \"index\"].values\n            del country_df, res, index, country_embed_d, country_embed_i  # ここで del するとなんとか動く\n            torch.cuda.empty_cache()\n    D = np.clip(D, 0, 1)\n    return D, I","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:07:51.974605Z","iopub.execute_input":"2022-07-05T14:07:51.974923Z","iopub.status.idle":"2022-07-05T14:07:52.345829Z","shell.execute_reply.started":"2022-07-05T14:07:51.974882Z","shell.execute_reply":"2022-07-05T14:07:52.344676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"D_concat, I_concat = gen_candidate_ranks(df, cupy.asnumpy(V_embed_concat), CFG.n_candidates, no_country=True)\n\ngc.collect()\ntorch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:07:52.348055Z","iopub.execute_input":"2022-07-05T14:07:52.348397Z","iopub.status.idle":"2022-07-05T14:08:50.917484Z","shell.execute_reply.started":"2022-07-05T14:07:52.348349Z","shell.execute_reply":"2022-07-05T14:08:50.915914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DBA/QE weighted","metadata":{}},{"cell_type":"code","source":"# https://www.kaggle.com/code/lyakaap/2nd-place-solution/notebook\ndef query_expansion(V, D, I, alpha=3, k=2):\n    weights = cupy.array(np.expand_dims(D[:, :k] ** alpha, axis=-1).astype(np.float32))\n    chunk_size = 100_000\n    for i in range(0, len(df), chunk_size):  # chunk に分けてやらないと `cudaErrorMemoryAllocation out of memory` になる…\n        V[i:i+chunk_size] = (V[I[i:i+chunk_size, :k]] * weights[i:i+chunk_size]).sum(axis=1)\n    return V\n\n\nV_embed_concat = query_expansion(V_embed_concat, D_concat, I_concat)\nV_embed_concat /= np.linalg.norm(V_embed_concat, 2, axis=1, keepdims=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:08:50.919483Z","iopub.execute_input":"2022-07-05T14:08:50.919957Z","iopub.status.idle":"2022-07-05T14:08:52.010397Z","shell.execute_reply.started":"2022-07-05T14:08:50.919899Z","shell.execute_reply":"2022-07-05T14:08:52.008938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del D_concat,I_concat\ngc.collect()\ntorch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:08:52.019690Z","iopub.execute_input":"2022-07-05T14:08:52.024700Z","iopub.status.idle":"2022-07-05T14:08:52.548973Z","shell.execute_reply.started":"2022-07-05T14:08:52.024557Z","shell.execute_reply":"2022-07-05T14:08:52.539748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"D_concat, I_concat = gen_candidate_ranks(df, cupy.asnumpy(V_embed_concat), CFG.n_candidates, no_country=True)\n\ngc.collect()\ntorch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:08:52.550799Z","iopub.execute_input":"2022-07-05T14:08:52.551165Z","iopub.status.idle":"2022-07-05T14:08:53.213301Z","shell.execute_reply.started":"2022-07-05T14:08:52.551122Z","shell.execute_reply":"2022-07-05T14:08:53.212480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import BallTree\n\ndef gen_candidate_rank_spatial(df, n_candidates):\n    \"\"\"lat/lon から曲面上の距離 (haversine) の近い順に候補点を作成する\n    \n    Returns:\n        I (np.array): I[s_index][r] -> s_index の r 番目に近い地点の index\n    \"\"\"\n\n    I = np.full((len(df), n_candidates), -1, dtype=np.int32)\n    for country, country_df in tqdm(df[[\"country\", \"latitude\", \"longitude\"]].to_pandas().groupby(\"country\")):\n        clip_n_candidates = min(len(country_df), n_candidates)\n        country_df = country_df.reset_index()\n        ball = BallTree(country_df[[\"latitude\", \"longitude\"]].values, metric='haversine')\n        _, indices = ball.query(\n            country_df[[\"latitude\", \"longitude\"]].values, \n            k = clip_n_candidates\n        )\n        indices = np.concatenate(\n            [indices, np.zeros((len(indices), n_candidates - clip_n_candidates), dtype=np.int32)], axis=1\n        )\n        for i in range(n_candidates):\n            I[country_df[\"index\"].values, i] = country_df.loc[indices[:, i], \"index\"].values\n    return I","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:08:53.218841Z","iopub.execute_input":"2022-07-05T14:08:53.221464Z","iopub.status.idle":"2022-07-05T14:08:53.236620Z","shell.execute_reply.started":"2022-07-05T14:08:53.221400Z","shell.execute_reply":"2022-07-05T14:08:53.235766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"I_spatial = gen_candidate_rank_spatial(df, 10)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:08:53.242830Z","iopub.execute_input":"2022-07-05T14:08:53.246301Z","iopub.status.idle":"2022-07-05T14:08:53.331459Z","shell.execute_reply.started":"2022-07-05T14:08:53.246239Z","shell.execute_reply":"2022-07-05T14:08:53.330629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## tfidf","metadata":{}},{"cell_type":"code","source":"from cuml.feature_extraction.text import TfidfVectorizer\n\ntfidf = TfidfVectorizer()\nV_name = tfidf.fit_transform(df[\"name\"])\nV_name.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:08:53.359634Z","iopub.execute_input":"2022-07-05T14:08:53.360022Z","iopub.status.idle":"2022-07-05T14:09:00.584134Z","shell.execute_reply.started":"2022-07-05T14:08:53.359982Z","shell.execute_reply":"2022-07-05T14:09:00.583249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfidf = TfidfVectorizer(stop_words='english')\ndf[\"full_address\"] = df[\"address\"] + \", \" + df[\"city\"] + \", \" + df[\"state\"] + \", \"  + df[\"country\"]\nV_full_address = tfidf.fit_transform(df[\"full_address\"])\nV_full_address.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:09:00.585706Z","iopub.execute_input":"2022-07-05T14:09:00.586019Z","iopub.status.idle":"2022-07-05T14:09:00.710814Z","shell.execute_reply.started":"2022-07-05T14:09:00.585975Z","shell.execute_reply":"2022-07-05T14:09:00.709536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfidf = TfidfVectorizer()\nV_cat = tfidf.fit_transform(df[\"categories\"])\nV_cat.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:09:00.713108Z","iopub.execute_input":"2022-07-05T14:09:00.715173Z","iopub.status.idle":"2022-07-05T14:09:00.801263Z","shell.execute_reply.started":"2022-07-05T14:09:00.715109Z","shell.execute_reply":"2022-07-05T14:09:00.800402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predict","metadata":{}},{"cell_type":"code","source":"from cuml import ForestInference\n\nTHRESHOLD = 0.3","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:09:00.802754Z","iopub.execute_input":"2022-07-05T14:09:00.803093Z","iopub.status.idle":"2022-07-05T14:09:00.808514Z","shell.execute_reply.started":"2022-07-05T14:09:00.803011Z","shell.execute_reply":"2022-07-05T14:09:00.807193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"R = 6371.0  # radius of the earth in km\n\ndef manhattan(lat1, long1, lat2, long2):\n    return np.abs(lat2 - lat1) + np.abs(long2 - long1)\n\n\ndef haversine_np(lon1, lat1, lon2, lat2):\n    \"\"\"https://www.kaggle.com/code/justfor/speedup-haversine/script\n    \n    Calculate the great circle distance between two points\n    on the earth (specified in decimal degrees)\n\n    All args must be of equal length.    \n\n    \"\"\"\n    lon1, lat1, lon2, lat2 = map(np.radians, [lon1, lat1, lon2, lat2])\n\n    dlon = lon2 - lon1\n    dlat = lat2 - lat1\n\n    a = np.sin(dlat/2.0)**2 + np.cos(lat1) * np.cos(lat2) * np.sin(dlon/2.0)**2\n\n    c = 2 * np.arcsin(np.sqrt(a))\n    km = R * c\n    \n    return km\n\n\ndef create_features(df, i, indices):\n    \n    prev_i = max(i-1, 0)\n    next_i = min(i+1, indices.shape[1] - 1)\n    prev_cand_index = indices[:, prev_i]\n    next_cand_index = indices[:, next_i]\n    \n    cand_index = indices[:, i]\n    \n    lon1 = df[\"longitude\"].to_pandas().to_numpy()\n    lat1 = df[\"latitude\"].to_pandas().to_numpy()\n    lon2 = df[\"longitude\"][cand_index].to_pandas().to_numpy()\n    lat2 = df[\"latitude\"][cand_index].to_pandas().to_numpy()\n    #df[\"diff_lon\"] = lon1 - lon2\n    #df[\"diff_lat\"] = lat1 - lat2\n    df[\"diff_lon\"] = np.abs(lon1 - lon2)\n    df[\"diff_lat\"] = np.abs(lat1 - lat2)\n    \n    df[\"lonlat_eucdist\"] =  (df['diff_lon'] ** 2 + df['diff_lat'] ** 2) ** 0.5\n    df[\"lonlat_manhattan\"] = manhattan(lat1, lon2, lat2, lon2)\n    df[\"lonlat_haversine_dist\"] = haversine_np(lon1, lat1, lon2, lat2)\n    \n    df[\"name_cossim\"] = V_name.multiply(V_name[cand_index]).sum(axis=1).ravel()\n    df[\"full_address_cossim\"] = V_full_address.multiply(V_full_address[cand_index]).sum(axis=1).ravel()\n    df[\"cat_cossim\"] = V_cat.multiply(V_cat[cand_index]).sum(axis=1).ravel()\n    \n    df[\"cand_hit_count_02\"] = df[\"hit_count_02\"][cand_index].to_pandas().to_numpy()\n    df[\"cand_hit_count_03\"] = df[\"hit_count_03\"][cand_index].to_pandas().to_numpy()\n    df[\"cand_hit_count_04\"] = df[\"hit_count_04\"][cand_index].to_pandas().to_numpy()\n    df[\"cand_hit_count_05\"] = df[\"hit_count_05\"][cand_index].to_pandas().to_numpy()\n    df[\"cand_hit_count_sum\"] = df[\"hit_count_sum\"][cand_index].to_pandas().to_numpy()\n    \n    df[\"hit_count_02_min\"] = df[[\"hit_count_02\", \"cand_hit_count_02\"]].min(axis=1)\n    df[\"hit_count_03_min\"] = df[[\"hit_count_03\", \"cand_hit_count_03\"]].min(axis=1)\n    df[\"hit_count_04_min\"] = df[[\"hit_count_04\", \"cand_hit_count_04\"]].min(axis=1)\n    df[\"hit_count_05_min\"] = df[[\"hit_count_05\", \"cand_hit_count_05\"]].min(axis=1)\n    df[\"hit_count_sum_min\"] = df[[\"hit_count_sum\", \"cand_hit_count_sum\"]].min(axis=1)\n    \n    df[\"hit_count_02_max\"] = df[[\"hit_count_02\", \"cand_hit_count_02\"]].max(axis=1)\n    df[\"hit_count_03_max\"] = df[[\"hit_count_03\", \"cand_hit_count_03\"]].max(axis=1)\n    df[\"hit_count_04_max\"] = df[[\"hit_count_04\", \"cand_hit_count_04\"]].max(axis=1)\n    df[\"hit_count_05_max\"] = df[[\"hit_count_05\", \"cand_hit_count_05\"]].max(axis=1)\n    df[\"hit_count_sum_max\"] = df[[\"hit_count_sum\", \"cand_hit_count_sum\"]].max(axis=1)\n    \n    cossim = []\n    eucdist = []\n    \n    eucdist1=[]\n    eucdist2=[]\n    eucdist3=[]\n    eucdist4=[]\n    eucdist5=[]\n    \n    chunk_size = 100_000\n    for i in range(0, len(df), chunk_size):  # chunk に分けてやらないと `cudaErrorMemoryAllocation out of memory` になる…\n        cossim.append(cupy.multiply(V_embed_concat[i:i+chunk_size], V_embed_concat[cand_index[i:i+chunk_size]]).sum(axis=1))\n        eucdist.append(cupy.sqrt(((V_embed_concat[i:i+chunk_size] - V_embed_concat[cand_index[i:i+chunk_size]]) ** 2).sum(axis=1)))\n        \n        eucdist1.append(cupy.sqrt(((V_embed_concat[prev_cand_index[i:i+chunk_size]] - V_embed_concat[cand_index[i:i+chunk_size]]) ** 2).sum(axis=1)))\n        eucdist2.append(cupy.sqrt(((V_embed_concat[next_cand_index[i:i+chunk_size]] - V_embed_concat[cand_index[i:i+chunk_size]]) ** 2).sum(axis=1)))\n        eucdist3.append(cupy.sqrt(((V_embed_concat[i:i+chunk_size]                  - V_embed_concat[prev_cand_index[i:i+chunk_size]]) ** 2).sum(axis=1)))\n        eucdist4.append(cupy.sqrt(((V_embed_concat[i:i+chunk_size]                  - V_embed_concat[next_cand_index[i:i+chunk_size]]) ** 2).sum(axis=1)))\n        eucdist5.append(cupy.sqrt(((V_embed_concat[prev_cand_index[i:i+chunk_size]] - V_embed_concat[next_cand_index[i:i+chunk_size]]) ** 2).sum(axis=1)))\n\n    df[\"embed_cossim\"] = cupy.concatenate(cossim)\n    df[\"embed_eucdist\"] = cupy.concatenate(eucdist)\n    \n    df[\"embed_eucdist1\"] = cupy.concatenate(eucdist1)\n    df[\"embed_eucdist2\"] = cupy.concatenate(eucdist2)\n    df[\"embed_eucdist3\"] = cupy.concatenate(eucdist3)\n    df[\"embed_eucdist4\"] = cupy.concatenate(eucdist4)\n    df[\"embed_eucdist5\"] = cupy.concatenate(eucdist5)\n    \n    df[\"d0_d1\"] = df[\"embed_eucdist\"] - df[\"embed_eucdist1\"]\n    df[\"d0_d2\"] = df[\"embed_eucdist\"] - df[\"embed_eucdist2\"]\n    df[\"d0_d3\"] = df[\"embed_eucdist\"] - df[\"embed_eucdist3\"]\n    df[\"d0_d4\"] = df[\"embed_eucdist\"] - df[\"embed_eucdist4\"]\n    df[\"d0_d5\"] = df[\"embed_eucdist\"] - df[\"embed_eucdist5\"]\n    \n    for col in [\"id\", \"name\", \"address\", \"city\", \"state\", \"zip\", \"country\", \"url\", \"phone\", \"categories\", \"full_address\"]:\n        df[f\"{col}_edit_dist\"] = df[col].str.edit_distance(df[col][cand_index])\n        df[f\"norm_{col}_edit_dist\"] = df[f\"{col}_edit_dist\"] / df[col].str.len()\n        df[f\"norm_{col}_edit_dist\"] = df[f\"norm_{col}_edit_dist\"].replace([np.inf, -np.inf], 0)\n    \n    features = [\n        \"diff_lon\",\n        \"diff_lat\",\n        \"lonlat_eucdist\",\n        \"lonlat_manhattan\",\n        \"lonlat_haversine_dist\",\n        \"name_cossim\",\n        \"full_address_cossim\",\n        \"cat_cossim\",\n        \"embed_cossim\",\n        \"embed_eucdist\",\n        \n        \"embed_eucdist1\",\n        \"embed_eucdist2\",\n        \"embed_eucdist3\",\n        \"embed_eucdist4\",\n        \"embed_eucdist5\",\n        \"d0_d1\",\n        \"d0_d2\",\n        \"d0_d3\",\n        \"d0_d4\",\n        \"d0_d5\",\n\n        \"hit_count_02\",\n        \"hit_count_03\",\n        \"hit_count_04\",\n        \"hit_count_05\",\n        \"hit_count_sum\",\n        \n        \"hit_count_02_min\",\n        \"hit_count_03_min\",\n        \"hit_count_04_min\",\n        \"hit_count_05_min\",\n        \"hit_count_sum_min\",\n        \"hit_count_02_max\",\n        \"hit_count_03_max\",\n        \"hit_count_04_max\",\n        \"hit_count_05_max\",\n        \"hit_count_sum_max\"\n    ]\n    \n    for col in [\"id\", \"name\", \"address\", \"city\", \"state\", \"zip\", \"country\", \"url\", \"phone\", \"categories\", \"full_address\"]:\n        features.append(f\"{col}_edit_dist\")\n        features.append(f\"norm_{col}_edit_dist\")\n    \n    return df, features","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:09:00.810815Z","iopub.execute_input":"2022-07-05T14:09:00.811419Z","iopub.status.idle":"2022-07-05T14:09:00.858309Z","shell.execute_reply.started":"2022-07-05T14:09:00.811372Z","shell.execute_reply":"2022-07-05T14:09:00.857377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import defaultdict\n\ndfs = []\n\ndef train(I, n_candiates, prefix, link):\n    # 0で初期化しておく\n    df[\"hit_count_02\"] = 0\n    df[\"hit_count_03\"] = 0\n    df[\"hit_count_04\"] = 0\n    df[\"hit_count_05\"] = 0\n    df[\"hit_count_sum\"] = 0\n    \n    df[\"cand_hit_count_02\"] = 0\n    df[\"cand_hit_count_03\"] = 0\n    df[\"cand_hit_count_04\"] = 0\n    df[\"cand_hit_count_05\"] = 0\n    df[\"cand_hit_count_sum\"] = 0\n    \n    for i in range(I.shape[1]):\n        print(f\"Candidate rank {i} ...\")\n        tmp_df = df.copy()\n        tmp_df[\"match_id\"] = tmp_df[\"id\"].to_pandas().values[I[:, i]]\n\n        tmp_df, features = create_features(tmp_df, i, I)\n        tmp_df[\"pred\"] = 0\n        for fold in range(CFG.n_splits):\n            if CFG.one_fold and fold != 0:\n                continue\n            print(f\"    ===== fold{fold} =====\")\n            xgb_model = ForestInference.load(\n                #f\"../input/fs-4-ensemble-xgb/fs_xgb_model_{prefix}_candidate{i}_fold{fold}.json\",\n                f\"../input/xgb-4ensemble-0705-v2/fs_xgb_model_{prefix}_candidate{i}_fold{fold}.json\",\n                output_class=True,\n                model_type=\"xgboost_json\"\n            )\n\n            pred = xgb_model.predict_proba(tmp_df[features].to_pandas())[:, 1]\n            if CFG.one_fold:\n                tmp_df[\"pred\"] = pred\n            else:\n                tmp_df[\"pred\"] += pred / CFG.n_splits\n\n#         dfs.append(tmp_df[tmp_df[\"pred\"] > THRESHOLD][[\"id\", \"match_id\", \"pred\"]])\n\n        df[\"hit_count_02\"] += (pred > 0.2)\n        df[\"hit_count_03\"] += (pred > 0.3)\n        df[\"hit_count_04\"] += (pred > 0.4)\n        df[\"hit_count_05\"] += (pred > 0.5)\n        df[\"hit_count_sum\"] += pred\n        \n        ids = tmp_df.to_pandas().id.values\n        match_ids = tmp_df.to_pandas().match_id.values\n        pred_lis = tmp_df.to_pandas().pred.values\n        for i in range(len(tmp_df)):\n            dist = 1 - pred_lis[i]\n            if dist > 0.9:\n                continue\n            link[ids[i]][match_ids[i]] = min(dist, link[ids[i]][match_ids[i]])\n        \n    return link","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:10:00.786709Z","iopub.execute_input":"2022-07-05T14:10:00.787533Z","iopub.status.idle":"2022-07-05T14:10:00.803906Z","shell.execute_reply.started":"2022-07-05T14:10:00.787491Z","shell.execute_reply":"2022-07-05T14:10:00.802877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import defaultdict\n\nlink = defaultdict(lambda:defaultdict(lambda: 1))\n\nlink = train(I_concat, CFG.n_candidates, \"embed\", link)\nlink = train(I_spatial, CFG.n_spatial_candidates, \"spatial\", link)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:10:01.691122Z","iopub.execute_input":"2022-07-05T14:10:01.692182Z","iopub.status.idle":"2022-07-05T14:10:30.468561Z","shell.execute_reply.started":"2022-07-05T14:10:01.692123Z","shell.execute_reply":"2022-07-05T14:10:30.467293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_COST=0.4","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:10:30.475798Z","iopub.execute_input":"2022-07-05T14:10:30.479576Z","iopub.status.idle":"2022-07-05T14:10:30.488122Z","shell.execute_reply.started":"2022-07-05T14:10:30.479512Z","shell.execute_reply":"2022-07-05T14:10:30.486073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from heapq import heappush, heappop\n\ndef dks(link, start, MAX_COST):\n    visited = set()\n    hq = []\n    heappush(hq, (0,start))\n\n    while hq:\n        shortest, now = heappop(hq)\n        \n        if (shortest > MAX_COST) :\n            break\n        \n        if now in visited:\n            continue\n        visited.add(now)\n            \n        for nxt, next_cost in link[now].items():\n            if nxt in visited:\n                continue\n            inv_next_cost = link[nxt][now]\n            heappush(hq, (shortest + (next_cost + inv_next_cost) / 2.0, nxt))\n            \n    return list(visited)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:10:30.496015Z","iopub.execute_input":"2022-07-05T14:10:30.497707Z","iopub.status.idle":"2022-07-05T14:10:30.513248Z","shell.execute_reply.started":"2022-07-05T14:10:30.497453Z","shell.execute_reply":"2022-07-05T14:10:30.512267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_lis = []\nmatches = []\nfor source_id in tqdm(set(df.to_pandas().id)):\n    ret=dks(link,source_id,MAX_COST)\n    matches.append(ret)\n    \n    id_lis.append(source_id)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:10:30.521087Z","iopub.execute_input":"2022-07-05T14:10:30.521938Z","iopub.status.idle":"2022-07-05T14:10:30.562221Z","shell.execute_reply.started":"2022-07-05T14:10:30.521884Z","shell.execute_reply":"2022-07-05T14:10:30.561378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dks_df = pd.DataFrame()\ndks_df[\"id\"] = id_lis\ndks_df[\"match_id\"] = matches\ndks_df[\"matches\"] = dks_df[\"match_id\"].apply(lambda x: \" \".join(set(x)))\n\ndks_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:10:30.565854Z","iopub.execute_input":"2022-07-05T14:10:30.569746Z","iopub.status.idle":"2022-07-05T14:10:30.600117Z","shell.execute_reply.started":"2022-07-05T14:10:30.569688Z","shell.execute_reply":"2022-07-05T14:10:30.599201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dks_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:10:30.605103Z","iopub.execute_input":"2022-07-05T14:10:30.605787Z","iopub.status.idle":"2022-07-05T14:10:30.621149Z","shell.execute_reply.started":"2022-07-05T14:10:30.605735Z","shell.execute_reply":"2022-07-05T14:10:30.620081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dks_df.to_csv(\"submission.csv\", index=False, columns=[\"id\", \"matches\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:10:30.626543Z","iopub.execute_input":"2022-07-05T14:10:30.627203Z","iopub.status.idle":"2022-07-05T14:10:30.645006Z","shell.execute_reply.started":"2022-07-05T14:10:30.627154Z","shell.execute_reply":"2022-07-05T14:10:30.644111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}