{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Here is my solution for this competition.\n\nIt received a silver medal without any ensembling or complicated feature engineering.\n\nThe pipeline is is simple:\n\n1. Train xlmroberta with ArcFace Loss\n2. Use the cos sims from the xlmroberta + coordinate distance to extract match candidates \n3. Add features (cos sim, distance, lcs, tfidf, etc...)\n4. Train a lightgbm model (with flaml hyperparameter optimization) to select the correct candidates as a binary classification task.\n5. Do 2-3 on the test data and inference with lgbm","metadata":{}},{"cell_type":"code","source":"import math \nimport torch\nimport pickle\nimport torch.nn as nn\nimport pandas as pd\nimport treelite\nimport treelite_runtime \nfrom numba import jit\nfrom tqdm import tqdm \nfrom cuml import ForestInference\nfrom lightgbm import LGBMClassifier\nfrom transformers import AutoTokenizer\nfrom sklearn.neighbors import NearestNeighbors\nfrom transformers import AutoModelForSequenceClassification, AutoConfig","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:47.07407Z","iopub.execute_input":"2022-06-22T06:03:47.074575Z","iopub.status.idle":"2022-06-22T06:03:47.079733Z","shell.execute_reply.started":"2022-06-22T06:03:47.074537Z","shell.execute_reply":"2022-06-22T06:03:47.079016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class FourSquareModel(nn.Module):\n    def __init__(self):\n        super(FourSquareModel, self).__init__()\n        config = AutoConfig.from_pretrained('../input/xlm-roberta-squad2/deepset/xlm-roberta-base-squad2', num_labels=128)\n        self.transformer = AutoModelForSequenceClassification.from_config(config)\n        \n    def forward(self, ids, mask):\n        embd = self.transformer(input_ids=ids.squeeze(), attention_mask=mask.squeeze()).logits\n        return embd","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:47.103753Z","iopub.execute_input":"2022-06-22T06:03:47.10408Z","iopub.status.idle":"2022-06-22T06:03:47.109567Z","shell.execute_reply.started":"2022-06-22T06:03:47.104047Z","shell.execute_reply":"2022-06-22T06:03:47.108866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_text_embd(df, model, tokenizer, device):\n    vectors = []\n\n    # Split dataframe into chunks\n    chunks = [df[i:i+128] for i in range(0, len(df), 128)]\n\n    for chunk in tqdm(chunks):\n        input_ids = []\n        attention_masks = []\n\n        for index, row in chunk.iterrows():\n            text = [row['name'], row['address'], row['city'], row['state'],\n                            row['zip'], row['country'], row['url'], row['phone'], row['categories']]\n            text = [str(x) for x in text]\n            text = ' '.join(text)\n            input = tokenizer(\n                text,\n                truncation=True,\n                max_length=128,\n                padding='max_length',\n                return_tensors=\"pt\"\n            )\n            input_ids.append(input['input_ids'])\n            attention_masks.append(input['attention_mask'])\n\n        input_ids = torch.cat(input_ids, dim=0)\n        attention_masks = torch.cat(attention_masks, dim=0)\n\n        with torch.no_grad():\n            embd = model(input_ids.to(device), attention_masks.to(device))\n\n        vectors.append(embd) \n\n    vectors = torch.cat(vectors, 0)\n    vectors = torch.nn.functional.normalize(vectors, p=2, dim=-1)\n    return vectors","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:47.128517Z","iopub.execute_input":"2022-06-22T06:03:47.1288Z","iopub.status.idle":"2022-06-22T06:03:47.138662Z","shell.execute_reply.started":"2022-06-22T06:03:47.128771Z","shell.execute_reply":"2022-06-22T06:03:47.137718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_geospatial_neightbors(k, df):\n    knn = NearestNeighbors(n_neighbors = k)\n    knn.fit(df[['latitude','longitude']], df.index)\n    dists, nears = knn.kneighbors(df[['latitude','longitude']])\n    return dists, nears","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:47.157681Z","iopub.execute_input":"2022-06-22T06:03:47.157899Z","iopub.status.idle":"2022-06-22T06:03:47.162655Z","shell.execute_reply.started":"2022-06-22T06:03:47.157874Z","shell.execute_reply":"2022-06-22T06:03:47.161933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_cos_sims(k, vectors, nears):\n    topk_indices = []\n    cos_sims_word = []\n    cos_sims_dist = []\n    cos = nn.CosineSimilarity(dim=1, eps=1e-6)\n\n    for i in tqdm(range(math.ceil(vectors.shape[0]/1000))):\n        end = min(1000*(i+1), vectors.shape[0])\n        dist = torch.mm(vectors[1000*i:end], vectors.t())    \n        value, indices = torch.topk(dist, k, dim=-1)\n        \n        temp = []\n        for j in range(1000*i, end):\n            temp.append(dist[j-1000*i][nears[j]])\n        temp = torch.stack(temp, 0)\n\n        cos_sims_dist.append(temp)\n        cos_sims_word.append(value)\n        topk_indices.append(indices)\n\n    topk_indices = torch.cat(topk_indices, 0).cpu().numpy()\n    cos_sims_word = torch.cat(cos_sims_word, 0).cpu().numpy()\n    cos_sims_dist = torch.cat(cos_sims_dist, 0).cpu().numpy()\n    return topk_indices, cos_sims_word, cos_sims_dist","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:47.181161Z","iopub.execute_input":"2022-06-22T06:03:47.181437Z","iopub.status.idle":"2022-06-22T06:03:47.19167Z","shell.execute_reply.started":"2022-06-22T06:03:47.181409Z","shell.execute_reply":"2022-06-22T06:03:47.190958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_pair_df(df, k, topk_indices, cos_sims_word, cos_sims_dist, nears, dists):\n    test_df = []\n\n    for i in range(k):  \n        # Add based on geospatial distance\n        cur_df = df[['id']]\n        cur_df['match_id'] = df['id'].values[nears[:, i]]\n        cur_df['cos_sim'] = cos_sims_dist[:,i]\n        cur_df['geospatial_k'] = k - i\n        cur_df['dist'] = dists[:, i]\n        if i > 10:\n            cur_df = cur_df[cur_df['dist'] < 2]\n        cur_df = cur_df.drop(['dist'], axis=1)\n        test_df.append(cur_df)\n        \n        # Add based on word similarity\n        cur_df = df[['id']]\n        cur_df['match_id'] = df['id'].values[topk_indices[:, i]]\n        cur_df['cos_sim'] = cos_sims_word[:,i]\n        cur_df['cos_sim_k'] = k - i\n        if i > 10:\n            cur_df = cur_df[cur_df['cos_sim'] > 0.5]\n        test_df.append(cur_df)\n\n    test_df = pd.concat(test_df)\n    test_df = test_df.drop_duplicates()\n    return test_df","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:47.204848Z","iopub.execute_input":"2022-06-22T06:03:47.205079Z","iopub.status.idle":"2022-06-22T06:03:47.214121Z","shell.execute_reply.started":"2022-06-22T06:03:47.205053Z","shell.execute_reply":"2022-06-22T06:03:47.213158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_features(test_df, df):\n    df = df[['id','longitude', 'latitude']]\n    test_df = test_df.merge(df, on='id')\n    test_df = test_df.rename(columns={\"longitude\":\"longitude_1\", 'latitude':'latitude_1'})\n    test_df = test_df.merge(df, left_on='match_id', right_on='id')\n    test_df = test_df.rename(columns={\"longitude\":\"longitude_2\", 'latitude':'latitude_2', 'id_x':'id'})\n    test_df = test_df.drop(['id_y'], axis=1)\n    \n    test_df['longitude_diff'] = abs(test_df['longitude_2'] - test_df['longitude_1'])\n    test_df['latitude_diff'] = abs(test_df['latitude_2'] - test_df['latitude_1'])\n    test_df['euc_dist'] = (test_df['longitude_diff'] ** 2 + test_df['latitude_diff'] ** 2) ** 0.5\n    test_df['manhattan_dist'] = test_df['longitude_diff'] + test_df['latitude_diff']\n    return test_df","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:47.255106Z","iopub.execute_input":"2022-06-22T06:03:47.255554Z","iopub.status.idle":"2022-06-22T06:03:47.262898Z","shell.execute_reply.started":"2022-06-22T06:03:47.255522Z","shell.execute_reply":"2022-06-22T06:03:47.262221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%load_ext Cython","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:47.264704Z","iopub.execute_input":"2022-06-22T06:03:47.265905Z","iopub.status.idle":"2022-06-22T06:03:47.273677Z","shell.execute_reply.started":"2022-06-22T06:03:47.265844Z","shell.execute_reply":"2022-06-22T06:03:47.272737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%cython\nimport numpy as np  # noqa\ncpdef int LCS(str S, str T):\n    cdef int i, j\n    cdef int cost\n    cdef int v1,v2,v3,v4\n    cdef int[:, :] dp = np.zeros((len(S) + 1, len(T) + 1), dtype=np.int32)\n    for i in range(len(S)):\n        for j in range(len(T)):\n            cost = (int)(S[i] == T[j])\n            v1 = dp[i, j] + cost\n            v2 = dp[i + 1, j]\n            v3 = dp[i, j + 1]\n            v4 = dp[i + 1, j + 1]\n            dp[i + 1, j + 1] = max((v1,v2,v3,v4))\n    return dp[len(S)][len(T)]","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:47.303377Z","iopub.execute_input":"2022-06-22T06:03:47.303685Z","iopub.status.idle":"2022-06-22T06:03:47.309683Z","shell.execute_reply.started":"2022-06-22T06:03:47.303655Z","shell.execute_reply":"2022-06-22T06:03:47.308831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_columns = ['name', 'address', 'city', \n            'state', 'zip', 'url', \n           'phone', 'categories', 'country']\nvec_columns = ['name', 'categories', 'address', \n               'state', 'url', 'country']\n\ndef create_more_features(df, test_df, tfidf_d, id2index_d):    \n    for col in feat_columns:       \n        if col in vec_columns:\n            tv_fit = tfidf_d[col]\n            indexs = [id2index_d[i] for i in test_df['id']]\n            match_indexs = [id2index_d[i] for i in test_df['match_id']]\n            test_df[f'{col}_sim'] = np.array(tv_fit[indexs].multiply(tv_fit[match_indexs]).sum(axis = 1)).ravel()\n        \n        col_values = df.loc[test_df['id']][col].values.astype(str)\n        matcol_values = df.loc[test_df['match_id']][col].values.astype(str)\n        \n        geshs = []\n        levens = []\n        jaros = []\n        lcss = []\n        for s, match_s in zip(col_values, matcol_values):\n            if s != 'nan' and match_s != 'nan':                    \n                geshs.append(difflib.SequenceMatcher(None, s, match_s).ratio())\n                levens.append(Levenshtein.distance(s, match_s))\n                jaros.append(Levenshtein.jaro_winkler(s, match_s))\n                lcss.append(LCS(str(s), str(match_s)))\n            else:\n                geshs.append(np.nan)\n                levens.append(np.nan)\n                jaros.append(np.nan)\n                lcss.append(np.nan)\n        \n        test_df[f'{col}_gesh'] = geshs\n        test_df[f'{col}_leven'] = levens\n        test_df[f'{col}_jaro'] = jaros\n        test_df[f'{col}_lcs'] = lcss\n        \n        if col not in ['phone', 'zip']:\n            test_df[f'{col}_len'] = list(map(len, col_values))\n            test_df[f'match_{col}_len'] = list(map(len, matcol_values)) \n            test_df[f'{col}_len_diff'] = np.abs(test_df[f'{col}_len'] - test_df[f'match_{col}_len'])\n            test_df[f'{col}_nleven'] = test_df[f'{col}_leven'] / \\\n                                    test_df[[f'{col}_len', f'match_{col}_len']].max(axis = 1)\n            \n            test_df[f'{col}_nlcsk'] = test_df[f'{col}_lcs'] / test_df[f'match_{col}_len']\n            test_df[f'{col}_nlcs'] = test_df[f'{col}_lcs'] / test_df[f'{col}_len']\n            \n            test_df = test_df.drop(f'{col}_len', axis = 1)\n            test_df = test_df.drop(f'match_{col}_len', axis = 1)\n            gc.collect()\n            \n    return test_df","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:47.32841Z","iopub.execute_input":"2022-06-22T06:03:47.328707Z","iopub.status.idle":"2022-06-22T06:03:47.344998Z","shell.execute_reply.started":"2022-06-22T06:03:47.328676Z","shell.execute_reply":"2022-06-22T06:03:47.344211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def infer(test_df, ml_models, use_treelite=False):\n    inference_df = test_df.drop(['id', 'match_id'], axis=1)\n\n    # inference with treelite\n    if use_treelite:\n        batch_size = 64\n        X = inference_df.values    \n        step = X.shape[0]//batch_size\n        if batch_size*step < X.shape[0]:\n            step += 1\n        ret = []\n        start = 0\n        for i in range(step):\n            end = start + batch_size \n            preds = []\n            for i, ml_model in enumerate(ml_models):\n                preds.append(ml_model.predict(X[start:end,:]))\n            pred = np.average(preds, 0)\n            ret.append(pred)\n            start += batch_size\n        result = np.concatenate(ret)    \n    else:\n        result = ml_model.predict(inference_df)\n        \n    test_df['match'] = result\n    return test_df[['id', 'match_id', 'match']]","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:47.351319Z","iopub.execute_input":"2022-06-22T06:03:47.351892Z","iopub.status.idle":"2022-06-22T06:03:47.361427Z","shell.execute_reply.started":"2022-06-22T06:03:47.35186Z","shell.execute_reply":"2022-06-22T06:03:47.360566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def post_process(df):\n    id2match = dict(zip(df['id'].values, df['matches'].str.split()))\n\n    for base, match in df[['id', 'matches']].values:\n        match = match.split()\n        if len(match) == 1:        \n            continue\n\n        for m in match:\n            if base not in id2match[m]:\n                id2match[m].append(base)\n    df['matches'] = df['id'].map(id2match).map(' '.join)\n    return df ","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:47.37541Z","iopub.execute_input":"2022-06-22T06:03:47.375825Z","iopub.status.idle":"2022-06-22T06:03:47.38161Z","shell.execute_reply.started":"2022-06-22T06:03:47.375793Z","shell.execute_reply":"2022-06-22T06:03:47.38082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_submission(test_df):\n    test_df = test_df[test_df['match']]\n    out_df = test_df[['id', 'match_id', 'match']]\n    out_df = out_df[out_df['match'].astype(bool)]\n    out_df = out_df.groupby('id')['match_id'].\\\n                            apply(list).reset_index()\n    out_df['matches'] = out_df['match_id'].apply(lambda x: ' '.join(set(x)))\n    out_df = post_process(out_df)\n    return out_df[['id', 'matches']]    ","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:47.397934Z","iopub.execute_input":"2022-06-22T06:03:47.398351Z","iopub.status.idle":"2022-06-22T06:03:47.404674Z","shell.execute_reply.started":"2022-06-22T06:03:47.398322Z","shell.execute_reply":"2022-06-22T06:03:47.403803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df_path = '../input/foursquare-location-matching/test.csv'\ndf = pd.read_csv(test_df_path)\ndf = df.fillna('')","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:47.421078Z","iopub.execute_input":"2022-06-22T06:03:47.421586Z","iopub.status.idle":"2022-06-22T06:03:47.432779Z","shell.execute_reply.started":"2022-06-22T06:03:47.421553Z","shell.execute_reply":"2022-06-22T06:03:47.431908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = FourSquareModel()\nmodel.transformer.load_state_dict(torch.load('../input/foursquare-weights/transformer_new_new.pth'))\nmodel.eval()\n\ntokenizer = AutoTokenizer.from_pretrained('../input/xlm-roberta-squad2/deepset/xlm-roberta-base-squad2')\ndevice = torch.device(\"cuda\")\nmodel.to(device) ## model to GPU\n\nvectors = create_text_embd(df, model, tokenizer, device)\ndel tokenizer","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:47.445263Z","iopub.execute_input":"2022-06-22T06:03:47.445574Z","iopub.status.idle":"2022-06-22T06:03:56.343895Z","shell.execute_reply.started":"2022-06-22T06:03:47.445543Z","shell.execute_reply":"2022-06-22T06:03:56.342923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"k = 10\nk = min(len(df), k)\n\ndists, nears = get_geospatial_neightbors(k, df)\ntopk_indices, cos_sims_word, cos_sims_dist = get_cos_sims(k, vectors, nears)\ntest_df = create_pair_df(df, k, topk_indices, cos_sims_word, cos_sims_dist, nears, dists)","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:56.348481Z","iopub.execute_input":"2022-06-22T06:03:56.351005Z","iopub.status.idle":"2022-06-22T06:03:56.413671Z","shell.execute_reply.started":"2022-06-22T06:03:56.350958Z","shell.execute_reply":"2022-06-22T06:03:56.412887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = create_features(test_df, df)","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:56.415283Z","iopub.execute_input":"2022-06-22T06:03:56.415957Z","iopub.status.idle":"2022-06-22T06:03:56.457974Z","shell.execute_reply.started":"2022-06-22T06:03:56.41591Z","shell.execute_reply":"2022-06-22T06:03:56.457196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nimport Levenshtein\nimport difflib\nimport gc\n\ntfidf_d = {}\nfor col in vec_columns:\n    tfidf = TfidfVectorizer()\n    tv_fit = tfidf.fit_transform(df[col].fillna('nan'))\n    tfidf_d[col] = tv_fit\nid2index_d = dict(zip(df['id'].values, df.index))\ndf = df.set_index('id')","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:56.460264Z","iopub.execute_input":"2022-06-22T06:03:56.460716Z","iopub.status.idle":"2022-06-22T06:03:56.505009Z","shell.execute_reply.started":"2022-06-22T06:03:56.460674Z","shell.execute_reply":"2022-06-22T06:03:56.5015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import dask\nfrom dask.diagnostics import ProgressBar\n\nml_model = ForestInference.load(filename='../input/foursquare-weights/lgbm_20_0.txt',model_type='lightgbm')\n\nchunks = [test_df[i:i+10000] for i in range(0, len(test_df), 10000)]\n\nfor i, chunk in tqdm(enumerate(chunks)):\n    chunk = create_more_features(df, chunk, tfidf_d, id2index_d)\n    chunk = infer(chunk, [ml_model], True)\n    chunks[i] = chunk\n    \ntest_df = pd.concat(chunks)","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:03:56.506703Z","iopub.execute_input":"2022-06-22T06:03:56.508326Z","iopub.status.idle":"2022-06-22T06:04:02.198757Z","shell.execute_reply.started":"2022-06-22T06:03:56.508282Z","shell.execute_reply":"2022-06-22T06:04:02.198039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['match'] = test_df['match']>0.55\ntest_df['match'] = test_df['match'].astype(bool)\ntest_df['match'] = test_df['match'] | (test_df['id'] == test_df['match_id'])","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:04:02.247393Z","iopub.execute_input":"2022-06-22T06:04:02.247683Z","iopub.status.idle":"2022-06-22T06:04:02.270515Z","shell.execute_reply.started":"2022-06-22T06:04:02.247641Z","shell.execute_reply":"2022-06-22T06:04:02.269772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"out_df = create_submission(test_df)\nout_df.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-06-22T06:04:02.272511Z","iopub.execute_input":"2022-06-22T06:04:02.272889Z","iopub.status.idle":"2022-06-22T06:04:02.32161Z","shell.execute_reply.started":"2022-06-22T06:04:02.272849Z","shell.execute_reply":"2022-06-22T06:04:02.32058Z"},"trusted":true},"execution_count":null,"outputs":[]}]}