{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Reference**\n\nThis notebook is based on [@ryotayoshinobu](https://www.kaggle.com/ryotayoshinobu)'s [baseline](https://www.kaggle.com/code/ryotayoshinobu/foursquare-lightgbm-baseline).\n\nI joined this competition in the middle of the competition period, but thanks to [@ryotayoshinobu](https://www.kaggle.com/ryotayoshinobu), I was able to work on it smoothly.\n\nPlease don't forget to upvote the original notebook!","metadata":{"id":"MepXK1V1eF5M"}},{"cell_type":"markdown","source":"**about this notebook**\n\nOn both Kaggle and Colab, training and inference can be run on this single notebook!\n\n1. Training\n    \n    Set `CFG.train = True` and run.\n\n2. Inference\n\n    Set `CFG.train = False` and run.","metadata":{}},{"cell_type":"code","source":"!nvidia-smi","metadata":{"id":"0ff71a01","outputId":"4e6c67a0-194d-4091-dd2b-2f50ac012303","execution":{"iopub.status.busy":"2022-07-09T21:02:43.704235Z","iopub.execute_input":"2022-07-09T21:02:43.705414Z","iopub.status.idle":"2022-07-09T21:02:44.600438Z","shell.execute_reply.started":"2022-07-09T21:02:43.705293Z","shell.execute_reply":"2022-07-09T21:02:44.599421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Libraries","metadata":{"id":"50361d58"}},{"cell_type":"code","source":"# ====================================================\n# import libraries1\n# ====================================================\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport os\nimport sys\nimport math\nimport random\nimport time\nimport numpy as np\nimport pandas as pd\nimport gc\nimport json\nimport joblib\nfrom tqdm import tqdm\nfrom pathlib import Path\nimport itertools\nimport collections\nfrom collections import Counter\n\nimport torch\nimport torch.nn.functional as F\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader, Dataset\nimport torch.optim as optim\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.optim.lr_scheduler import CosineAnnealingWarmRestarts, CosineAnnealingLR, ReduceLROnPlateau, _LRScheduler\nfrom torch.nn import Parameter\n\nimport datetime\nfrom datetime import timedelta\nimport hashlib\nimport difflib\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom requests import get\nfrom PIL import Image\nimport pickle\nfrom contextlib import contextmanager\nimport multiprocessing\n\nfrom sklearn.model_selection import StratifiedKFold, GroupKFold, KFold\nfrom sklearn.neighbors import KNeighborsRegressor, NearestNeighbors\nfrom sklearn.metrics import mean_squared_error, f1_score\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.utils.class_weight import compute_sample_weight\nfrom sklearn.metrics.pairwise import cosine_similarity\nfrom sklearn.decomposition import TruncatedSVD\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom gensim.models import word2vec\n\nimport lightgbm as lgb\nimport typing as tp\n\nfrom logging import getLogger, INFO, StreamHandler, FileHandler, Formatter\n\ntqdm.pandas()\npd.set_option('display.max_rows', 500)\npd.set_option('display.max_columns', 500)\n\nif torch.cuda.is_available():\n    device = torch.device('cuda')\nelse:\n    device = torch.device('cpu')\n    \nprint(f'Using device: {device}')","metadata":{"id":"ffcf3391","outputId":"743e7684-c22d-4ff9-cc18-05441c98d53c","execution":{"iopub.status.busy":"2022-07-09T21:02:44.605524Z","iopub.execute_input":"2022-07-09T21:02:44.606165Z","iopub.status.idle":"2022-07-09T21:02:50.752134Z","shell.execute_reply.started":"2022-07-09T21:02:44.606122Z","shell.execute_reply":"2022-07-09T21:02:50.751079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Config","metadata":{"id":"04cf2fd6"}},{"cell_type":"code","source":"# ==============================================\n#  Config\n# ==============================================\n\nclass CFG:\n    colab = \"google.colab\" in sys.modules\n    exp = \"064\"\n    train = False\n    debug = False\n    api_path = '/content/drive/My Drive/kaggle.json'\n    seed = 42\n    used_fold = [0,1,2,3,4]\n    fold = 5\n    target = \"label\"\n    n_neighbors = 300\n    threshold = 0.56\n    data_split = 10\n    name_low_bound = 20\n    name_th = 0.65\n    phone_th = 0.7\n    \n    # ====================================================\n    # Stage1 (fine-tuning)\n    # ====================================================\n    model = \"sentence-transformers/paraphrase-multilingual-mpnet-base-v2\"\n    batch_size = 32\n    max_length = 32\n    epochs = 15\n    num_workers = 8\n    lr = 1e-5\n    scheduler = 'CosineAnnealingLR'\n\nif not CFG.colab:\n    CFG.model = \"../input/sbert-models/paraphrase-multilingual-mpnet-base-v2\"","metadata":{"id":"146715c7","execution":{"iopub.status.busy":"2022-07-09T21:02:50.753540Z","iopub.execute_input":"2022-07-09T21:02:50.753922Z","iopub.status.idle":"2022-07-09T21:02:50.769102Z","shell.execute_reply.started":"2022-07-09T21:02:50.753885Z","shell.execute_reply":"2022-07-09T21:02:50.768334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ==============================================\n#  Catboost parameters\n# ==============================================\n\nCATEGORICAL_COL = []\nDROP_COLS = [\"id\", \"match_id\"]\n\nPARAMS = {\n    'loss_function': 'Logloss', # ['Logloss', 'AUC']\n    'learning_rate': 0.5,\n    'max_depth': 7,\n    'random_state': CFG.seed,\n    'thread_count': 2,\n    'task_type': 'GPU',\n    #'scale_pos_weight': 4,\n    'num_boost_round': 150000,\n}","metadata":{"id":"6penqhLdjk9t","execution":{"iopub.status.busy":"2022-07-09T21:02:50.771838Z","iopub.execute_input":"2022-07-09T21:02:50.772618Z","iopub.status.idle":"2022-07-09T21:02:50.784532Z","shell.execute_reply.started":"2022-07-09T21:02:50.772581Z","shell.execute_reply":"2022-07-09T21:02:50.783573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if CFG.colab:\n    print(\"==============================================\")\n    print(\"This environment is Google Colab\")\n    print(\"==============================================\")\n\n    # Google Drive\n    from google.colab import drive, files\n    drive.mount('/content/drive')\n    %cd \"drive/My Drive/foursquare/\"\n\n    # Kaggle API\n    f = open(CFG.api_path, 'r')\n    json_data = json.load(f) \n    os.environ[\"KAGGLE_USERNAME\"] = json_data[\"username\"]\n    os.environ[\"KAGGLE_KEY\"] = json_data[\"key\"]\n\n    # Directory Setting\n    if not os.path.exists(f\"output/exp{CFG.exp}/\"):\n        os.makedirs(f\"output/exp{CFG.exp}/\")\n    \n    DATA_DIR = \"input/\"\n    OUTPUT_DIR = f\"output/exp{CFG.exp}/\"\n    MODEL_DIR = OUTPUT_DIR\n    \n    # Data Loading\n    if not os.path.isfile(os.path.join(DATA_DIR, \"foursquare-location-matching.zip\")):\n        !kaggle competitions download -c foursquare-location-matching -p $DATA_DIR\n\n    # Libraries\n    !pip install -q catboost\n    !pip install -q Levenshtein\n    #!pip install -q textdistance==4.2.2\n    !pip install -q pylcs==0.0.6\n    !pip install -q fasttext\n    !pip install -q reverse_geocode\n    !pip install -q transformers\n    !pip install -q sentence_transformers==2.2.0\n\nelse:\n    print(\"==============================================\")\n    print(\" This environment is Kaggle Notebook\")\n    print(\"==============================================\")\n\n    # Directory Setting\n    DATA_DIR = \"../input/foursquare-location-matching/\"\n    OUTPUT_DIR = \"./\"\n    MODEL_DIR = f\"../input/foursquare-dataset-exp{CFG.exp}/\"\n    MODEL_DIR2 = f\"../input/foursquare-dataset-exp094/\"\n\n    # Libraries\n    !pip install /kaggle/input/reversegeocode/reverse_geocode-1.4.1-py3-none-any.whl\n    #!pip install ../input/textdistance-install/textdistance-4.2.2-py3-none-any.whl\n    !pip install --force-reinstall ../input/pylcs-install/pybind11-2.9.2-py2.py3-none-any.whl\n\n    !rm -r mypip\n    !mkdir mypip\n    !tar -czvf mypip/pylcs-0.0.6.tar.gz -C ../input/pylcs-install/pylcs-0.0.6/pylcs-0.0.6 .\n    !ls -l mypip\n\n    !pip install --no-index mypip/pylcs-0.0.6.tar.gz\n    \n    sys.path.append(\"../input/sentencetransformersinstall/sentence-transformers-2.2.0\")\n\n# ====================================================\n# import libraries2\n# ====================================================\n\nfrom catboost import CatBoost, Pool\nimport Levenshtein\nimport pylcs\n#import textdistance\nimport reverse_geocode\nfrom fasttext import load_model\nfrom sentence_transformers import SentenceTransformer, LoggingHandler, losses, InputExample, evaluation\nfrom transformers import DistilBertModel, DistilBertTokenizer, AutoTokenizer, AutoModel, AutoConfig\nfrom transformers import AdamW\nfrom transformers import get_linear_schedule_with_warmup,get_cosine_schedule_with_warmup\nfrom transformers import get_cosine_with_hard_restarts_schedule_with_warmup","metadata":{"id":"36db515c","outputId":"76d14acf-0aec-4d93-990e-91f8b3ba3c35","execution":{"iopub.status.busy":"2022-07-09T21:02:50.786103Z","iopub.execute_input":"2022-07-09T21:02:50.786864Z","iopub.status.idle":"2022-07-09T21:04:23.840695Z","shell.execute_reply.started":"2022-07-09T21:02:50.786827Z","shell.execute_reply":"2022-07-09T21:04:23.839805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Helper Functions","metadata":{"id":"43fc4d81"}},{"cell_type":"code","source":"def reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2    \n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)    \n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose: print('Memory usage decreased to {:5.2f} Mb ({:.1f}% reduction)'.format(end_mem, 100 * (start_mem - end_mem) / start_mem))\n    return df","metadata":{"id":"a0e877b0","execution":{"iopub.status.busy":"2022-07-09T21:04:23.842313Z","iopub.execute_input":"2022-07-09T21:04:23.842690Z","iopub.status.idle":"2022-07-09T21:04:23.860454Z","shell.execute_reply.started":"2022-07-09T21:04:23.842639Z","shell.execute_reply":"2022-07-09T21:04:23.859523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@contextmanager\ndef timer(name: str):\n    t0 = time.time()\n    print(f\"[{name}] start\")\n    yield\n    msg = f\"[{name}] done in {time.time() - t0:.0f} s\"\n    print(msg)","metadata":{"id":"311a9cd0","execution":{"iopub.status.busy":"2022-07-09T21:04:23.862091Z","iopub.execute_input":"2022-07-09T21:04:23.862435Z","iopub.status.idle":"2022-07-09T21:04:23.873738Z","shell.execute_reply.started":"2022-07-09T21:04:23.862400Z","shell.execute_reply":"2022-07-09T21:04:23.873041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed=42):\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    os.environ[\"TOKENIZERS_PARALLELISM\"] = \"false\"\n\nseed_everything(CFG.seed)","metadata":{"id":"e3d77c9b","execution":{"iopub.status.busy":"2022-07-09T21:04:23.875247Z","iopub.execute_input":"2022-07-09T21:04:23.875628Z","iopub.status.idle":"2022-07-09T21:04:23.886515Z","shell.execute_reply.started":"2022-07-09T21:04:23.875590Z","shell.execute_reply":"2022-07-09T21:04:23.885800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def init_logger(log_file=OUTPUT_DIR+'train.log'):\n    logger = getLogger(__name__)\n    logger.setLevel(INFO)\n    logger.hasHandlers()\n    handler1 = StreamHandler()\n    handler1.setFormatter(Formatter(\"%(message)s\"))\n    handler2 = FileHandler(filename=log_file)\n    handler2.setFormatter(Formatter(\"%(message)s\"))\n    logger.addHandler(handler1)\n    logger.addHandler(handler2)\n    return logger\n\nLOGGER = init_logger()","metadata":{"id":"c4a09e1a","execution":{"iopub.status.busy":"2022-07-09T21:04:23.887915Z","iopub.execute_input":"2022-07-09T21:04:23.888391Z","iopub.status.idle":"2022-07-09T21:04:23.897384Z","shell.execute_reply.started":"2022-07-09T21:04:23.888353Z","shell.execute_reply":"2022-07-09T21:04:23.896469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_id2poi(input_df: pd.DataFrame) -> dict:\n    return dict(zip(input_df['id'], input_df['point_of_interest']))\n\ndef get_poi2ids(input_df: pd.DataFrame) -> dict:\n    return input_df.groupby('point_of_interest')['id'].apply(set).to_dict()\n\ndef get_score(input_df: pd.DataFrame):\n    scores = []\n    for id_str, matches in zip(input_df['id'].to_numpy(), input_df['matches'].to_numpy()):\n        targets = poi2ids[id2poi[id_str]]\n        preds = set(matches.split())\n        score = len((targets & preds)) / len((targets | preds))\n        scores.append(score)\n    scores = np.array(scores)\n    return scores.mean()\n\ndef analysis(df):\n    print('Num of data: %s' % len(df))\n    print('Num of unique id: %s' % df['id'].nunique())\n    print('Num of unique poi: %s' % df['point_of_interest'].nunique())\n    \n    poi_grouped = df.groupby('point_of_interest')['id'].count().reset_index()\n    print('Mean num of unique poi: %s' % poi_grouped['id'].mean())","metadata":{"id":"a5b6be05","execution":{"iopub.status.busy":"2022-07-09T21:04:23.901501Z","iopub.execute_input":"2022-07-09T21:04:23.901766Z","iopub.status.idle":"2022-07-09T21:04:23.913370Z","shell.execute_reply.started":"2022-07-09T21:04:23.901742Z","shell.execute_reply":"2022-07-09T21:04:23.912621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cos_sim(v1, v2):\n    return np.dot(v1, v2) / (np.linalg.norm(v1) * np.linalg.norm(v2))","metadata":{"id":"rQvgr7P2L5Hj","execution":{"iopub.status.busy":"2022-07-09T21:04:23.914543Z","iopub.execute_input":"2022-07-09T21:04:23.914996Z","iopub.status.idle":"2022-07-09T21:04:23.924253Z","shell.execute_reply.started":"2022-07-09T21:04:23.914961Z","shell.execute_reply":"2022-07-09T21:04:23.923385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ====================================================\n#  Inference\n# ====================================================\n\ndef inference(df):\n    pred = np.zeros(len(df))\n    for fold in tqdm(range(CFG.fold)):\n        if fold in CFG.used_fold:\n            \n            # two seeds averaging\n            \n            # seed 42\n            with open(f\"../input/foursquare-dataset-exp064/model_fold{fold}.pkl\", 'rb') as f:\n                model = pickle.load(f)\n            pred += model.predict(df.drop(columns=DROP_COLS), prediction_type='Probability').T[1]\n            del model; gc_clear()\n            \n            # seed 31\n            with open(f\"../input/foursquare-dataset-exp064-seed31/model_fold{fold}.pkl\", 'rb') as f:\n                model = pickle.load(f)\n            pred += model.predict(df.drop(columns=DROP_COLS), prediction_type='Probability').T[1]\n            del model; gc_clear()\n            \n    pred = pred / (len(CFG.used_fold)*2)\n    df = df[[\"id\", \"match_id\"]]\n    df[\"prediction\"] = pred\n\n    df = df[df[\"prediction\"]>CFG.threshold].groupby(\"id\")[\"match_id\"].apply(list).reset_index()\n    df.columns = [\"id\", \"matches\"]\n    df[\"matches\"] = df[\"matches\"].apply(lambda x: \" \".join(x))\n\n    return df","metadata":{"id":"e0fce192","execution":{"iopub.status.busy":"2022-07-09T21:04:23.926100Z","iopub.execute_input":"2022-07-09T21:04:23.926764Z","iopub.status.idle":"2022-07-09T21:04:23.936358Z","shell.execute_reply.started":"2022-07-09T21:04:23.926728Z","shell.execute_reply":"2022-07-09T21:04:23.935535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ====================================================\n#  Post-processing\n# ====================================================\n\ndef post_process(df):\n    id2match = dict(zip(df['id'].values, df['matches'].str.split()))\n\n    for base, match in df[['id', 'matches']].values:\n        match = match.split()\n        if len(match) == 1:        \n            continue\n\n        for m in match:\n            if base not in id2match[m]:\n                id2match[m].append(base)\n    df['matches'] = df['id'].map(id2match).map(' '.join)\n    return df ","metadata":{"id":"d3a16b81","execution":{"iopub.status.busy":"2022-07-09T21:04:23.937764Z","iopub.execute_input":"2022-07-09T21:04:23.938223Z","iopub.status.idle":"2022-07-09T21:04:23.949457Z","shell.execute_reply.started":"2022-07-09T21:04:23.938188Z","shell.execute_reply":"2022-07-09T21:04:23.948591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gc_clear():\n    for i in range(5):\n        gc.collect()","metadata":{"id":"WwYwVmvtjPrf","execution":{"iopub.status.busy":"2022-07-09T21:04:23.950541Z","iopub.execute_input":"2022-07-09T21:04:23.951052Z","iopub.status.idle":"2022-07-09T21:04:23.959581Z","shell.execute_reply.started":"2022-07-09T21:04:23.951015Z","shell.execute_reply":"2022-07-09T21:04:23.958692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def flatten(x):\n    return [e for i in x for e in i]","metadata":{"id":"bfrYKmJlLWS_","execution":{"iopub.status.busy":"2022-07-09T21:04:23.961186Z","iopub.execute_input":"2022-07-09T21:04:23.961589Z","iopub.status.idle":"2022-07-09T21:04:23.969771Z","shell.execute_reply.started":"2022-07-09T21:04:23.961553Z","shell.execute_reply":"2022-07-09T21:04:23.968934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vec_columns = ['name', 'categories', 'address', 'text']\nfeat_columns = ['name', 'address', 'city', 'state', 'zip', 'country', 'url', 'phone', 'categories', 'name_lang', 'city2', 'country2', 'text'] ","metadata":{"id":"37deb317","execution":{"iopub.status.busy":"2022-07-09T21:04:23.971085Z","iopub.execute_input":"2022-07-09T21:04:23.971538Z","iopub.status.idle":"2022-07-09T21:04:23.980173Z","shell.execute_reply.started":"2022-07-09T21:04:23.971502Z","shell.execute_reply":"2022-07-09T21:04:23.979298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Loading","metadata":{"id":"1d709f14"}},{"cell_type":"code","source":"with timer(\"Data Loading\"):\n    if CFG.train:\n        original_df = pd.read_csv(DATA_DIR + \"train.csv\")\n        if CFG.debug:\n            original_df = original_df[:100]\n    else:\n        original_df = pd.read_csv(DATA_DIR + \"test.csv\")\n        original_df[\"point_of_interest\"] = \"match\"\ndisplay(original_df)","metadata":{"id":"80066067","outputId":"0fb72f6c-c18d-42c4-8dae-7d2f0003da47","execution":{"iopub.status.busy":"2022-07-09T21:04:23.982551Z","iopub.execute_input":"2022-07-09T21:04:23.983204Z","iopub.status.idle":"2022-07-09T21:04:24.028125Z","shell.execute_reply.started":"2022-07-09T21:04:23.983167Z","shell.execute_reply":"2022-07-09T21:04:24.027246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id2poi = get_id2poi(original_df)\npoi2ids = get_poi2ids(original_df)","metadata":{"id":"11281a08","execution":{"iopub.status.busy":"2022-07-09T21:04:24.029405Z","iopub.execute_input":"2022-07-09T21:04:24.029968Z","iopub.status.idle":"2022-07-09T21:04:24.039713Z","shell.execute_reply.started":"2022-07-09T21:04:24.029930Z","shell.execute_reply":"2022-07-09T21:04:24.038781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calc_maximum_score(original_df, df):\n\n    eval_df = pd.DataFrame()\n    eval_df['id'] = original_df['id'].unique().tolist()\n    eval_df['match_id'] = eval_df['id']\n\n    eval_df_ = df[df['label'] == 1][['id', 'match_id']]\n    eval_df = pd.concat([eval_df, eval_df_])\n\n    eval_df = eval_df.groupby('id')['match_id'].apply(list).reset_index()\n    eval_df['matches'] = eval_df['match_id'].apply(lambda x: ' '.join(set(x)))\n\n    score_before_pp = get_score(eval_df)\n    print(f\"maximum socre (before pp): {score_before_pp}\")\n    score_after_pp = get_score(post_process(eval_df))\n    print(f\"maximum socre (after pp): {score_after_pp}\")","metadata":{"id":"efXSBz5LfB62","execution":{"iopub.status.busy":"2022-07-09T21:04:24.042693Z","iopub.execute_input":"2022-07-09T21:04:24.043173Z","iopub.status.idle":"2022-07-09T21:04:24.051933Z","shell.execute_reply.started":"2022-07-09T21:04:24.043137Z","shell.execute_reply":"2022-07-09T21:04:24.051030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def categorical_similarity(A, B):\n    if A==\"nan\" or B==\"nan\":\n        return np.nan\n\n    A = set(str(A).split(\", \"))\n    B = set(str(B).split(\", \"))\n\n    nominator = A.intersection(B)\n\n    similarity_1 = len(nominator) / len(A)\n    similarity_2 = len(nominator) / len(B)\n\n    return max(similarity_1, similarity_2)\n\ndef gesh(A, B):\n    if A==\"nan\" or B==\"nan\":\n        return np.nan\n    else:\n        return difflib.SequenceMatcher(None, A, B).ratio()\n\ndef leven(A, B):\n    if A==\"nan\" or B==\"nan\":\n        return np.nan\n    else:\n        return Levenshtein.distance(A, B)\n\ndef jaro(A, B):\n    if A==\"nan\" or B==\"nan\":\n        return np.nan\n    else:\n        return Levenshtein.jaro_winkler(A, B)\n\ndef lcs_sequence(A, B):\n    if A==\"nan\" or B==\"nan\":\n        return np.nan\n    else:\n        return pylcs.lcs(A, B)\n\ndef lcs_string(A, B):\n    if A==\"nan\" or B==\"nan\":\n        return np.nan\n    else:\n        return pylcs.lcs2(A, B)\n\ndef equal(A, B):\n    if A==\"nan\" or B==\"nan\":\n        return np.nan\n    else:\n        return int(A==B)","metadata":{"id":"5-jXhfFexY8j","execution":{"iopub.status.busy":"2022-07-09T21:04:24.053060Z","iopub.execute_input":"2022-07-09T21:04:24.053626Z","iopub.status.idle":"2022-07-09T21:04:24.066131Z","shell.execute_reply.started":"2022-07-09T21:04:24.053580Z","shell.execute_reply":"2022-07-09T21:04:24.065262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocess","metadata":{"id":"6af14709"}},{"cell_type":"markdown","source":"## Devide Train Data into about 600K×2","metadata":{"id":"b181d18d"}},{"cell_type":"code","source":"if CFG.train:\n    kf = GroupKFold(n_splits=2)\n    for i_fold, (trn_idx, val_idx) in enumerate(kf.split(original_df, original_df[\"point_of_interest\"], original_df[\"point_of_interest\"])):\n        original_df.loc[val_idx, \"set\"] = i_fold\n    original_df[\"set\"] = original_df[\"set\"].astype(\"int8\")\n    print(original_df[\"set\"].value_counts())","metadata":{"id":"ec993462","outputId":"624eddd1-7538-48bd-fbe4-c2d5381eca22","execution":{"iopub.status.busy":"2022-07-09T21:04:24.067399Z","iopub.execute_input":"2022-07-09T21:04:24.067840Z","iopub.status.idle":"2022-07-09T21:04:24.080393Z","shell.execute_reply.started":"2022-07-09T21:04:24.067796Z","shell.execute_reply":"2022-07-09T21:04:24.079619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Candidates Generation","metadata":{"id":"b05093dc"}},{"cell_type":"markdown","source":"In generating candidates, I used not only haversine distance, but also `name` similarity and `phone` similarity to select candidates with larger theoretical maximum IoU score.","metadata":{}},{"cell_type":"code","source":"def recall_knn(original_df, Neighbors=CFG.n_neighbors):\n\n    # ==============================================\n    #  Candidates Generation\n    # ==============================================\n    # ideas and codes from my teammate @mrt0933!\n\n    near_ids = {}\n    near_dists = {}\n\n    for country, country_df in tqdm(original_df.groupby(\"country\")):\n        country_df = country_df.reset_index(drop=True)\n    \n        knn = KNeighborsRegressor(n_neighbors=min(len(country_df), Neighbors), \n                                    metric='haversine', n_jobs=-1)\n        knn.fit(country_df[['latitude','longitude']], country_df.index)\n        dists, nears = knn.kneighbors(country_df[['latitude','longitude']], return_distance=True)\n\n        ids = country_df['id'].values\n        nids = country_df['id'].values[nears]\n        for j in range(len(country_df)):\n            near_ids[ids[j]] = nids[j]\n            near_dists[ids[j]] = dists[j]\n\n    def get_lcs_rate(str1, str2):\n        if str1!=\"nan\":\n            return lcs_sequence(str1, str2) / len(str1)\n        else :\n            return np.nan\n\n    ids = original_df['id'].values\n    id2index = {ids[i]:i for i in range(len(ids))}\n\n    res = {idi:near_ids[idi][:CFG.name_low_bound].tolist() for idi in near_ids}\n    names = original_df['name'].values.astype(str)\n\n    original_df['phone'] = original_df['phone'].replace(\"\", \"nan\")\n \n    phone = original_df['phone'].values\n\n    for id0 in tqdm(near_ids):\n        idx0 = id2index[id0]\n        n0 = names[idx0]\n        ph0 = phone[idx0]\n\n        for k, id1 in enumerate(near_ids[id0]):\n            idx1 = id2index[id1]\n            ph1 = phone[idx1]\n            n1 = names[idx1]\n\n            if k < CFG.name_low_bound:\n                continue\n\n            jrate = jaro(n0, n1)\n            lrate = get_lcs_rate(ph0, ph1)\n            if jrate >= CFG.name_th or lrate >= CFG.phone_th:\n                res[id0].append(id1)\n\n    id_list = []\n    match_id_list = []\n\n    for id0 in tqdm(res):\n        num = len(res[id0])\n        id_list.append([[id0]*num])\n        match_id_list.append(res[id0])\n\n    df = pd.DataFrame()\n    df[\"id\"] = flatten(flatten(id_list))\n    df[\"match_id\"] = flatten(match_id_list)\n\n    del id_list, match_id_list\n    gc_clear()\n\n    # ==============================================\n    #  Delete unnecessary rows\n    # ==============================================\n\n    # Delete same id\n    df = df[df[\"id\"]!=df[\"match_id\"]].reset_index(drop=True)\n\n    # Delete duplicate pairs\n    if not CFG.train:\n        df = df[~pd.DataFrame(np.sort(df[['id','match_id']].values,1)).duplicated()].reset_index(drop=True)\n\n    print(f\"len: {len(df)}\")\n\n    # ==============================================\n    #  Make 0/1 label\n    # ==============================================\n\n    if CFG.train:\n\n        ids = df['id'].tolist()\n        match_ids = df['match_id'].tolist()\n\n        poi = original_df.set_index('id').loc[ids]['point_of_interest'].values\n        match_poi = original_df.set_index('id').loc[match_ids]['point_of_interest'].values\n\n        df['label'] = np.array(poi == match_poi, dtype = np.int8)\n\n        print(df['label'].value_counts())\n        \n        del poi, match_poi, ids, match_ids\n        gc_clear()\n\n    # ==============================================\n    #  Max score\n    # ==============================================\n    if CFG.train:\n        calc_maximum_score(original_df, df)\n    \n    # ==============================================\n    #  Make test fold\n    # ==============================================\n            \n    if len(df)<10000:\n        CFG.data_split = 1\n        df[\"data_split\"] = 0\n        df[\"data_split\"] = df[\"data_split\"].astype(\"int8\")\n    else:\n        kf = GroupKFold(n_splits=CFG.data_split)\n        for i_fold, (trn_idx, val_idx) in enumerate(kf.split(df, df[\"id\"], df[\"id\"])):\n            df.loc[val_idx, \"data_split\"] = i_fold\n        df[\"data_split\"] = df[\"data_split\"].astype(\"int8\")\n    \n    return df","metadata":{"id":"0b2a34f7","execution":{"iopub.status.busy":"2022-07-09T21:04:24.081810Z","iopub.execute_input":"2022-07-09T21:04:24.082195Z","iopub.status.idle":"2022-07-09T21:04:24.107450Z","shell.execute_reply.started":"2022-07-09T21:04:24.082158Z","shell.execute_reply":"2022-07-09T21:04:24.106688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{"id":"280d00cf"}},{"cell_type":"markdown","source":"### Preprocess original_df","metadata":{}},{"cell_type":"code","source":"def original_df_preprocess(original_df):\n    # ==============================================\n    # Language identification\n    # ==============================================\n\n    if CFG.colab:\n        model = load_model(\"models/lid.176.bin\")\n    else:\n        model = load_model(\"../input/language-predictor/lid.176.bin\")\n\n    original_df[\"name_lang\"] = original_df[\"name\"].fillna(\" \").progress_apply(lambda x: model.predict(x)[0][0][9:])\n    original_df.loc[original_df[\"name\"]==\"nan\", \"name_lang\"] = \"nan\"\n\n    del model\n    gc_clear()\n\n    # ==============================================\n    # Reverse geocode\n    # ==============================================\n\n    def get_geo_info(coords):\n        data = reverse_geocode.search(coords)\n        return [v['country_code'] for v in data], [v['city'] for v in data]\n\n    original_df['country2'] = get_geo_info(original_df[['latitude', 'longitude']])[0]\n    original_df['city2'] = get_geo_info(original_df[['latitude', 'longitude']])[1]\n\n    # ==============================================\n    # Degree to radian\n    # ==============================================\n    original_df[\"latitude\"] = original_df[\"latitude\"] * np.pi / 180\n    original_df[\"longitude\"] = original_df[\"longitude\"] * np.pi / 180\n\n    # ==============================================\n    # Make features\n    # ==============================================\n    original_df[\"text\"] = original_df[\"name\"].fillna(\"\") + \" \" + \\\n                            original_df[\"address\"].fillna(\"\") + \" \" + \\\n                            original_df[\"city\"].fillna(\"\") + \" \" + \\\n                            original_df[\"state\"].fillna(\"\") + \" \" + \\\n                            original_df[\"categories\"].fillna(\"\")\n\n    # ==============================================\n    # Clean Sentences\n    # ==============================================\n\n    #def clean_text(text):\n    #    try:\n    #        text = text.encode('utf-8').decode(\"unicode_escape\")\n    #        text = text.encode('ascii', 'ignore').decode(\"unicode_escape\")\n    #    except:\n    #        pass\n    #    return text\n    # \n    #original_df[\"name\"] = original_df[\"name\"].apply(lambda x: clean_text(x))\n\n    #def remove_char(x):\n    #    return re.sub(r'[^0-9]', '', x)\n\n    #func = np.frompyfunc(remove_char, 1, 1)\n    #original_df['phone'] = func(original_df['phone'].values)\n    #original_df['phone'] = original_df['phone'].replace(\"\", np.nan)\n\n    for c in original_df.columns:\n        if not c in [\"id\", \"latitude\", \"longitude\"]:\n            \n            original_df[c] = original_df[c].astype(str) #.str.lower()\n                \n    return original_df","metadata":{"id":"c6cd5bac","execution":{"iopub.status.busy":"2022-07-09T21:04:24.110365Z","iopub.execute_input":"2022-07-09T21:04:24.110612Z","iopub.status.idle":"2022-07-09T21:04:24.124264Z","shell.execute_reply.started":"2022-07-09T21:04:24.110582Z","shell.execute_reply":"2022-07-09T21:04:24.123273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Distance features","metadata":{}},{"cell_type":"code","source":"# ==============================================\n# Distance features\n# ==============================================\n\ndef vectorized_haversine(lats1, lats2, longs1, longs2):\n    dlat = np.radians(lats2 - lats1)\n    dlon = np.radians(longs2 - longs1)\n    a = np.sin(dlat/2) * np.sin(dlat/2) + np.cos(np.radians(lats1)) \\\n        * np.cos(np.radians(lats2)) * np.sin(dlon/2) * np.sin(dlon/2)\n    c = 2 * np.arctan2(np.sqrt(a), np.sqrt(1-a))\n    return c\n\ndef add_distance_features(original_df, df):\n    original_df = original_df.set_index('id')\n\n    lat1 = original_df.loc[df[\"id\"].tolist()][\"latitude\"].values\n    lat2 = original_df.loc[df[\"match_id\"].tolist()][\"latitude\"].values\n    lon1 = original_df.loc[df[\"id\"].tolist()][\"longitude\"].values\n    lon2 = original_df.loc[df[\"match_id\"].tolist()][\"longitude\"].values\n    \n    df['latdiff'] = abs(lat1 - lat2).astype(\"float16\")\n    df['londiff'] = abs(lon1 - lon2).astype(\"float16\")\n\n    df['haversine'] = vectorized_haversine(lat1, lat2, lon1, lon2).astype(\"float32\")\n\n    return df","metadata":{"id":"pC2pFcpQKeIg","execution":{"iopub.status.busy":"2022-07-09T21:04:24.126472Z","iopub.execute_input":"2022-07-09T21:04:24.127144Z","iopub.status.idle":"2022-07-09T21:04:24.139262Z","shell.execute_reply.started":"2022-07-09T21:04:24.127066Z","shell.execute_reply":"2022-07-09T21:04:24.138472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### tfidf features","metadata":{}},{"cell_type":"code","source":"def add_tfidf_features(original_df, df):\n    \n    id2index_d = dict(zip(original_df['id'].values, original_df.index))\n    indexs = [id2index_d[i] for i in df['id']]\n    match_indexs = [id2index_d[i] for i in df['match_id']]\n\n    tfidf1 = TfidfVectorizer(analyzer=\"word\", #[\"word\", \"char\", \"char_wb\"]\n                            #strip_accents=\"unicode\",\n                            ngram_range=(1, 1))\n    tfidf2 = TfidfVectorizer(analyzer=\"char_wb\", #[\"word\", \"char\", \"char_wb\"]\n                            #strip_accents=\"unicode\",\n                            ngram_range=(3, 3))\n    \n    tfidf_dict = {\"tfidf1\": tfidf1,\n                  \"tfidf2\": tfidf2,\n                  }\n    \n    for tfidf_name in [\"tfidf1\",\"tfidf2\"]:\n        for col in tqdm(vec_columns):\n            \n            tv_fit = tfidf_dict[tfidf_name].fit_transform(original_df[col])\n            \n            output = np.array([])\n            for i in range(5):\n                chunk = len(indexs)//5 + 1\n                s = i*chunk\n                e = (i+1)*chunk\n                output = np.append(output, tv_fit[indexs[s:e]].multiply(tv_fit[match_indexs[s:e]]).sum(axis = 1).A.ravel())\n\n            df[f'{col}_{tfidf_name}_sim'] = output\n            df[f'{col}_{tfidf_name}_sim'] = output.astype(\"float16\")\n\n            del tv_fit, output\n            gc_clear()\n\n    del id2index_d, indexs, match_indexs\n    gc_clear()\n        \n    return df","metadata":{"id":"36iKn8Nh6Tdt","execution":{"iopub.status.busy":"2022-07-09T21:04:24.140679Z","iopub.execute_input":"2022-07-09T21:04:24.141298Z","iopub.status.idle":"2022-07-09T21:04:24.152638Z","shell.execute_reply.started":"2022-07-09T21:04:24.141262Z","shell.execute_reply":"2022-07-09T21:04:24.151882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Pretrained BERT features","metadata":{}},{"cell_type":"code","source":"def add_bert_features(original_df, df):\n\n    if CFG.colab:\n        model_dict = {\"mpnet\": \"paraphrase-multilingual-mpnet-base-v2\",\n                      \"para_xlm\": \"paraphrase-xlm-r-multilingual-v1\",\n                      \"xlm\": \"xlm-roberta-base\",\n                      \"MiniLM\": \"paraphrase-multilingual-MiniLM-L12-v2\",\n                      }\n    else:\n        model_dict = {\"mpnet\": \"../input/sbert-models/paraphrase-multilingual-mpnet-base-v2\",\n                      \"para_xlm\": \"../input/sbert-models/paraphrase-xlm-r-multilingual-v1\",\n                      \"xlm\": \"../input/sbert-models/xlm-roberta-base\",\n                      \"MiniLM\": \"../input/sbert-models/paraphrase-multilingual-MiniLM-L12-v2\",\n                    }\n\n    for model_name in model_dict:\n\n        original_df[\"name_categories\"] = original_df[\"name\"] + \"[SEP]\" + original_df[\"categories\"]\n\n        original_df[\"index\"] = original_df.index\n        original_df[\"index\"] = original_df[\"index\"].astype(\"int32\")\n        id_indexes = original_df.set_index(\"id\").loc[df[\"id\"]][\"index\"].values\n        match_id_indexes = original_df.set_index(\"id\").loc[df[\"match_id\"]][\"index\"].values\n\n        if model_name in [\"mpnet\", \"para_xlm\", \"xlm\", \"MiniLM\"]:\n\n            model = SentenceTransformer(model_dict[model_name], device=device)\n\n            for col in [\"name\", \"categories\", \"name_categories\"]:\n                vec = model.encode(original_df[col], device=device)\n\n                output = []\n                for idx1, idx2 in tqdm(zip(id_indexes, match_id_indexes), total=len(id_indexes)):\n                    output.append(cos_sim(vec[idx1], vec[idx2]))\n\n                df[f\"{col}_{model_name}_sim\"] = output\n                df[f\"{col}_{model_name}_sim\"] = df[f\"{col}_{model_name}_sim\"].astype(\"float16\")\n\n                del vec, output\n                gc_clear()\n\n            del model\n            torch.cuda.empty_cache()\n            gc_clear()\n        \n        del original_df[\"name_categories\"], original_df[\"index\"], id_indexes, match_id_indexes\n        gc_clear()\n\n    return df","metadata":{"id":"BotVfcpG0Y2g","execution":{"iopub.status.busy":"2022-07-09T21:04:24.154036Z","iopub.execute_input":"2022-07-09T21:04:24.154599Z","iopub.status.idle":"2022-07-09T21:04:24.167565Z","shell.execute_reply.started":"2022-07-09T21:04:24.154564Z","shell.execute_reply":"2022-07-09T21:04:24.166812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fine-tuned BERT feature (Stage1)","metadata":{}},{"cell_type":"markdown","source":"Reference: [My other notebook](https://www.kaggle.com/code/shkanda/unsupervised-baseline-curricularface?scriptVersionId=100429606)","metadata":{}},{"cell_type":"code","source":"# ====================================================\n#  CurricularFace\n# ====================================================   \n\ndef l2_norm(input, axis = 1):\n    norm = torch.norm(input, 2, axis, True)\n    output = torch.div(input, norm)\n\n    return output\n\nclass CurricularFace(nn.Module):\n    def __init__(self, in_features, out_features, s = 5, m = 0.050):\n        super(CurricularFace, self).__init__()\n\n        self.in_features = in_features\n        self.out_features = out_features\n        self.m = m\n        self.s = s\n        self.cos_m = math.cos(m)\n        self.sin_m = math.sin(m)\n        self.threshold = math.cos(math.pi - m)\n        self.mm = math.sin(math.pi - m) * m\n        self.kernel = nn.Parameter(torch.Tensor(in_features, out_features))\n        self.register_buffer('t', torch.zeros(1))\n        nn.init.normal_(self.kernel, std=0.01)\n\n    def forward(self, embbedings, label):\n        embbedings = l2_norm(embbedings, axis = 1)\n        kernel_norm = l2_norm(self.kernel, axis = 0)\n        cos_theta = torch.mm(embbedings, kernel_norm)\n        cos_theta = cos_theta.clamp(-1, 1)\n        with torch.no_grad():\n            origin_cos = cos_theta.clone()\n        target_logit = cos_theta[torch.arange(0, embbedings.size(0)), label].view(-1, 1)\n\n        sin_theta = torch.sqrt(1.0 - torch.pow(target_logit, 2))\n        cos_theta_m = target_logit * self.cos_m - sin_theta * self.sin_m\n        mask = cos_theta > cos_theta_m\n        final_target_logit = torch.where(target_logit > self.threshold, cos_theta_m, target_logit - self.mm)\n\n        hard_example = cos_theta[mask]\n        with torch.no_grad():\n            self.t = target_logit.mean() * 0.01 + (1 - 0.01) * self.t\n        cos_theta[mask] = hard_example * (self.t + hard_example)\n        cos_theta.scatter_(1, label.view(-1, 1).long(), final_target_logit)\n        output = cos_theta * self.s\n        return output","metadata":{"execution":{"iopub.status.busy":"2022-07-09T21:04:24.168959Z","iopub.execute_input":"2022-07-09T21:04:24.169532Z","iopub.status.idle":"2022-07-09T21:04:24.184590Z","shell.execute_reply.started":"2022-07-09T21:04:24.169496Z","shell.execute_reply":"2022-07-09T21:04:24.183935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ====================================================\n#  Fine-tuned Bert Model\n# ==================================================== \n\nclass CustomModel(nn.Module):\n    def __init__(self, model_name, n_classes=369985, embedding_size=128):                 \n        super(CustomModel, self).__init__()\n\n        self.config = AutoConfig.from_pretrained(model_name)\n        self.model = AutoModel.from_pretrained(model_name, \n                                               config=self.config)\n        #self.fc = ArcMarginProduct(embedding_size, CFG.n_classes)\n        self.fc = CurricularFace(embedding_size, n_classes)\n        self.head = nn.Sequential(\n            nn.Linear(self.config.hidden_size + 2, embedding_size),\n            nn.BatchNorm1d(embedding_size),\n        )\n\n    def forward(self, ids, mask, lat, lon, labels):\n        embedding = self.extract(ids=ids, mask=mask, lat=lat, lon=lon)\n        output = self.fc(embedding, labels)\n        return output\n    \n    def extract(self, ids, mask, lat, lon):\n        lat, lon = lat.view(-1, 1), lon.view(-1, 1)\n        out = self.model(input_ids=ids, attention_mask=mask)\n        embedding = out[0][:, 0, :] # CLS Token\n        embedding = torch.cat([embedding, lat, lon], axis=1)\n        embedding = self.head(embedding)\n        return embedding\n    \nprint(CustomModel(CFG.model))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T21:04:24.188781Z","iopub.execute_input":"2022-07-09T21:04:24.189494Z","iopub.status.idle":"2022-07-09T21:04:42.756923Z","shell.execute_reply.started":"2022-07-09T21:04:24.189457Z","shell.execute_reply":"2022-07-09T21:04:42.756065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ====================================================\n#  Dataset\n# ====================================================\n\nclass FoursquareDataset(Dataset):\n    def __init__(self, df, include_labels=True):\n        tokenizer = AutoTokenizer.from_pretrained(CFG.model)\n\n        self.df = df\n        self.include_labels = include_labels\n\n        self.text = df['text'].tolist()\n        self.lat = df['latitude'].values\n        self.lon = df['longitude'].values\n        self.labels = df['point_of_interest'].values\n\n        self.encoded = tokenizer.batch_encode_plus(\n            self.text,\n            padding = 'max_length',            \n            max_length = CFG.max_length,\n            truncation = True,\n            return_attention_mask=True\n        )\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n\n        input_ids = torch.tensor(self.encoded['input_ids'][idx], dtype=torch.long)\n        attention_mask = torch.tensor(self.encoded['attention_mask'][idx], dtype=torch.long)\n        lat = torch.tensor(self.lat[idx], dtype=torch.float)\n        lon = torch.tensor(self.lon[idx], dtype=torch.float)\n\n        if self.include_labels:\n            label = torch.tensor(self.labels[idx], dtype=torch.long)\n            return input_ids, attention_mask, lat, lon, label\n\n        return input_ids, attention_mask, lat, lon","metadata":{"execution":{"iopub.status.busy":"2022-07-09T21:04:42.758337Z","iopub.execute_input":"2022-07-09T21:04:42.758902Z","iopub.status.idle":"2022-07-09T21:04:42.769433Z","shell.execute_reply.started":"2022-07-09T21:04:42.758862Z","shell.execute_reply":"2022-07-09T21:04:42.768543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def inference_fn(test, i_set):\n    \n    if int(i_set) == int(0):\n        n_classes = 369987\n    elif int(i_set) == int(1):\n        n_classes = 369985\n\n    test_dataset = FoursquareDataset(test, include_labels=False)\n\n    test_loader = DataLoader(\n        test_dataset,\n        batch_size=CFG.batch_size,\n        shuffle=False,\n        num_workers=CFG.num_workers,\n        pin_memory=True,\n        drop_last=False,\n        worker_init_fn=seed_worker,\n        generator=g\n    )\n\n    model = CustomModel(CFG.model, n_classes=n_classes)\n    path = MODEL_DIR2 + f\"set{1-int(i_set)}_bert_epoch15.pth\"\n    state = torch.load(path, map_location=torch.device('cpu'))\n    model.load_state_dict(state)\n    model.to(device)\n    model.eval()\n\n    preds = []\n    for step, (input_ids, attention_mask, lat, lon) in tqdm(enumerate(test_loader), total=len(test_loader)):\n        input_ids = input_ids.to(device)\n        attention_mask = attention_mask.to(device)\n        lat = lat.to(device)\n        lon = lon.to(device)\n        \n        with torch.no_grad():\n            pred = model.extract(input_ids, attention_mask, lat, lon)\n        preds.append(pred.detach().cpu().numpy())\n    preds = np.concatenate(preds)\n\n    del model\n    torch.cuda.empty_cache()\n    gc_clear()\n\n    return preds","metadata":{"execution":{"iopub.status.busy":"2022-07-09T21:04:42.772135Z","iopub.execute_input":"2022-07-09T21:04:42.772394Z","iopub.status.idle":"2022-07-09T21:04:42.791417Z","shell.execute_reply.started":"2022-07-09T21:04:42.772370Z","shell.execute_reply":"2022-07-09T21:04:42.790381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_finetuned_bert_features(original_df, df):\n\n    original_df[\"index\"] = original_df.index\n    original_df[\"index\"] = original_df[\"index\"].astype(\"int32\")\n    id_indexes = original_df.set_index(\"id\").loc[df[\"id\"]][\"index\"].values\n    match_id_indexes = original_df.set_index(\"id\").loc[df[\"match_id\"]][\"index\"].values\n\n    if CFG.train:\n\n        i_set = original_df[\"set\"][0]\n        embedding = inference_fn(original_df, i_set)\n\n        output = []\n        for idx1, idx2 in tqdm(zip(id_indexes, match_id_indexes), total=len(id_indexes)):\n            output.append(cos_sim(embedding[idx1], embedding[idx2]))\n\n        df[f\"finetuned_mpnet_sim\"] = output\n        df[f\"finetuned_mpnet_sim\"] = df[f\"finetuned_mpnet_sim\"].astype(\"float16\")\n\n        del embedding, output\n        gc_clear()\n\n    else:\n\n        embedding0 = inference_fn(original_df, 0)\n        embedding1 = inference_fn(original_df, 1)\n\n        output0 = []\n        for idx1, idx2 in tqdm(zip(id_indexes, match_id_indexes), total=len(id_indexes)):\n            output0.append(cos_sim(embedding0[idx1], embedding0[idx2]))\n\n        output1 = []\n        for idx1, idx2 in tqdm(zip(id_indexes, match_id_indexes), total=len(id_indexes)):\n            output1.append(cos_sim(embedding1[idx1], embedding1[idx2]))\n\n        output = (np.array(output0) + np.array(output1))/2\n\n        df[f\"finetuned_mpnet_sim\"] = output\n        df[f\"finetuned_mpnet_sim\"] = df[f\"finetuned_mpnet_sim\"].astype(\"float16\")\n\n        del embedding0, embedding1, output0, output1, output\n        gc_clear()\n        \n    del original_df[\"index\"], id_indexes, match_id_indexes\n    gc_clear()\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-09T21:04:42.792994Z","iopub.execute_input":"2022-07-09T21:04:42.793625Z","iopub.status.idle":"2022-07-09T21:04:42.807629Z","shell.execute_reply.started":"2022-07-09T21:04:42.793586Z","shell.execute_reply":"2022-07-09T21:04:42.806783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Other features","metadata":{}},{"cell_type":"markdown","source":"`multiprocessing` is used to speed up feature generation.\n\nSince `multiprocessing` requires large memory,I reduced memory usage by creating features and making predictions on each segmented data (each `data_split`).","metadata":{}},{"cell_type":"code","source":"def _add_other_features(args):\n    (_, df), original_df = args\n\n    for col in tqdm(feat_columns):\n        \n        col_values = original_df.set_index('id').loc[df['id']][col].values.astype(str)\n        matcol_values = original_df.set_index('id').loc[df['match_id']][col].values.astype(str)\n\n        if not col in ['country', 'name_lang', 'city2', 'country2']:\n            df[f'{col}_gesh'] = [gesh(s1, s2) for s1, s2 in zip(col_values, matcol_values)]\n            df[f'{col}_leven'] = [leven(s1, s2) for s1, s2 in zip(col_values, matcol_values)]\n            df[f'{col}_jaro'] = [jaro(s1, s2) for s1, s2 in zip(col_values, matcol_values)]\n            df[f'{col}_lcs_sequence'] = [lcs_sequence(s1, s2) for s1, s2 in zip(col_values, matcol_values)]\n            df[f'{col}_lcs_string'] = [lcs_string(s1, s2) for s1, s2 in zip(col_values, matcol_values)]\n\n            # # Measure against memory limitation\n            df[f'{col}_gesh'] = df[f'{col}_gesh'].astype(\"float16\")\n            df[f'{col}_jaro'] = df[f'{col}_jaro'].astype(\"float16\")\n        \n        if col not in ['country', 'name_lang', 'phone', 'zip', 'city2', 'country2']:\n            df[f'{col}_len'] = list(map(len, col_values))\n            df[f'match_{col}_len'] = list(map(len, matcol_values)) \n            df[f'{col}_len_diff'] = (np.abs(df[f'{col}_len'] - df[f'match_{col}_len'])).astype(\"int16\")\n            df[f'{col}_nleven'] = (df[f'{col}_leven'] / np.sqrt(df[f'{col}_len']*df[f'match_{col}_len'])).astype(\"float16\")\n            df[f'{col}_nlcs_sequence'] = (df[f'{col}_lcs_sequence'] / np.sqrt(df[f'{col}_len']*df[f'match_{col}_len'])).astype(\"float16\")\n            df[f'{col}_nlcs_string'] = (df[f'{col}_lcs_string'] / np.sqrt(df[f'{col}_len']*df[f'match_{col}_len'])).astype(\"float16\")\n\n            del df[f'match_{col}_len'], df[f'{col}_len']\n            gc_clear()\n                              \n        if col in ['categories']:\n            df[f'{col}_similarity'] = [categorical_similarity(s1, s2) for s1, s2 in zip(col_values, matcol_values)]\n            df[f'{col}_similarity'] = df[f'{col}_similarity'].astype(\"float16\")\n        \n        if col in ['name_lang', 'country', 'city2', 'country2']:\n            df[f'{col}_equal'] = [equal(s1, s2) for s1, s2 in zip(col_values, matcol_values)]\n            df[f'{col}_equal'] = df[f'{col}_equal'].fillna(-1).astype('int8')\n\n        del col_values, matcol_values\n        gc_clear()\n\n    # ====================================================\n    # Mean features\n    # ====================================================    \n    feat_columns2 = ['name', 'address', 'city', 'state', 'url', 'categories']\n\n    df[\"gesh_mean\"] = df[[f\"{col}_gesh\" for col in feat_columns2]].mean(axis=1).astype(\"float16\")\n    df[\"tfidf1_mean\"] = df[[f\"{col}_tfidf1_sim\" for col in vec_columns]].mean(axis=1).astype(\"float16\")\n    df[\"tfidf2_mean\"] = df[[f\"{col}_tfidf2_sim\" for col in vec_columns]].mean(axis=1).astype(\"float16\")\n    df[\"leven_mean\"] = df[[f\"{col}_leven\" for col in feat_columns2]].mean(axis=1).astype(\"float16\")\n    df[\"jaro_mean\"] = df[[f\"{col}_jaro\" for col in feat_columns2]].mean(axis=1).astype(\"float16\")\n    df[\"nlcs_sequence_mean\"] = df[[f\"{col}_nlcs_sequence\" for col in feat_columns2]].mean(axis=1).astype(\"float16\")\n    df[\"nlcs_string_mean\"] = df[[f\"{col}_nlcs_string\" for col in feat_columns2]].mean(axis=1).astype(\"float16\")\n\n    # Measure against memory limitation\n    for col in feat_columns:\n        if not col in ['country', 'name_lang', 'city2', 'country2']:\n            df[f'{col}_leven'] = df[f'{col}_leven'].fillna(99).astype(\"int16\")\n            df[f'{col}_lcs_sequence'] = df[f'{col}_lcs_sequence'].fillna(99).astype(\"int16\")\n            df[f'{col}_lcs_string'] = df[f'{col}_lcs_string'].fillna(99).astype(\"int16\")\n                    \n    return df\n\ndef add_other_features(original_df, df):\n    if len(original_df)>5:\n        processes = multiprocessing.cpu_count()\n        with multiprocessing.Pool(processes=processes) as pool:\n            df[\"idx_group\"] = df.index // (len(df) / processes)\n            len_df_gby = len(df.groupby('idx_group'))\n            dfs = pool.imap_unordered(_add_other_features, zip(df.groupby('idx_group'), [original_df for _ in range(len_df_gby)]))\n            dfs = tqdm(dfs, total=len_df_gby)\n            dfs = list(dfs)\n        df = pd.concat(dfs)\n        df.drop(columns=\"idx_group\", axis=1, inplace=True)\n        del dfs\n        return df\n    else:  \n        df = _add_other_features(((_, df), original_df))\n        return df","metadata":{"id":"59caea65","execution":{"iopub.status.busy":"2022-07-09T21:04:42.809171Z","iopub.execute_input":"2022-07-09T21:04:42.809806Z","iopub.status.idle":"2022-07-09T21:04:42.876155Z","shell.execute_reply.started":"2022-07-09T21:04:42.809766Z","shell.execute_reply":"2022-07-09T21:04:42.875373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Run","metadata":{}},{"cell_type":"code","source":"def preprocess(original_df):\n    original_df = original_df_preprocess(original_df)\n    df = recall_knn(original_df, Neighbors=min(CFG.n_neighbors, len(original_df)))\n    df = add_distance_features(original_df, df)\n    df = add_tfidf_features(original_df, df)\n    df = add_bert_features(original_df, df)\n    # df = add_finetuned_bert_features(original_df, df)\n    # ====================================================\n    #  preprocessing (and prediction when testing)\n    # ====================================================\n    new_df = pd.DataFrame()\n\n    for i in range(CFG.data_split):\n        df_temp = df[df[\"data_split\"]==i].reset_index(drop=True)\n        df_temp = add_other_features(original_df, df_temp)\n\n        # Prediction\n        if not CFG.train:\n            df_temp = inference(df_temp)\n    \n        new_df = new_df.append(df_temp, ignore_index=True)\n    \n        del df_temp\n        gc_clear()\n    \n    del df\n    gc_clear()\n\n    display(new_df)\n\n    return new_df","metadata":{"id":"6c6a1391","execution":{"iopub.status.busy":"2022-07-09T21:04:42.877357Z","iopub.execute_input":"2022-07-09T21:04:42.877861Z","iopub.status.idle":"2022-07-09T21:04:42.891365Z","shell.execute_reply.started":"2022-07-09T21:04:42.877821Z","shell.execute_reply":"2022-07-09T21:04:42.890455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if CFG.train:\n    df = pd.concat([\n        preprocess(original_df[original_df[\"set\"]==0].reset_index(drop=True)), \n        preprocess(original_df[original_df[\"set\"]==1].reset_index(drop=True)), \n    ]).reset_index(drop=True)\nelse:\n    df = preprocess(original_df)\ndisplay(df)","metadata":{"id":"f924117b","outputId":"cfdb9309-611b-4dc1-885c-eeb10b01e7b8","execution":{"iopub.status.busy":"2022-07-09T21:04:42.892575Z","iopub.execute_input":"2022-07-09T21:04:42.893329Z","iopub.status.idle":"2022-07-09T21:07:15.312176Z","shell.execute_reply.started":"2022-07-09T21:04:42.893292Z","shell.execute_reply":"2022-07-09T21:07:15.311362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if CFG.train:\n    with open(DATA_DIR + f\"df-exp{CFG.exp}.pkl\", mode=\"wb\") as f:\n        pickle.dump(df, f, protocol=4)","metadata":{"id":"0ba3ee1c","execution":{"iopub.status.busy":"2022-07-09T21:07:15.313274Z","iopub.execute_input":"2022-07-09T21:07:15.313824Z","iopub.status.idle":"2022-07-09T21:07:15.320910Z","shell.execute_reply.started":"2022-07-09T21:07:15.313784Z","shell.execute_reply":"2022-07-09T21:07:15.319824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if CFG.train:\n    with open(DATA_DIR + f'df-exp{CFG.exp}.pkl', 'rb') as f:\n        df = pickle.load(f)\n    display(df)","metadata":{"id":"b4b2fed8","outputId":"9ef5eed9-c6ff-47a4-cc3d-09d3ebd49575","execution":{"iopub.status.busy":"2022-07-09T21:07:15.322395Z","iopub.execute_input":"2022-07-09T21:07:15.322888Z","iopub.status.idle":"2022-07-09T21:07:15.340831Z","shell.execute_reply.started":"2022-07-09T21:07:15.322850Z","shell.execute_reply":"2022-07-09T21:07:15.335509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{"id":"03c68f30"}},{"cell_type":"code","source":"def make_fold(df):\n    unique_id = df[\"id\"].unique()\n    fold = np.zeros(len(df), dtype=int)\n    kf = KFold(n_splits=CFG.fold, shuffle=True, random_state=CFG.seed)\n    for i_fold, (_, va_group_idx) in enumerate(kf.split(unique_id)):\n        va_groups = unique_id[va_group_idx]\n        is_va = df[df[\"id\"].isin(va_groups)].index\n        fold[is_va] = i_fold\n    return fold","metadata":{"id":"7641bbdd","execution":{"iopub.status.busy":"2022-07-09T21:07:15.342823Z","iopub.execute_input":"2022-07-09T21:07:15.344822Z","iopub.status.idle":"2022-07-09T21:07:15.357159Z","shell.execute_reply.started":"2022-07-09T21:07:15.344781Z","shell.execute_reply":"2022-07-09T21:07:15.355602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calc_score(X_val, val_pred, threshold=CFG.threshold, pp=False):\n\n    new_df = pd.DataFrame(X_val[\"id\"].unique()).rename(columns={0: \"id\"})\n\n    tmp_df = X_val[val_pred >= threshold].groupby(\"id\")[\"match_id\"].apply(list).reset_index()\n    tmp_df.columns = [\"id\", \"matches\"]\n    tmp_df[\"matches\"] = tmp_df[\"matches\"].apply(lambda x: \" \".join(x))\n\n    new_df = pd.merge(new_df, tmp_df, on=\"id\", how=\"left\")\n    new_df[\"matches\"] = new_df[\"id\"] + \" \" + new_df[\"matches\"].fillna(\"\")\n\n    if pp:\n        score = get_score(post_process(new_df))\n    else:\n        score = get_score(new_df)\n\n    return score","metadata":{"id":"16dcb2c7","execution":{"iopub.status.busy":"2022-07-09T21:07:15.360508Z","iopub.execute_input":"2022-07-09T21:07:15.361994Z","iopub.status.idle":"2022-07-09T21:07:15.374757Z","shell.execute_reply.started":"2022-07-09T21:07:15.361953Z","shell.execute_reply":"2022-07-09T21:07:15.373296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run_catboost(param, df):\n    \n    oof_pred = np.zeros(len(df))\n    feature_importance_df = pd.DataFrame()\n    score_list = []\n    \n    folds_idx = make_fold(df)\n    oof_pred = np.zeros(len(df))\n\n    for fold in range(CFG.fold):\n\n        if fold in CFG.used_fold:\n\n            LOGGER.info(f\"==============================================\")\n            LOGGER.info(f\"▶︎ Start fold{fold} Training\")\n            LOGGER.info(f\"==============================================\")\n\n            tr_idx = np.argwhere(folds_idx != fold).reshape(-1)\n            va_idx = np.argwhere(folds_idx == fold).reshape(-1)\n            X_trn, X_val = df.loc[tr_idx].reset_index(drop=True), df.loc[va_idx].reset_index(drop=True)\n            y_trn, y_val = df.loc[tr_idx, CFG.target].reset_index(drop=True), df.loc[va_idx, CFG.target].reset_index(drop=True)\n\n            LOGGER.info(f\"train_shape: {X_trn.shape}, val_shape: {X_val.shape}\")\n\n            train = Pool(X_trn.drop(columns=[CFG.target]+DROP_COLS), y_trn, cat_features=CATEGORICAL_COL)\n            valid = Pool(X_val.drop(columns=[CFG.target]+DROP_COLS), y_val, cat_features=CATEGORICAL_COL)\n\n            model = CatBoost(param)\n            model = model.fit(\n                        train,\n                        eval_set=valid,\n                        use_best_model=True,\n                        early_stopping_rounds=100,\n                        verbose_eval=200\n                        )\n            \n            # ==============================================\n            # Feature Importances\n            # ==============================================\n\n            fold_importance_df = pd.DataFrame()\n            fold_importance_df[\"feature\"] = model.feature_names_\n            fold_importance_df[\"importance\"] = model.feature_importances_\n            fold_importance_df[\"fold\"] = fold\n            feature_importance_df = pd.concat([feature_importance_df, fold_importance_df], axis=0)\n\n            # ==============================================\n            # Calculate Score\n            # ==============================================\n            val_pred = model.predict(X_val.drop(columns=[CFG.target]+DROP_COLS), prediction_type='Probability').T[1]\n            score = calc_score(X_val, val_pred)\n\n            LOGGER.info(f\"fold{fold} score: {score:.6f}\")\n\n            oof_pred[va_idx] = val_pred\n            score_list.append([fold, score])\n\n            # ==============================================\n            # Save model\n            # ==============================================\n            pickle.dump(model, open(OUTPUT_DIR + f'model_fold{fold}.pkl', 'wb'))\n            \n            del model, X_trn, X_val, y_trn, y_val, train, valid\n            gc_clear()\n    \n    score_df = pd.DataFrame(\n        score_list, columns=[\"fold\", \"IoU\"])\n    \n    # ==============================================\n    # Create Kaggle Dataset\n    # ==============================================\n\n    !kaggle datasets init -p $OUTPUT_DIR\n\n    metadata = {\"id\": f\"shkanda/foursquare-dataset-exp{CFG.exp}\",\n                    \"title\": f\"foursquare-dataset-exp{CFG.exp}\",\n                    \"licenses\": [{\"name\": \"CC0-1.0\"}]}\n\n    with open(OUTPUT_DIR+'dataset-metadata.json', 'w') as fp:\n        json.dump(metadata, fp)\n\n    !kaggle datasets create -p $OUTPUT_DIR\n \n    return oof_pred, score_df, feature_importance_df\n\ndef show_feature_importance(feature_importance_df):\n    order = list(feature_importance_df.groupby(\"feature\").mean().sort_values(\"importance\", ascending=False).index)\n    plt.figure(figsize=(10, 15))\n    sns.barplot(x=\"importance\", y=\"feature\", data=feature_importance_df, order=order)\n    plt.title(\"feature importance\")\n    plt.tight_layout()\n    plt.savefig(OUTPUT_DIR+'feature_importance.png')","metadata":{"id":"a8b2b301","execution":{"iopub.status.busy":"2022-07-09T21:07:15.376623Z","iopub.execute_input":"2022-07-09T21:07:15.377229Z","iopub.status.idle":"2022-07-09T21:07:15.439274Z","shell.execute_reply.started":"2022-07-09T21:07:15.377189Z","shell.execute_reply":"2022-07-09T21:07:15.438629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if CFG.train:\n    oof_pred, score_df, feature_importance_df = run_catboost(PARAMS, df)\n    display(score_df)","metadata":{"id":"3285c2b7","outputId":"43108586-6469-4a91-9ba0-e75387c881f6","execution":{"iopub.status.busy":"2022-07-09T21:07:15.443284Z","iopub.execute_input":"2022-07-09T21:07:15.444058Z","iopub.status.idle":"2022-07-09T21:07:15.452054Z","shell.execute_reply.started":"2022-07-09T21:07:15.443993Z","shell.execute_reply":"2022-07-09T21:07:15.451314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if CFG.train:\n    show_feature_importance(feature_importance_df)","metadata":{"id":"ccc33409","outputId":"961751ff-8714-49de-9b46-cebef2566afc","execution":{"iopub.status.busy":"2022-07-09T21:07:15.454815Z","iopub.execute_input":"2022-07-09T21:07:15.456027Z","iopub.status.idle":"2022-07-09T21:07:15.464262Z","shell.execute_reply.started":"2022-07-09T21:07:15.455821Z","shell.execute_reply":"2022-07-09T21:07:15.463607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if CFG.train:\n    df[\"pred\"] = oof_pred\n    with open(DATA_DIR + f\"oof_df-exp{CFG.exp}.pkl\", mode=\"wb\") as f:\n        pickle.dump(df, f, protocol=4)","metadata":{"id":"aS3t_z8L9eCx","execution":{"iopub.status.busy":"2022-07-09T21:07:15.465449Z","iopub.execute_input":"2022-07-09T21:07:15.466306Z","iopub.status.idle":"2022-07-09T21:07:15.477733Z","shell.execute_reply.started":"2022-07-09T21:07:15.466267Z","shell.execute_reply":"2022-07-09T21:07:15.476788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submit","metadata":{"id":"3f78508c"}},{"cell_type":"code","source":"if not(CFG.train):\n    sub = pd.merge(original_df[[\"id\"]], df, on=\"id\", how=\"left\")\n    sub[\"matches\"] = (sub[\"id\"] + \" \" + sub[\"matches\"].fillna(\"\")).apply(lambda x: x.strip())\n    sub = post_process(sub)\n    sub.to_csv(\"submission.csv\", index=False)\n    display(sub)","metadata":{"id":"da4d4f06","execution":{"iopub.status.busy":"2022-07-09T21:07:15.479265Z","iopub.execute_input":"2022-07-09T21:07:15.479605Z","iopub.status.idle":"2022-07-09T21:07:15.518011Z","shell.execute_reply.started":"2022-07-09T21:07:15.479572Z","shell.execute_reply":"2022-07-09T21:07:15.517289Z"},"trusted":true},"execution_count":null,"outputs":[]}]}