{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"nadare_data_dir = \"/kaggle/input/nadare-4sq-data\"\nnadare_feature_dir = \"/kaggle/input/nadare-4sq-data-public/FEAT32/\"\nnadare_util_dir = \"/kaggle/input/nadare-4sq-data-public/common_utils/\"","metadata":{"ExecuteTime":{"end_time":"2022-06-25T11:17:58.286052Z","start_time":"2022-06-25T11:17:58.275053Z"},"papermill":{"duration":0.040326,"end_time":"2022-07-07T12:59:36.255614","exception":false,"start_time":"2022-07-07T12:59:36.215288","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:49:51.317396Z","iopub.execute_input":"2022-07-08T02:49:51.317909Z","iopub.status.idle":"2022-07-08T02:49:51.326021Z","shell.execute_reply.started":"2022-07-08T02:49:51.317864Z","shell.execute_reply":"2022-07-08T02:49:51.325202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport tensorflow as tf\ngpus = tf.config.experimental.list_physical_devices('GPU')\nif gpus:\n  # Restrict TensorFlow to only allocate 1GB of memory on the first GPU\n  try:\n    tf.config.experimental.set_virtual_device_configuration(\n        gpus[0],\n        [tf.config.experimental.VirtualDeviceConfiguration(memory_limit=1024*13)])\n    logical_gpus = tf.config.experimental.list_logical_devices('GPU')\n    print(len(gpus), \"Physical GPUs,\", len(logical_gpus), \"Logical GPUs\")\n  except RuntimeError as e:\n    # Virtual devices must be set before GPUs have been initialized\n    print(e)\n","metadata":{"papermill":{"duration":7.693821,"end_time":"2022-07-07T12:59:43.972590","exception":false,"start_time":"2022-07-07T12:59:36.278769","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:49:53.286312Z","iopub.execute_input":"2022-07-08T02:49:53.286814Z","iopub.status.idle":"2022-07-08T02:50:00.146143Z","shell.execute_reply.started":"2022-07-08T02:49:53.286775Z","shell.execute_reply":"2022-07-08T02:50:00.145266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cuml\ncuml.set_global_output_type('numpy')","metadata":{"papermill":{"duration":4.593628,"end_time":"2022-07-07T12:59:48.620082","exception":false,"start_time":"2022-07-07T12:59:44.026454","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:50:00.147477Z","iopub.execute_input":"2022-07-08T02:50:00.148144Z","iopub.status.idle":"2022-07-08T02:50:03.859850Z","shell.execute_reply.started":"2022-07-08T02:50:00.148104Z","shell.execute_reply":"2022-07-08T02:50:03.859013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install {nadare_util_dir}cdifflib-1.2.5-cp37-cp37m-linux_x86_64.whl --no-deps\n!pip install {nadare_util_dir}polyleven-0.7-cp37-cp37m-linux_x86_64.whl --no-deps\n!pip install {nadare_util_dir}jarowinkler-1.0.5-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl --no-deps","metadata":{"papermill":{"duration":66.688938,"end_time":"2022-07-07T13:00:55.352788","exception":false,"start_time":"2022-07-07T12:59:48.663850","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:50:20.182352Z","iopub.execute_input":"2022-07-08T02:50:20.182825Z","iopub.status.idle":"2022-07-08T02:51:27.256893Z","shell.execute_reply.started":"2022-07-08T02:50:20.182778Z","shell.execute_reply":"2022-07-08T02:51:27.255781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!unzip {nadare_util_dir}pykakasi_deps.dontopenthiskaggle/pykakasi_deps.dontopenthiskaggle -d .","metadata":{"ExecuteTime":{"end_time":"2022-06-25T11:17:48.103193Z","start_time":"2022-06-25T11:17:48.059193Z"},"papermill":{"duration":0.823217,"end_time":"2022-07-07T13:00:56.198949","exception":false,"start_time":"2022-07-07T13:00:55.375732","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:51:27.262230Z","iopub.execute_input":"2022-07-08T02:51:27.263956Z","iopub.status.idle":"2022-07-08T02:51:28.189631Z","shell.execute_reply.started":"2022-07-08T02:51:27.263913Z","shell.execute_reply":"2022-07-08T02:51:28.188337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!conda install ./pykakasi_deps/offline_pykakasi.tar.bz2\n!conda install ./pykakasi_deps/offline_jaconv.tar.bz2\n!conda install ./pykakasi_deps/offline_deprecated.tar.bz2","metadata":{"ExecuteTime":{"start_time":"2022-06-25T11:17:48.298Z"},"papermill":{"duration":30.302686,"end_time":"2022-07-07T13:01:26.524511","exception":false,"start_time":"2022-07-07T13:00:56.221825","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:51:28.193399Z","iopub.execute_input":"2022-07-08T02:51:28.196699Z","iopub.status.idle":"2022-07-08T02:51:58.712202Z","shell.execute_reply.started":"2022-07-08T02:51:28.196655Z","shell.execute_reply":"2022-07-08T02:51:58.711200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install {nadare_util_dir}tensorflow_text-2.6.0-cp37-cp37m-manylinux1_x86_64.whl --no-deps\n!pip install {nadare_util_dir}tensorflow_ranking-0.5.0-py2.py3-none-any.whl --no-deps\n!pip install {nadare_util_dir}mojimoji-0.0.12-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl --no-deps\n!pip install {nadare_util_dir}num2words-0.5.10-py3-none-any.whl\n!pip install {nadare_util_dir}phonenumbers-8.12.48-py2.py3-none-any.whl --no-deps\n!pip install {nadare_util_dir}pycnnum-1.0.1-py3-none-any.whl\n!pip install {nadare_util_dir}SudachiPy-0.6.5-cp37-cp37m-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_12_x86_64.manylinux2010_x86_64.whl --no-deps\n!pip install {nadare_util_dir}SudachiDict_core-20220519-py3-none-any.whl --no-deps\n!pip install {nadare_util_dir}SudachiDict_full-20220519-py3-none-any.whl --no-deps\n!pip install {nadare_util_dir}pythainlp-3.0.8-py3-none-any.whl --no-deps\n!pip install {nadare_util_dir}tinydb-4.7.0-py3-none-any.whl --no-deps","metadata":{"papermill":{"duration":273.557136,"end_time":"2022-07-07T13:06:00.110298","exception":false,"start_time":"2022-07-07T13:01:26.553162","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:51:58.716201Z","iopub.execute_input":"2022-07-08T02:51:58.716784Z","iopub.status.idle":"2022-07-08T02:56:32.263480Z","shell.execute_reply.started":"2022-07-08T02:51:58.716748Z","shell.execute_reply":"2022-07-08T02:56:32.262468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport tensorflow_text as tf_text\n\nfrom tqdm.notebook import tqdm\nimport re\n\nimport gc","metadata":{"ExecuteTime":{"end_time":"2022-06-25T11:18:02.272391Z","start_time":"2022-06-25T11:17:58.769254Z"},"papermill":{"duration":1.338265,"end_time":"2022-07-07T13:06:01.476012","exception":false,"start_time":"2022-07-07T13:06:00.137747","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:56:32.266297Z","iopub.execute_input":"2022-07-08T02:56:32.267132Z","iopub.status.idle":"2022-07-08T02:56:33.736800Z","shell.execute_reply.started":"2022-07-08T02:56:32.267081Z","shell.execute_reply":"2022-07-08T02:56:33.735807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_df = pd.read_csv(\"../input/foursquare-location-matching/train.csv\", encoding=\"utf-8\")\ntest_df = pd.read_csv(\"../input/foursquare-location-matching/test.csv\", encoding=\"utf-8\")\nDEBUG = False\nif len(test_df) == 5:\n    DEBUG = True\n    test_df = pd.read_csv(\"../input/foursquare-location-matching/train.csv\", encoding=\"utf-8\").drop(\"point_of_interest\", axis=1).sample(10000).reset_index(drop=True)\n#test_df.loc[0, \"name\"] = np.NaN","metadata":{"ExecuteTime":{"end_time":"2022-06-25T11:18:04.825627Z","start_time":"2022-06-25T11:18:02.289393Z"},"papermill":{"duration":8.95065,"end_time":"2022-07-07T13:06:10.457064","exception":false,"start_time":"2022-07-07T13:06:01.506414","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:56:33.741379Z","iopub.execute_input":"2022-07-08T02:56:33.741781Z","iopub.status.idle":"2022-07-08T02:56:42.200680Z","shell.execute_reply.started":"2022-07-08T02:56:33.741742Z","shell.execute_reply":"2022-07-08T02:56:42.199819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_id_map = {v: i for i, v in enumerate(test_df[\"id\"].values)}\ntest_df[\"ix\"] = test_df[\"id\"].map(test_id_map)\ntest_df[\"categories\"] = test_df[\"categories\"].fillna(\"\")\ntest_df[\"pid\"] = -1\ntest_df[\"group\"] = -1","metadata":{"ExecuteTime":{"end_time":"2022-06-25T11:18:05.367627Z","start_time":"2022-06-25T11:18:04.826628Z"},"papermill":{"duration":0.054995,"end_time":"2022-07-07T13:06:10.539879","exception":false,"start_time":"2022-07-07T13:06:10.484884","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:56:42.202028Z","iopub.execute_input":"2022-07-08T02:56:42.202484Z","iopub.status.idle":"2022-07-08T02:56:42.225195Z","shell.execute_reply.started":"2022-07-08T02:56:42.202445Z","shell.execute_reply":"2022-07-08T02:56:42.224323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import unicodedata\ntest_df[\"name\"] = test_df[\"name\"].fillna(\"\").apply(lambda x: unicodedata.normalize(\"NFKC\", x))\ntest_df[\"address\"] = test_df[\"address\"].fillna(\"\").apply(lambda x: unicodedata.normalize(\"NFKC\", x))\n","metadata":{"papermill":{"duration":0.042438,"end_time":"2022-07-07T13:06:10.610314","exception":false,"start_time":"2022-07-07T13:06:10.567876","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:56:42.226750Z","iopub.execute_input":"2022-07-08T02:56:42.227297Z","iopub.status.idle":"2022-07-08T02:56:42.248406Z","shell.execute_reply.started":"2022-07-08T02:56:42.227255Z","shell.execute_reply":"2022-07-08T02:56:42.247553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from fasttext import load_model\nft_model = load_model(nadare_util_dir + \"lid.176.bin\")\ndef predict_language(text):\n    label, prob = ft_model.predict(text, 1)\n    return list(zip([l.replace(\"__label__\", \"\") for l in label], prob))[0][0]\n\ntest_df[\"name_language\"] = np.vectorize(predict_language)(test_df[\"name\"].fillna(\"\"))\ntest_df[\"address_language\"] = np.vectorize(predict_language)(test_df[\"address\"].fillna(\"\"))\n\ndef language_pattern(country, language):\n    if country == \"JP\":\n        return 0\n    elif country == \"TH\":\n        return 1\n    elif country == \"CN\":\n        return 2\n    elif language == \"ja\":\n        return 0\n    elif language == \"th\":\n        return 1\n    elif language == \"zh\":\n        return 2\n    else:\n        return 3\ntest_df[\"name_handle_pattern\"] = np.vectorize(language_pattern)(test_df[\"country\"], test_df[\"name_language\"])     \ntest_df[\"address_handle_pattern\"] = np.vectorize(language_pattern)(test_df[\"country\"], test_df[\"address_language\"])\nlang_vc = test_df[\"name_language\"].value_counts()\ndel ft_model","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:11:24.951488Z","start_time":"2022-06-25T01:11:11.655662Z"},"papermill":{"duration":1.586885,"end_time":"2022-07-07T13:06:12.224475","exception":false,"start_time":"2022-07-07T13:06:10.637590","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:56:42.249627Z","iopub.execute_input":"2022-07-08T02:56:42.249961Z","iopub.status.idle":"2022-07-08T02:56:43.623819Z","shell.execute_reply.started":"2022-07-08T02:56:42.249920Z","shell.execute_reply":"2022-07-08T02:56:43.622980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from num2words import num2words\nfrom collections import defaultdict\n\nnum2words_languages = ['en', 'tr', 'ru', 'es', 'de', 'th', 'id', 'fr', 'pt', 'it', 'nl',\n       'ko', 'ar', 'sv', 'uk', \"pl\"]\n\nword_number_df = []\nfor lang in tqdm(num2words_languages):\n    for i in range(1001):\n        try:\n            word = num2words(i, lang=lang)\n        except:\n            word = np.NaN\n        try:\n            ordinal = num2words(i, lang=lang, to=\"ordinal\")\n        except:\n            ordinal = np.NaN\n        try:\n            ordinal_num = num2words(i, lang=lang, to=\"ordinal_num\")\n        except:\n            ordinal_num = np.NaN\n        if (pd.isna(ordinal)) or (pd.isna(ordinal_num)) or (ordinal == ordinal_num):\n            ordinal_num = i\n        if lang == \"ko\" and not pd.isna(ordinal):\n            ordinal = ordinal[:-3]\n            ordinal_num = ordinal_num[:-3] \n        \n        word_number_df.append({\"lang\": lang,\n                               \"number\": i,\n                               \"word\": word,\n                               \"ordinal\": ordinal,\n                               \"ordinal_num\": ordinal_num})\n\nword_number_df = pd.DataFrame(word_number_df)\nword_number_df[\"lang_count\"] = word_number_df[\"lang\"].apply(lambda x: lang_vc.get(x, 0))\n\nword2num_dict = {}\npossibly_have_next_dict = {}\n\nfor i, row in word_number_df.sort_values(by=\"lang_count\").iterrows():\n    if not pd.isna(row[\"word\"]):\n        word = row[\"word\"]\n        if not word in possibly_have_next_dict.keys():\n            possibly_have_next_dict[word] = False\n\n        word2num_dict[word] = str(row[\"number\"])\n        word2num_dict[word.replace(\"-\", \" \")] = str(row[\"number\"])\n        \n        wsp = word.split()\n        for i in range(1, len(wsp)-1):\n            possibly_have_next_dict[\" \".join(wsp[:i])] = True\n        wsp = word.replace(\"-\", \" \").split()[:-1]\n        for i in range(1, len(wsp)-1):\n            possibly_have_next_dict[\" \".join(wsp[:i])] = True\n        \n    if (not pd.isna(row[\"ordinal\"])) and (not pd.isna(row[\"ordinal_num\"])):\n        word = row[\"ordinal\"]\n        if not word in possibly_have_next_dict.keys():\n            possibly_have_next_dict[word] = False\n        word2num_dict[word] = str(row[\"ordinal_num\"])\n        word2num_dict[word.replace(\"-\", \" \")] = str(row[\"ordinal_num\"])\n        \n        wsp = word.split()\n        for i in range(1, len(wsp)-1):\n            possibly_have_next_dict[\" \".join(wsp[:i])] = True\n        wsp = word.replace(\"-\", \" \").split()[:-1]\n        for i in range(1, len(wsp)-1):\n            possibly_have_next_dict[\" \".join(wsp[:i])] = True\n        ","metadata":{"papermill":{"duration":3.232895,"end_time":"2022-07-07T13:06:15.484199","exception":false,"start_time":"2022-07-07T13:06:12.251304","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:56:43.627218Z","iopub.execute_input":"2022-07-08T02:56:43.627540Z","iopub.status.idle":"2022-07-08T02:56:46.349469Z","shell.execute_reply.started":"2022-07-08T02:56:43.627514Z","shell.execute_reply":"2022-07-08T02:56:46.348588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import string\nimport pycnnum\nimport pythainlp\nfrom pythainlp.util import thai_digit_to_arabic_digit\nfrom sudachipy import Dictionary, SplitMode\nimport mojimoji\nimport pykakasi\nfrom unidecode import unidecode\nimport emoji\n\nemoji_set = set()\nfor k, v in emoji.UNICODE_EMOJI.items():\n    emoji_set.update(v.keys())\nthai_word2num_dict = {num2words(i, lang=\"th\"):str(i) for i in range(10001)}\nCHINESE_DIGITS_PATTERN = re.compile(r'[〇〇零一二三四五六七八九零壹贰叁肆伍陆柒捌玖十百千万拾佰仟萬]+')\nremove_punctuations = '!\"#$%\\'()*+,./:;<=>?@[\\\\]^_`{|}~'\npunctuation_translater = str.maketrans(remove_punctuations, \" \"*len(remove_punctuations))\njapanese_char2num_translater = str.maketrans(\"〇零一壱二弐三参四肆五伍六陸七漆八捌九玖\", \"00112233445566778899\")\n\n\nkakasi = pykakasi.kakasi()\nkakasi.setMode('H', 'a')  # Convert Hiragana into alphabet\nkakasi.setMode('K', 'a')  # Convert Katakana into alphabet\nkakasi.setMode('J', 'a')  # Convert Kanji into alphabet\nconversion = kakasi.getConverter()\n\nforce_katakana = pykakasi.kakasi()\nkakasi.setMode('H', 'K')  # Convert Hiragana into alphabet\nkakasi.setMode('K', 'K')  # Convert Katakana into alphabet\nforce_katakana.setMode('J', 'K')  # Convert Hiragana into alphabet\nforce_katakana_conversion = force_katakana.getConverter()\n\nsudachi_tokenizer = Dictionary(dict_type=\"full\").create()\nsudachi_tokenize_mode = SplitMode.A\n\nzh_segmenter = tf_text.HubModuleTokenizer(nadare_util_dir + \"zh_segmentation\")\n\ndef remove_emoji(x):\n    return ''.join([c for c in x if c not in emoji_set])\n\ndef remove_marks(x):\n    x = re.sub(r'[^a-zA-Z0-9& ]', ' ', x)\n    x = re.sub(r' +', ' ', x)\n    return x\n\ndef zenhan_normalize(x):\n    return mojimoji.han_to_zen(mojimoji.zen_to_han(x, kana=False), digit=False, ascii=False)\n\ndef japanese_wakachi_reading(morphemes):\n    readings = []\n    for m in morphemes:\n        if (m.normalized_form() == \" \") or (m.reading_form() == \"キゴウ\"):\n            continue\n        if m.normalized_form().endswith(\"丁目\") and m.normalized_form() != \"丁目\":\n            norm = m.normalized_form().translate(japanese_char2num_translater)\n            norm = re.sub(r\"(\\d+)\", r\" \\1 \", norm)\n            rs = japanese_wakachi_reading(sudachi_tokenizer.tokenize(norm, sudachi_tokenize_mode))\n            readings.extend(rs)\n        elif m.normalized_form().isdigit():\n            readings.append(m.normalized_form())\n        else:\n            readings.append(force_katakana_conversion.do(m.reading_form()))\n    return readings\n\ndef handle_japanese(x):\n    x = zenhan_normalize(x)\n    morphemes = sudachi_tokenizer.tokenize(x, sudachi_tokenize_mode)\n    reading_forms = japanese_wakachi_reading(morphemes)\n    romanize_forms = conversion.do(\" \".join(reading_forms))\n    return \" \".join(reading_forms), romanize_forms\n\ndef decode_list(x):\n    if type(x) is list:\n        return list(map(decode_list, x))\n    return x.decode(\"UTF-8\")\n\ndef decode_utf8_tensor(x):\n    return list(map(decode_list, x.to_list()))\n\ndef handle_chinese(x):\n    x = x.replace(\"'\", \"\").translate(punctuation_translater).lower()\n    if re.fullmatch(\"\\s*\", x):\n        return \"\", \"\"\n    with tf.device(\"CPU:0\"):\n        words_list = decode_utf8_tensor(zh_segmenter.tokenize(x.split()))\n        \n    number_words = []\n    romanized_words = []\n    for words in words_list:\n        num_word = []\n        roman_word = []\n        for word in words:\n            if re.fullmatch(CHINESE_DIGITS_PATTERN, word):\n                word = str(pycnnum.cn2num(word))\n            if (word[0]) == \"第\" and re.fullmatch(CHINESE_DIGITS_PATTERN, word[1:]):\n                num = str(pycnnum.cn2num(word[1:]))\n                word = \"第\"\n                num_word.append(word + num)\n                roman_word.append(unidecode(word[0]).lower().replace(\" \", \"\") + \" \" + num)\n            else:\n                num_word.append(word)\n                roman_word.append(unidecode(word).lower().replace(\" \", \"\"))\n        number_words.append(\"\".join(num_word))\n        romanized_words.append(\" \".join(roman_word))\n    return \" \".join(number_words), \" \".join(romanized_words)\n\ndef handle_thai(x):\n    x = thai_digit_to_arabic_digit(x)\n    tokenized_word = pythainlp.tokenize.word_tokenize(x, engine=\"newmm\")\n    number_words = []\n    romanized_words = []\n    for word in tokenized_word:\n        word = thai_word2num_dict.get(word, word)\n        number_words.append(word)\n        try:\n            romanized_words.append(pythainlp.romanize(word, engine='royin'))\n        except:\n            pass\n    return \"\".join(number_words), \" \".join(romanized_words)\n\n\ndef language_normalize(word):\n    result = []\n    word = word.replace(\"'\", \"\").translate(punctuation_translater).lower()\n    tmp = \"\"\n    for w in word.split():\n        if len(tmp):\n            tmp = tmp + \" \" + w\n            if tmp in possibly_have_next_dict.keys():\n                if possibly_have_next_dict[tmp]:\n                    continue\n                else:\n                    result.append(word2num_dict[tmp])\n                    tmp = \"\"\n        else:\n            if w in possibly_have_next_dict.keys():\n                if possibly_have_next_dict[w]:\n                    tmp = w\n                    continue\n                else:\n                    result.append(word2num_dict[w])\n            else:\n                result.append(w)\n    if len(tmp):\n        result.append(word2num_dict.get(tmp, tmp))\n            \n    return re.sub(r' +', ' ', \" \".join(result))\n\ndef language_handle(handle_pattern, text):\n    if handle_pattern == 0:\n        normalized, romanized = handle_japanese(text)\n    elif handle_pattern == 1:\n        normalized, romanized = handle_thai(text)\n    elif handle_pattern == 2:\n        normalized, romanized = handle_chinese(text)\n    return normalized, romanized\n","metadata":{"papermill":{"duration":6.916585,"end_time":"2022-07-07T13:06:22.433229","exception":false,"start_time":"2022-07-07T13:06:15.516644","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:56:46.354279Z","iopub.execute_input":"2022-07-08T02:56:46.354676Z","iopub.status.idle":"2022-07-08T02:56:53.755178Z","shell.execute_reply.started":"2022-07-08T02:56:46.354637Z","shell.execute_reply":"2022-07-08T02:56:53.754374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"name_handle_index = np.where((test_df[\"name_handle_pattern\"] <= 2) & (test_df[\"name\"] != \"\"))[0]\n\nname_normalized = []\nname_romanized = []\nfor pattern, name in tqdm(zip(test_df[\"name_handle_pattern\"].values[name_handle_index],\n                              test_df[\"name\"].fillna(\"\").astype(str).values[name_handle_index]), total=len(name_handle_index)):\n    normalized, romanized = language_handle(pattern, name)\n    name_normalized.append(normalized)\n    name_romanized.append(romanized)","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:20:08.941066Z","start_time":"2022-06-25T01:11:28.19259Z"},"papermill":{"duration":5.519473,"end_time":"2022-07-07T13:06:27.980175","exception":false,"start_time":"2022-07-07T13:06:22.460702","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:56:53.756315Z","iopub.execute_input":"2022-07-08T02:56:53.756969Z","iopub.status.idle":"2022-07-08T02:56:59.415284Z","shell.execute_reply.started":"2022-07-08T02:56:53.756933Z","shell.execute_reply":"2022-07-08T02:56:59.414478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"address_handle_index = np.where((test_df[\"address_handle_pattern\"] <= 2) & (test_df[\"name\"] != \"\"))[0]\n\naddress_normalized = []\naddress_romanized = []\nfor pattern, address in tqdm(zip(test_df[\"address_handle_pattern\"].values[address_handle_index],\n                              test_df[\"address\"].fillna(\"\").astype(str).values[address_handle_index]), total=len(address_handle_index)):\n    normalized, romanized = language_handle(pattern, address)\n    address_normalized.append(normalized)\n    address_romanized.append(romanized)","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:32:53.123103Z","start_time":"2022-06-25T01:20:08.942067Z"},"papermill":{"duration":0.8046,"end_time":"2022-07-07T13:06:28.812745","exception":false,"start_time":"2022-07-07T13:06:28.008145","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:56:59.416480Z","iopub.execute_input":"2022-07-08T02:56:59.417230Z","iopub.status.idle":"2022-07-08T02:57:00.380344Z","shell.execute_reply.started":"2022-07-08T02:56:59.417192Z","shell.execute_reply":"2022-07-08T02:57:00.379497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ndel zh_segmenter, kakasi, conversion, sudachi_tokenizer\ngc.collect()","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:32:53.556559Z","start_time":"2022-06-25T01:32:53.124105Z"},"papermill":{"duration":1.237558,"end_time":"2022-07-07T13:06:30.078678","exception":false,"start_time":"2022-07-07T13:06:28.841120","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:00.384226Z","iopub.execute_input":"2022-07-08T02:57:00.386672Z","iopub.status.idle":"2022-07-08T02:57:01.973050Z","shell.execute_reply.started":"2022-07-08T02:57:00.386633Z","shell.execute_reply":"2022-07-08T02:57:01.972305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[\"name_normalized\"] = np.vectorize(language_normalize)(test_df[\"name\"].fillna(\"\").astype(str))\ntest_df[\"address_normalized\"] = np.vectorize(language_normalize)(test_df[\"address\"].fillna(\"\").astype(str))\n\n\nfrom_cols = [\"name_normalized\", \"address_normalized\"]\nto_cols = ['sp_name', \"sp_address_raw\"]\n\nfor from_, to_ in zip(from_cols, to_cols):\n    test_df[to_] = test_df[from_].fillna('').apply(unidecode)\n    \ntest_df.loc[name_handle_index, \"name_normalized\"] = name_normalized\ntest_df.loc[name_handle_index, \"sp_name\"] = name_romanized\n\ntest_df.loc[address_handle_index, \"address_normalized\"] = address_normalized\ntest_df.loc[address_handle_index, \"sp_address_raw\"] = address_romanized\n\nfor from_, to_ in zip(from_cols, to_cols):\n    test_df[to_] = test_df[to_].apply(unidecode)\n    test_df[to_] = test_df[to_].str.lower()\n    test_df[to_] = test_df[to_].apply(remove_emoji)\n    test_df[to_] = test_df[to_].apply(remove_marks)","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:33:01.045885Z","start_time":"2022-06-25T01:32:53.556559Z"},"papermill":{"duration":0.446629,"end_time":"2022-07-07T13:06:30.553138","exception":false,"start_time":"2022-07-07T13:06:30.106509","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:01.974268Z","iopub.execute_input":"2022-07-08T02:57:01.974616Z","iopub.status.idle":"2022-07-08T02:57:02.373130Z","shell.execute_reply.started":"2022-07-08T02:57:01.974580Z","shell.execute_reply":"2022-07-08T02:57:02.372278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsRegressor\naddress_index = np.where(~np.vectorize(lambda x: bool(re.fullmatch(\"\\s*\", x)))(test_df[\"sp_address_raw\"].fillna(\"\")))[0]\nknn = KNeighborsRegressor(n_neighbors=min(3, (~test_df[\"address\"].isna()).sum()), \n                          metric='haversine', \n                          n_jobs=-1)\nknn.fit(np.deg2rad(test_df[['latitude','longitude']].values)[address_index], address_index)\naddress_neighbor = address_index[knn.kneighbors(np.deg2rad(test_df[['latitude','longitude']].values))[1]]\ntest_df[\"sp_address\"] = np.vectorize(lambda x: \" \".join(x))(pd.Series(test_df[\"sp_address_raw\"].values[address_neighbor].tolist()))\ntest_df[\"use_address\"] = np.vectorize(lambda x: \"[SEP]\".join(x))(pd.Series(test_df[\"address\"].values[address_neighbor].tolist()))\n\ntest_df[\"sp_name\"] = test_df[\"sp_name\"].apply(remove_marks)\ntest_df[\"sp_address\"] = test_df[\"sp_address\"].apply(remove_marks)","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:35:36.095326Z","start_time":"2022-06-25T01:33:55.36536Z"},"papermill":{"duration":0.807467,"end_time":"2022-07-07T13:06:31.388688","exception":false,"start_time":"2022-07-07T13:06:30.581221","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:02.374596Z","iopub.execute_input":"2022-07-08T02:57:02.374948Z","iopub.status.idle":"2022-07-08T02:57:03.139272Z","shell.execute_reply.started":"2022-07-08T02:57:02.374912Z","shell.execute_reply":"2022-07-08T02:57:03.138424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\ncountry_index_df = pd.read_csv(nadare_feature_dir + 'country_tsp_index.csv', encoding=\"utf-8\")\nstate_index_df = pd.read_csv(nadare_feature_dir + 'state_tsp_index.csv', encoding=\"utf-8\")\ncity_index_df = pd.read_csv(nadare_feature_dir + 'city_tsp_index.csv', encoding=\"utf-8\")\ngeo_name_index_df = pd.read_csv(nadare_feature_dir + 'geo_name_tsp_index.csv', encoding=\"utf-8\")\ncategory_index_df = pd.read_csv(nadare_feature_dir + 'category_tsp_index.csv', encoding=\"utf-8\").fillna(\"nan\")\n","metadata":{"ExecuteTime":{"end_time":"2022-06-25T11:18:40.487979Z","start_time":"2022-06-25T11:18:40.461977Z"},"papermill":{"duration":0.107417,"end_time":"2022-07-07T13:06:31.524051","exception":false,"start_time":"2022-07-07T13:06:31.416634","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:03.140462Z","iopub.execute_input":"2022-07-08T02:57:03.142161Z","iopub.status.idle":"2022-07-08T02:57:03.199579Z","shell.execute_reply.started":"2022-07-08T02:57:03.142121Z","shell.execute_reply":"2022-07-08T02:57:03.198788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"country_index_map = {country: i for country, i in country_index_df[[\"country\", \"country_index\"]].values}\nstate_index_map = {state: i for state, i in state_index_df[[\"state\", \"state_index\"]].values}\ncity_index_map = {city: i for city, i in city_index_df[[\"city\", \"city_index\"]].values}\ngeo_name_index_map = {geo_name: i for geo_name, i in geo_name_index_df[[\"geo_name\", \"geo_name_index\"]].values}","metadata":{"ExecuteTime":{"end_time":"2022-06-25T11:18:51.76745Z","start_time":"2022-06-25T11:18:51.751447Z"},"papermill":{"duration":0.049339,"end_time":"2022-07-07T13:06:31.601392","exception":false,"start_time":"2022-07-07T13:06:31.552053","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:03.200983Z","iopub.execute_input":"2022-07-08T02:57:03.201506Z","iopub.status.idle":"2022-07-08T02:57:03.217861Z","shell.execute_reply.started":"2022-07-08T02:57:03.201465Z","shell.execute_reply":"2022-07-08T02:57:03.216464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def haversine(X, Y):\n    delta = Y - X\n    x_lats = tf.gather(X, 0, axis=-1)\n    y_lats = tf.gather(Y, 0, axis=-1)\n    dlat = tf.gather(delta, 0, axis=-1)\n    dlon = tf.gather(delta, 1, axis=-1)\n\n    a = tf.sin(dlat/2) * tf.sin(dlat/2) + tf.cos(x_lats) * tf.cos(y_lats) * tf.sin(dlon/2) * tf.sin(dlon/2)\n    c = 2 * tf.math.atan2(tf.sqrt(a), tf.sqrt(1-a))\n    return c","metadata":{"ExecuteTime":{"end_time":"2022-06-25T11:22:27.304434Z","start_time":"2022-06-25T11:22:27.293436Z"},"papermill":{"duration":0.03816,"end_time":"2022-07-07T13:06:31.667510","exception":false,"start_time":"2022-07-07T13:06:31.629350","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:03.219464Z","iopub.execute_input":"2022-07-08T02:57:03.220153Z","iopub.status.idle":"2022-07-08T02:57:03.228831Z","shell.execute_reply.started":"2022-07-08T02:57:03.220076Z","shell.execute_reply":"2022-07-08T02:57:03.228041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"category_ix_dict = {k: v for k, v in category_index_df[[\"category\", \"category_ix\"]].values}","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:36:08.438683Z","start_time":"2022-06-25T01:36:08.424308Z"},"papermill":{"duration":0.037779,"end_time":"2022-07-07T13:06:31.733128","exception":false,"start_time":"2022-07-07T13:06:31.695349","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:03.230513Z","iopub.execute_input":"2022-07-08T02:57:03.230848Z","iopub.status.idle":"2022-07-08T02:57:03.239348Z","shell.execute_reply.started":"2022-07-08T02:57:03.230812Z","shell.execute_reply":"2022-07-08T02:57:03.238601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(category_ix_dict)","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:36:08.973546Z","start_time":"2022-06-25T01:36:08.956036Z"},"papermill":{"duration":0.03685,"end_time":"2022-07-07T13:06:31.797342","exception":false,"start_time":"2022-07-07T13:06:31.760492","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:03.241631Z","iopub.execute_input":"2022-07-08T02:57:03.242227Z","iopub.status.idle":"2022-07-08T02:57:03.249226Z","shell.execute_reply.started":"2022-07-08T02:57:03.242187Z","shell.execute_reply":"2022-07-08T02:57:03.248231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_value_with_default(dict_, value, default=0):\n    return np.vectorize(lambda x: dict_.get(x, default))(value)\n\ndef get_category_ix(df, category_ix_dict):\n    cat_pairs = []\n    all_category = set()\n    for row in df[\"categories\"].drop_duplicates().dropna().values:\n        for cat in row.split(\", \"):\n            if len(cat):\n                cat_pairs.append([row, cat])\n                all_category.add(cat)\n\n    category_df = pd.DataFrame(cat_pairs, columns=[\"categories\", \"category\"])\n    categories_df = df[[\"ix\", \"categories\"]].merge(category_df, on=\"categories\", how=\"left\").sort_values(by=\"ix\")\n\n    categories_df[\"category_ix\"] = get_value_with_default(category_ix_dict, categories_df[\"category\"].values, -1)\n    categories_df = categories_df[categories_df[\"category_ix\"] >= 0]\n    categories_ix = tf.RaggedTensor.from_value_rowids(values=tf.constant(categories_df[\"category_ix\"].values, \"int32\"),\n                                                              value_rowids=tf.constant(categories_df[\"ix\"].values, \"int32\"),\n                                                              nrows=len(df))\n    return categories_ix\ncategory_ix_dict = {k: v for k, v in category_index_df[[\"category\", \"category_ix\"]].values}\ntest_categories_ix = get_category_ix(test_df, category_ix_dict)","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:36:09.978598Z","start_time":"2022-06-25T01:36:09.386281Z"},"papermill":{"duration":0.111832,"end_time":"2022-07-07T13:06:31.936214","exception":false,"start_time":"2022-07-07T13:06:31.824382","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:03.250848Z","iopub.execute_input":"2022-07-08T02:57:03.251232Z","iopub.status.idle":"2022-07-08T02:57:03.324385Z","shell.execute_reply.started":"2022-07-08T02:57:03.251197Z","shell.execute_reply":"2022-07-08T02:57:03.323607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsRegressor\nknn = KNeighborsRegressor(n_neighbors=1, \n                          metric='haversine', \n                          n_jobs=-1)\nknn.fit(np.deg2rad(geo_name_index_df[['latitude','longitude']]), geo_name_index_df[\"geo_name_index\"])\npseudo_index = knn.kneighbors(np.deg2rad(test_df[['latitude','longitude']]))[1].T[0]\ntest_df[\"pseudo_geo_name_ix\"] = pseudo_index","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:36:53.787862Z","start_time":"2022-06-25T01:36:10.262518Z"},"papermill":{"duration":0.252758,"end_time":"2022-07-07T13:06:32.218109","exception":false,"start_time":"2022-07-07T13:06:31.965351","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:03.325522Z","iopub.execute_input":"2022-07-08T02:57:03.325894Z","iopub.status.idle":"2022-07-08T02:57:03.544237Z","shell.execute_reply.started":"2022-07-08T02:57:03.325859Z","shell.execute_reply":"2022-07-08T02:57:03.543395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsRegressor\nknn = KNeighborsRegressor(n_neighbors=1, \n                          metric='haversine', \n                          n_jobs=-1)\nknn.fit(np.deg2rad(city_index_df[['latitude','longitude']]), city_index_df[\"city_index\"])\npseudo_index = knn.kneighbors(np.deg2rad(test_df[['latitude','longitude']]))[1].T[0]\ntest_df[\"pseudo_city\"] = np.where(test_df[\"city\"].isin(city_index_df[\"city\"].values), test_df[\"city\"].values, city_index_df[\"city\"][pseudo_index].values)\ntest_df[\"pseudo_city_ix\"] = test_df[\"pseudo_city\"].map(city_index_map)","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:37:10.357208Z","start_time":"2022-06-25T01:37:03.627325Z"},"papermill":{"duration":0.271423,"end_time":"2022-07-07T13:06:32.517972","exception":false,"start_time":"2022-07-07T13:06:32.246549","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:03.545581Z","iopub.execute_input":"2022-07-08T02:57:03.545931Z","iopub.status.idle":"2022-07-08T02:57:03.782321Z","shell.execute_reply.started":"2022-07-08T02:57:03.545895Z","shell.execute_reply":"2022-07-08T02:57:03.781524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsRegressor\nknn = KNeighborsRegressor(n_neighbors=1, \n                          metric='haversine', \n                          n_jobs=-1)\nknn.fit(np.deg2rad(state_index_df[['latitude','longitude']]), state_index_df[\"state_index\"])\npseudo_index = knn.kneighbors(np.deg2rad(test_df[['latitude','longitude']]))[1].T[0]\ntest_df[\"pseudo_state\"] = np.where(test_df[\"state\"].isin(state_index_df[\"state\"].values), test_df[\"state\"].values, state_index_df[\"state\"][pseudo_index].values)\ntest_df[\"pseudo_state_ix\"] = test_df[\"pseudo_state\"].map(state_index_map)","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:37:17.879463Z","start_time":"2022-06-25T01:37:10.358129Z"},"papermill":{"duration":0.371784,"end_time":"2022-07-07T13:06:32.918481","exception":false,"start_time":"2022-07-07T13:06:32.546697","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:03.783604Z","iopub.execute_input":"2022-07-08T02:57:03.783968Z","iopub.status.idle":"2022-07-08T02:57:04.016222Z","shell.execute_reply.started":"2022-07-08T02:57:03.783934Z","shell.execute_reply":"2022-07-08T02:57:04.015259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[\"country_ix\"] = get_value_with_default(country_index_map, test_df[\"country\"], np.where(country_index_df[\"country\"] == \"NAN\")[0][0])","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:37:17.988594Z","start_time":"2022-06-25T01:37:17.880463Z"},"papermill":{"duration":0.043331,"end_time":"2022-07-07T13:06:32.990656","exception":false,"start_time":"2022-07-07T13:06:32.947325","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:04.017612Z","iopub.execute_input":"2022-07-08T02:57:04.017963Z","iopub.status.idle":"2022-07-08T02:57:04.025731Z","shell.execute_reply.started":"2022-07-08T02:57:04.017926Z","shell.execute_reply":"2022-07-08T02:57:04.024978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"remove_cols = ['zip', 'name_language',\n       'address_language',\n       'pseudo_city', 'pseudo_state']\nfor col in remove_cols:\n    del test_df[col]\ngc.collect()","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:37:18.391073Z","start_time":"2022-06-25T01:37:17.989595Z"},"papermill":{"duration":1.207552,"end_time":"2022-07-07T13:06:34.229666","exception":false,"start_time":"2022-07-07T13:06:33.022114","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:04.027126Z","iopub.execute_input":"2022-07-08T02:57:04.027702Z","iopub.status.idle":"2022-07-08T02:57:05.100310Z","shell.execute_reply.started":"2022-07-08T02:57:04.027666Z","shell.execute_reply":"2022-07-08T02:57:05.099445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ix_params_dict = {\"country\": {\"dimention\": 64, \"num_category\": len(country_index_df)+1},\n                  \"state\": {\"dimention\": 64, \"num_category\": len(state_index_df)},\n                  \"city\": {\"dimention\": 64, \"num_category\": len(city_index_df)},\n                  \"geo_name\": {\"dimention\": 64, \"num_category\": len(geo_name_index_df)}\n                  }\ntest_ix_values = {\"country\": tf.convert_to_tensor(test_df[\"country_ix\"].values.astype(np.int32)),\n                   \"state\": tf.convert_to_tensor(test_df[\"pseudo_state_ix\"].values.astype(np.int32)),\n                   \"city\": tf.convert_to_tensor(test_df[\"pseudo_city_ix\"].values.astype(np.int32)),\n                   \"geo_name\": tf.convert_to_tensor(test_df[\"pseudo_geo_name_ix\"].values.astype(np.int32)),\n                   }\n\nclass IxEmbeddingLayer(tf.keras.layers.Layer):\n    def __init__(self, ix_params_dict, num_layer=1, out_dim=128):\n        super(IxEmbeddingLayer, self).__init__()\n        self.ix_params_dict = ix_params_dict\n        self.num_layer = num_layer\n        self.out_dim = out_dim\n        self.embedding_layers = {k: tf.keras.layers.Embedding(v[\"num_category\"], v[\"dimention\"]) for k, v in ix_params_dict.items()}\n        self.denses = [tf.keras.layers.Dense(out_dim, activation=\"gelu\")]\n        self.out = tf.keras.layers.Dense(out_dim)\n        self.drop_out = tf.keras.layers.Dropout(0.1)\n    \n    def call(self, ix_values):\n        results = []\n        for k, v in ix_values.items():\n            results.append(self.embedding_layers[k](v))\n        X = tf.concat(results, axis=-1)\n        \n        for i in range(self.num_layer):\n            X = self.drop_out(self.denses[i](X))\n        return self.out(X)\n        ","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:37:18.407168Z","start_time":"2022-06-25T01:37:18.392076Z"},"papermill":{"duration":0.047163,"end_time":"2022-07-07T13:06:34.305403","exception":false,"start_time":"2022-07-07T13:06:34.258240","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:05.105341Z","iopub.execute_input":"2022-07-08T02:57:05.105716Z","iopub.status.idle":"2022-07-08T02:57:05.119117Z","shell.execute_reply.started":"2022-07-08T02:57:05.105682Z","shell.execute_reply":"2022-07-08T02:57:05.118285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MixerModel(tf.keras.Model):\n    \n    def __init__(self, name_model, address_model, ix_emb_layer, cat_emb_layer, num_layer=2, num_dim=128):\n        super(MixerModel, self).__init__()\n        self.num_layer = num_layer\n        self.num_dim = num_dim\n        self.name_model = name_model\n        self.address_model = address_model\n        self.ix_emb_layer = ix_emb_layer\n        self.cat_emb_layer = cat_emb_layer\n        \n        self.tokenize_norm = [tf.keras.layers.LayerNormalization() for i in range(4)]\n    \n        self.drop_out = tf.keras.layers.Dropout(0.1)\n        \n        self.filter_denses = [tf.keras.layers.DepthwiseConv2D(kernel_size=(1, 4),\n                                                              strides=1,\n                                                              padding=\"valid\",\n                                                              depth_multiplier=4,\n                                                              activation=\"gelu\") for _ in range(self.num_layer)]\n        self.out_denses = [tf.keras.layers.Dense(num_dim) for _ in range(self.num_layer)]\n        \n        #self.pointwise_filter_denses = [tf.keras.layers.Dense(num_dim, activation=\"gelu\") for _ in range(self.num_layer)]\n        #self.pointwise_out_denses = [tf.keras.layers.Dense(num_dim) for _ in range(self.num_layer)]\n        self.norm_layers = [tf.keras.layers.LayerNormalization() for _ in range(self.num_layer)]\n        \n    def check_tensor(self, x, ix):\n        if isinstance(x, tf.RaggedTensor):\n            x = x.to_tensor()\n        return x\n        \n    def call(self, info, training=False):\n        ix, name, address, position, ix_values, categories, pids, size = info\n        X_name = self.tokenize_norm[0](tf.reduce_sum(self.name_model(name), axis=-2))\n        X_address = self.tokenize_norm[1](tf.reduce_sum(self.address_model(address), axis=-2))\n        X_ix = self.tokenize_norm[2](self.drop_out(self.ix_emb_layer(ix_values)))\n        X_categories = self.tokenize_norm[3](tf.reduce_sum(self.drop_out(self.cat_emb_layer(categories)), axis=-2))\n        \n        X = tf.stack([X_name, X_address, X_ix, X_categories], axis=-2)\n        original_shape = tf.shape(X)\n        X = tf.expand_dims(X, axis=1)\n        expand_shape = tf.shape(X)\n        for i in range(self.num_layer):\n            X_ = self.norm_layers[i](X)\n            X_ = self.out_denses[i](tf.reshape(self.filter_denses[i](X_), expand_shape))\n            X = X + X_\n        X = tf.reshape(X, original_shape)\n        return tf.reduce_mean(X, axis=-2)","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:37:18.438149Z","start_time":"2022-06-25T01:37:18.423609Z"},"papermill":{"duration":0.052669,"end_time":"2022-07-07T13:06:34.464428","exception":false,"start_time":"2022-07-07T13:06:34.411759","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:05.120897Z","iopub.execute_input":"2022-07-08T02:57:05.121316Z","iopub.status.idle":"2022-07-08T02:57:05.139128Z","shell.execute_reply.started":"2022-07-08T02:57:05.121280Z","shell.execute_reply":"2022-07-08T02:57:05.138316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CategoryEmbeddingLayer(tf.keras.layers.Layer):\n    \n    def __init__(self, num_categories, dim_categories, init_embedding=None):\n        super(CategoryEmbeddingLayer, self).__init__()\n        self.num_categories = num_categories\n        self.dim_categories = dim_categories\n        self.category_dense = tf.Variable(tf.keras.initializers.GlorotUniform()(shape=(num_categories, dim_categories)), trainable=True, name=self.name + \"/category_embedding\")\n        if not init_embedding is None:\n            self.category_dense.assign(init_embedding)\n            \n    def call(self, C):\n        X = tf.gather(self.category_dense, C)\n        return X","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:37:18.468757Z","start_time":"2022-06-25T01:37:18.454372Z"},"papermill":{"duration":0.041127,"end_time":"2022-07-07T13:06:34.537621","exception":false,"start_time":"2022-07-07T13:06:34.496494","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:05.141847Z","iopub.execute_input":"2022-07-08T02:57:05.142582Z","iopub.status.idle":"2022-07-08T02:57:05.156260Z","shell.execute_reply.started":"2022-07-08T02:57:05.142531Z","shell.execute_reply":"2022-07-08T02:57:05.155592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow_text import SentencepieceTokenizer\n\nclass SupervisedContrastiveLoss(tf.keras.layers.Layer):\n    def __init__(self, from_logits=False):\n        super(SupervisedContrastiveLoss, self).__init__()\n        if not from_logits:\n            NotImplementedError\n        \n    @tf.function\n    def call(self, true, pred, sample_weight=None):\n        if sample_weight is None:\n            sample_weight = tf.ones(tf.shape(pred), dtype=tf.float32)\n        epred = tf.exp(pred) * sample_weight\n        scale = tf.reduce_sum(epred, axis=1, keepdims=True)\n        loss = tf.reduce_sum((tf.math.log(scale - epred * true) - true*pred) * true * sample_weight) / tf.reduce_sum(true*sample_weight)\n        return loss\n    \nclass SentencePieceEmbeddingLayer(tf.keras.layers.Layer):\n    \n    def __init__(self, vocab_size, out_dim, sp_model_path, init_embedding=None):\n        super(SentencePieceEmbeddingLayer, self).__init__()\n        self.vocab_size = vocab_size\n        self.out_dim = out_dim\n        model = open(sp_model_path, \"rb\").read()\n        self.tokenizer = SentencepieceTokenizer(model)\n        self.embedding = tf.keras.layers.Embedding(vocab_size, out_dim)\n        if not init_embedding is None:\n            self.embedding(0)\n            self.embedding.trainable_variables[0].assign(init_embedding)\n        self.drop_out = tf.keras.layers.Dropout(0.1)\n    \n    def call(self, X):\n        token = self.tokenizer.tokenize(X)\n        X = self.drop_out(self.embedding(token))\n        return X\n    \n    \nclass ClassifyTrainModel(tf.keras.Model):\n    \n    def __init__(self, name_model, address_model, ix_emb_layer, cat_emb_layer):\n        super(ClassifyTrainModel, self).__init__()\n        self.name_model = name_model\n        self.address_model = address_model\n        self.ix_emb_layer = ix_emb_layer\n        self.cat_emb_layer = cat_emb_layer\n        self.dense = tf.keras.layers.Dense(cat_emb_layer.dim_categories, use_bias=True)\n        self.scale = tf.sqrt(2.) * tf.math.log(tf.cast(cat_emb_layer.num_categories, \"float32\") - 1.)\n        self.loss_function = SupervisedContrastiveLoss(from_logits=True)\n        \n        self.layer_norms = [tf.keras.layers.LayerNormalization() for i in range(3)]\n        self.drop_out = tf.keras.layers.Dropout(0.1)\n        self.nan_mask = tf.concat([tf.zeros([1, 1]), tf.ones([1, cat_emb_layer.num_categories-1])], axis=1)\n        \n    def transform(self, name, address, ix_values):\n        X_name = self.layer_norms[0](tf.reduce_sum(self.drop_out(self.name_model(name)), axis=1))\n        X_address = self.layer_norms[1](tf.reduce_sum(self.drop_out(self.address_model(address)), axis=1))\n        X_ix = self.layer_norms[2](self.ix_emb_layer(ix_values))\n        X = self.dense(tf.concat([X_name, X_address, X_ix], axis=-1))\n        return X\n        \n    def call(self, name, address, ix_values, categories):\n        X = self.transform(name, address, ix_values)\n        classify_loss = self.classify_task(X, categories)\n        return classify_loss\n    \n    def classify_task(self, X, C):\n        cossim = tf.einsum(\"ND,CD->NC\", tf.nn.l2_normalize(X, axis=1), tf.nn.l2_normalize(self.cat_emb_layer.category_dense, axis=1))\n        \n        true = tf.scatter_nd(tf.stack([C.value_rowids(), C.values], axis=1),\n                              tf.ones(tf.shape(C.values)),\n                              [tf.shape(X)[0], self.cat_emb_layer.num_categories])\n        true = true * self.nan_mask\n        calc_loc = tf.where(tf.reduce_sum(true, axis=1) > 0.)\n        pred = self.scale * cossim\n        loss = self.loss_function(tf.gather_nd(true, calc_loc), tf.gather_nd(pred, calc_loc))\n        return loss","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:37:18.499935Z","start_time":"2022-06-25T01:37:18.46976Z"},"papermill":{"duration":0.083148,"end_time":"2022-07-07T13:06:34.648392","exception":false,"start_time":"2022-07-07T13:06:34.565244","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:05.157753Z","iopub.execute_input":"2022-07-08T02:57:05.158845Z","iopub.status.idle":"2022-07-08T02:57:05.188004Z","shell.execute_reply.started":"2022-07-08T02:57:05.158811Z","shell.execute_reply":"2022-07-08T02:57:05.187209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def haversine(X, Y):\n    delta = Y - X\n    x_lats = tf.gather(X, 0, axis=-1)\n    y_lats = tf.gather(Y, 0, axis=-1)\n    dlat = tf.gather(delta, 0, axis=-1)\n    dlon = tf.gather(delta, 1, axis=-1)\n\n    a = tf.sin(dlat/2) * tf.sin(dlat/2) + tf.cos(x_lats) * tf.cos(y_lats) * tf.sin(dlon/2) * tf.sin(dlon/2)\n    c = 2 * tf.math.atan2(tf.sqrt(a), tf.sqrt(1-a))\n    return c\n\n\nclass TFMiniBatchHKmeans(tf.keras.Model):\n    def __init__(self, n_cluster=64, learning_rate=0.1):\n        super(TFMiniBatchHKmeans, self).__init__()\n        self.n_cluster = n_cluster\n        self.learning_rate = learning_rate\n        self.centroids = tf.Variable(tf.zeros((n_cluster, 2), dtype=\"float32\"), trainable=False)\n    \n    def init_centroid(self, X):\n        new_centroids = []\n        choosed = tf.zeros(len(X), dtype=\"bool\")\n\n        ix = tf.cast(tf.random.uniform([1], maxval=len(X))//1, \"int32\")[0]\n        choosed = tf.where(tf.range(len(X)) == ix, True, choosed)\n        new_centroids.append(tf.gather(X, ix))\n        dist = tf.ones([tf.shape(X)[0]]) * tf.float32.max / 10\n\n        for i in tqdm(range(self.n_cluster-1)):\n            centroid = tf.gather(X, ix)\n    \n            dist = tf.where(choosed, tf.keras.backend.epsilon() , tf.minimum(dist, tf.math.square(hkmeans.haversine(centroid, X))))\n            prob = dist / tf.reduce_sum(dist, axis=0, keepdims=True)\n            logit = tf.math.log(prob) - tf.math.log(1-prob)        \n            gumbel = logit - tf.math.log(-tf.math.log(tf.random.uniform(tf.shape(logit))))\n            ix = tf.cast(tf.argmax(gumbel), \"int32\")\n            choosed = tf.where(tf.range(len(X)) == ix, True, choosed)\n            new_centroids.append(tf.gather(X, ix))\n        self.centroids.assign(tf.stack(new_centroids, axis=0))\n        \n    def partial_fit(self, X):\n        score = self.haversine(tf.expand_dims(X, axis=1), tf.expand_dims(self.centroids, axis=0))\n        maxix = tf.argmin(score, axis=1)\n        new_centroids = tf.math.unsorted_segment_mean(X, maxix, num_segments=self.n_cluster)\n        num_member = tf.math.unsorted_segment_sum(tf.ones((len(X), 1)), maxix, num_segments=self.n_cluster)\n        new_centroids = tf.where(num_member > 0, new_centroids, self.centroids)\n        new_centroids = new_centroids * self.learning_rate + self.centroids * (1 - self.learning_rate)\n        dist = tf.reduce_mean(tf.reduce_min(score, axis=1))\n        self.centroids.assign(new_centroids)\n        return dist\n    \n    def haversine(self, X, Y):\n        delta = Y - X\n        x_lats = tf.gather(X, 0, axis=-1)\n        y_lats = tf.gather(Y, 0, axis=-1)\n        dlat = tf.gather(delta, 0, axis=-1)\n        dlon = tf.gather(delta, 1, axis=-1)\n\n        a = tf.sin(dlat/2) * tf.sin(dlat/2) + tf.cos(x_lats) * tf.cos(y_lats) * tf.sin(dlon/2) * tf.sin(dlon/2)\n        c = tf.math.atan2(tf.sqrt(a), tf.sqrt(1-a))\n        return c\n    \n    def transform(self, X, return_score=False):\n        score = tf.math.log(self.haversine(tf.expand_dims(X, axis=1), tf.expand_dims(self.centroids, axis=0)) + tf.keras.backend.epsilon())\n        res = tf.argmin(score, axis=-1)\n        if return_score:\n            dist = tf.reduce_min(score, axis=-1)\n            return res, dist\n        else:\n            return res\n\nclass TFMiniBatchSKmeans(tf.keras.Model):\n    def __init__(self, n_cluster=64, n_dim=100, alpha=3/2, learning_rate=0.1):\n        super(TFMiniBatchSKmeans, self).__init__()\n        self.n_dim = n_dim\n        self.n_cluster = n_cluster\n        self.alpha = alpha\n        self.learning_rate = learning_rate\n\n        self.centroids = tf.Variable(tf.zeros((n_cluster, n_dim), dtype=\"float32\"), trainable=False)\n    \n    def init_centroid(self, X):\n        new_centroids = []\n        X = tf.nn.l2_normalize(X, axis=-1)\n        choosed = tf.zeros(len(X), dtype=\"bool\")\n\n        ix = tf.cast(tf.random.uniform([1], maxval=len(X))//1, \"int32\")[0]\n        choosed = tf.where(tf.range(len(X)) == ix, True, choosed)\n        new_centroids.append(tf.gather(X, ix))\n        \n        dist = tf.ones([tf.shape(X)[0]]) * tf.float32.max / 10\n\n        for i in tqdm(range(self.n_cluster-1)):\n            centroid = tf.gather(X, ix)\n            cos = tf.einsum(\"d,nd->n\", centroid, X)\n            dist = tf.where(choosed, tf.keras.backend.epsilon() , tf.maximum(tf.keras.backend.epsilon(), tf.minimum(dist, 1. - cos)))         \n            prob = dist / tf.reduce_sum(dist, axis=0, keepdims=True)\n            logit = tf.math.log(prob) - tf.math.log(1-prob)        \n            gumbel = logit - tf.math.log(-tf.math.log(tf.random.uniform(tf.shape(logit))))\n            ix = tf.cast(tf.argmax(gumbel), \"int32\")\n            choosed = tf.where(tf.range(len(X)) == ix, True, choosed)\n            new_centroids.append(X[ix])\n        self.centroids.assign(tf.stack(new_centroids, axis=0))\n        \n    @tf.function\n    def partial_fit(self, X):\n        score = tf.einsum(\"nd,kd->nk\", X, self.centroids)\n        maxix = tf.argmax(score, axis=1)\n        new_centroids = tf.math.unsorted_segment_mean(X, maxix, num_segments=self.n_cluster)\n        num_member = tf.math.unsorted_segment_sum(tf.ones((len(X), 1)), maxix, num_segments=self.n_cluster)\n        new_centroids = tf.where(num_member > 0, new_centroids, self.centroids)\n        new_centroids = tf.nn.l2_normalize(new_centroids, axis=1)\n        new_centroids = new_centroids * self.learning_rate + self.centroids * (1 - self.learning_rate)\n        new_centroids = tf.nn.l2_normalize(new_centroids, axis=1)\n        dist = tf.reduce_mean(self.alpha - tf.reduce_max(score, axis=1))\n        self.centroids.assign(new_centroids)\n        return dist\n    \n    def transform(self, X, return_score=False):\n        score = tf.einsum(\"...d,kd->...k\", X, self.centroids)\n        res = tf.argmax(score, axis=-1)\n        if return_score:\n            cossim = tf.reduce_max(score, axis=-1)\n            return res, cossim\n        else:\n            return res","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:37:18.515367Z","start_time":"2022-06-25T01:37:18.500936Z"},"papermill":{"duration":0.073409,"end_time":"2022-07-07T13:06:34.750671","exception":false,"start_time":"2022-07-07T13:06:34.677262","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:05.189433Z","iopub.execute_input":"2022-07-08T02:57:05.190604Z","iopub.status.idle":"2022-07-08T02:57:05.230880Z","shell.execute_reply.started":"2022-07-08T02:57:05.190546Z","shell.execute_reply":"2022-07-08T02:57:05.230105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataContainer():\n    def __init__(self, df, ix_values, category_ix, positive_ix, neighbor_ix, neighbor_dist):\n        self.name = tf.constant(df[\"sp_name\"].values)\n        self.address = tf.constant(df[\"sp_address\"].values)\n        self.position = tf.expand_dims(tf.constant(np.deg2rad(df[[\"latitude\", \"longitude\"]].astype(np.float32).values), dtype=\"float32\"), axis=0)\n        self.pid = tf.constant(df[\"pid\"].astype(np.int32).values, dtype=\"int32\")\n        self.group = tf.constant(df[\"group\"].astype(np.int32).values, dtype=\"int32\")\n        self.ix_values = ix_values\n        self.category_ix = category_ix\n        self.positive_ix = positive_ix\n        self.neighbor_ix = neighbor_ix\n        self.neighbor_dist = neighbor_dist\n        \n    def get_position(self, ix):\n        return tf.gather(tf.gather(self.position, ix, axis=1), 0)\n\n    def call(self, ix, size=1):\n        name = tf.gather(self.name, ix)\n        address = tf.gather(self.address, ix)\n        position = self.get_position(ix)\n        categories = tf.gather(self.category_ix, ix)\n        ix_values = {k: tf.gather(self.ix_values[k], ix) for k in self.ix_values.keys()}\n        pids = tf.gather(self.pid, ix)\n        return ix, name, address, position, ix_values, categories, pids, size\n    \n    @tf.function(experimental_relax_shapes=True)\n    def log_haversine(self, X, Y):\n        delta = Y - X\n        x_lats = tf.gather(X, 0, axis=-1)\n        y_lats = tf.gather(Y, 0, axis=-1)\n        dlat = tf.gather(delta, 0, axis=-1)\n        dlon = tf.gather(delta, 1, axis=-1)\n\n        a = tf.sin(dlat/2) * tf.sin(dlat/2) + tf.cos(x_lats) * tf.cos(y_lats) * tf.sin(dlon/2) * tf.sin(dlon/2)\n        c = tf.math.log(2 * tf.math.atan2(tf.sqrt(a), tf.sqrt(1-a)) + tf.keras.backend.epsilon())\n        return c","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:37:18.531256Z","start_time":"2022-06-25T01:37:18.516226Z"},"papermill":{"duration":0.06406,"end_time":"2022-07-07T13:06:34.844147","exception":false,"start_time":"2022-07-07T13:06:34.780087","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:05.232404Z","iopub.execute_input":"2022-07-08T02:57:05.232812Z","iopub.status.idle":"2022-07-08T02:57:05.248500Z","shell.execute_reply.started":"2022-07-08T02:57:05.232777Z","shell.execute_reply":"2022-07-08T02:57:05.247633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DIMSIZE = 256\nWALK_STEP = 128\nWALK_TOPK = 2\nnum_categories = max(category_ix_dict.values()) + 1\n\nname_spe = SentencePieceEmbeddingLayer(32000, DIMSIZE, nadare_feature_dir + \"sp_name.model\", None)\naddress_spe = SentencePieceEmbeddingLayer(32000, DIMSIZE, nadare_feature_dir + \"sp_address.model\", None)\n\nix_emb_layer = IxEmbeddingLayer(ix_params_dict, num_layer=1, out_dim=DIMSIZE)\ncat_emb_layer = CategoryEmbeddingLayer(num_categories, DIMSIZE)\n\nclassify_model = ClassifyTrainModel(name_spe, address_spe, ix_emb_layer, cat_emb_layer)\nmixer_model = MixerModel(name_spe, address_spe, ix_emb_layer, cat_emb_layer, num_layer=3, num_dim=DIMSIZE)\nmixer_loss_func = tf.keras.losses.CategoricalCrossentropy(from_logits=True)\n\n\nname_spe.embedding(0)\naddress_spe.embedding(0)\nfor k, v in ix_emb_layer.embedding_layers.items():\n    v(0)","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:37:18.639622Z","start_time":"2022-06-25T01:37:18.547602Z"},"papermill":{"duration":0.290803,"end_time":"2022-07-07T13:06:35.163309","exception":false,"start_time":"2022-07-07T13:06:34.872506","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:05.249937Z","iopub.execute_input":"2022-07-08T02:57:05.250508Z","iopub.status.idle":"2022-07-08T02:57:05.487778Z","shell.execute_reply.started":"2022-07-08T02:57:05.250471Z","shell.execute_reply":"2022-07-08T02:57:05.486949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"container_cols = [\"sp_name\", \"sp_address\", \"latitude\", \"longitude\", \"pid\", \"group\"] \ntest_container =  DataContainer(test_df[container_cols],\n                                   test_ix_values,\n                                    test_categories_ix,\n                                    None,\n                                None,\n                                None)\ntest_ixs = tf.range(len(test_container.name), dtype=\"int32\")","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:37:19.124135Z","start_time":"2022-06-25T01:37:18.735015Z"},"papermill":{"duration":0.05379,"end_time":"2022-07-07T13:06:35.245634","exception":false,"start_time":"2022-07-07T13:06:35.191844","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:05.488914Z","iopub.execute_input":"2022-07-08T02:57:05.490918Z","iopub.status.idle":"2022-07-08T02:57:05.511225Z","shell.execute_reply.started":"2022-07-08T02:57:05.490886Z","shell.execute_reply":"2022-07-08T02:57:05.510479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_info = test_container.call([0])\nix, name, address, position, ix_values, categories, pid, size = tmp_info\nclassify_model.transform(name, address, ix_values)\n_ = mixer_model(tmp_info)","metadata":{"papermill":{"duration":2.098488,"end_time":"2022-07-07T13:06:37.372342","exception":false,"start_time":"2022-07-07T13:06:35.273854","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:05.512416Z","iopub.execute_input":"2022-07-08T02:57:05.512815Z","iopub.status.idle":"2022-07-08T02:57:07.631013Z","shell.execute_reply.started":"2022-07-08T02:57:05.512780Z","shell.execute_reply":"2022-07-08T02:57:07.630129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classify_model.load_weights(nadare_feature_dir + \"classify_model\")\nmixer_model.load_weights(nadare_feature_dir + \"mixer_model\")","metadata":{"papermill":{"duration":2.124711,"end_time":"2022-07-07T13:06:39.543821","exception":false,"start_time":"2022-07-07T13:06:37.419110","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:07.632202Z","iopub.execute_input":"2022-07-08T02:57:07.632821Z","iopub.status.idle":"2022-07-08T02:57:09.608081Z","shell.execute_reply.started":"2022-07-08T02:57:07.632784Z","shell.execute_reply":"2022-07-08T02:57:09.607262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"normalized_category_dense = tf.nn.l2_normalize(cat_emb_layer.category_dense, axis=1)\npseudo_categories = []\ncategory_ix_dict_r = {v: k for k, v in category_ix_dict.items()}\neval_ds = tf.data.Dataset.from_tensor_slices(test_ixs)\\\n                .batch(128)\\\n                .map(test_container.call)\n\nfor ix, name, address, position, ix_values, categories, pid, size in tqdm(eval_ds):\n    X = tf.nn.l2_normalize(classify_model.transform(name, address, ix_values), axis=1)\n    label = tf.argmax(tf.einsum(\"nd,md->nm\", X, normalized_category_dense), axis=1)\n    for l in label.numpy():\n        pseudo_categories.append(\"nan, \" + category_ix_dict_r[l])\n        \ndef remove_test_only_category(text):\n    res = []\n    for x in text.split(\", \"):\n        if x in category_ix_dict.keys():\n            res.append(x)\n    return \", \".join(res)\n\ntest_df[\"categories\"] = np.vectorize(remove_test_only_category)(test_df[\"categories\"])\ntest_df[\"categories\"] = np.where(test_df[\"categories\"] == \"\", pseudo_categories, test_df[\"categories\"])\ntest_categories_ix = get_category_ix(test_df, category_ix_dict)","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:39:06.452961Z","start_time":"2022-06-25T01:37:19.125135Z"},"papermill":{"duration":2.891809,"end_time":"2022-07-07T13:06:42.464311","exception":false,"start_time":"2022-07-07T13:06:39.572502","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:09.611880Z","iopub.execute_input":"2022-07-08T02:57:09.612586Z","iopub.status.idle":"2022-07-08T02:57:13.309187Z","shell.execute_reply.started":"2022-07-08T02:57:09.612527Z","shell.execute_reply":"2022-07-08T02:57:13.308292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"container_cols = [\"sp_name\", \"sp_address\", \"latitude\", \"longitude\", \"pid\", \"group\"] \ntest_container =  DataContainer(test_df[container_cols],\n                                   test_ix_values,\n                                    test_categories_ix,\n                                    None,\n                                None,\n                                None)","metadata":{"papermill":{"duration":0.04289,"end_time":"2022-07-07T13:06:42.536524","exception":false,"start_time":"2022-07-07T13:06:42.493634","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:13.313630Z","iopub.execute_input":"2022-07-08T02:57:13.314384Z","iopub.status.idle":"2022-07-08T02:57:13.328772Z","shell.execute_reply.started":"2022-07-08T02:57:13.314345Z","shell.execute_reply":"2022-07-08T02:57:13.327889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hkmeans = TFMiniBatchHKmeans(255*8, learning_rate=1.)\nname_kmeans = TFMiniBatchSKmeans(255*8, n_dim=DIMSIZE, learning_rate=1.)\naddress_kmeans = TFMiniBatchSKmeans(255*8, n_dim=DIMSIZE, learning_rate=1.)\ncategory_kmeans = TFMiniBatchSKmeans(255*8, n_dim=DIMSIZE, learning_rate=1.)\nix_values_kmeans = TFMiniBatchSKmeans(255*8, n_dim=DIMSIZE, learning_rate=1.)\nmix_kmeans = TFMiniBatchSKmeans(255*8, n_dim=DIMSIZE, learning_rate=1.)\nbrand_kmeans = TFMiniBatchSKmeans(255*8, n_dim=DIMSIZE, learning_rate=1.)\n\nhkmeans.load_weights(nadare_feature_dir + \"haversine_kmeans\")\nname_kmeans.load_weights(nadare_feature_dir + \"name_kmeans\")\naddress_kmeans.load_weights(nadare_feature_dir + \"address_kmeans\")\nix_values_kmeans.load_weights(nadare_feature_dir + \"ix_value_kmeans\")\nmix_kmeans.load_weights(nadare_feature_dir + \"mix_kmeans\")\ncategory_kmeans.load_weights(nadare_feature_dir + \"category_kmeans\")\nbrand_kmeans.load_weights(nadare_feature_dir + \"brand_kmeans\")","metadata":{"papermill":{"duration":0.556297,"end_time":"2022-07-07T13:06:43.121548","exception":false,"start_time":"2022-07-07T13:06:42.565251","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:13.330254Z","iopub.execute_input":"2022-07-08T02:57:13.330857Z","iopub.status.idle":"2022-07-08T02:57:13.575221Z","shell.execute_reply.started":"2022-07-08T02:57:13.330818Z","shell.execute_reply":"2022-07-08T02:57:13.574496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# name\ntest_ds = tf.data.Dataset.from_tensor_slices(tf.range(len(test_df)))\\\n                .batch(1024)\\\n                .map(test_container.call)\n\nname_embeddings = []\naddress_embeddings = []\nix_values_embeddings = []\nmix_embeddings = []\nfor query_info in tqdm(test_ds):\n    ix, name, address, position, ix_values, categories, pid, size = query_info\n    name_embeddings.append(tf.nn.l2_normalize(tf.reduce_sum(name_spe(name), axis=1), axis=1))\n    address_embeddings.append(tf.nn.l2_normalize(tf.reduce_sum(address_spe(address), axis=1), axis=1))\n    ix_values_embeddings.append(tf.nn.l2_normalize(ix_emb_layer(ix_values), axis=1))\n    mix_embeddings.append(tf.nn.l2_normalize(mixer_model(query_info), axis=1))\n    \n\nname_embeddings = tf.nn.l2_normalize(tf.concat(name_embeddings, axis=0), axis=1)\naddress_embeddings = tf.nn.l2_normalize(tf.concat(address_embeddings, axis=0), axis=1)\nix_values_embeddings = tf.nn.l2_normalize(tf.concat(ix_values_embeddings, axis=0), axis=1)\nmix_embeddings = tf.nn.l2_normalize(tf.concat(mix_embeddings, axis=0), axis=1)\n\ncategory_embeddings = tf.nn.l2_normalize(tf.math.reduce_sum(tf.gather(cat_emb_layer.category_dense, test_categories_ix), axis=1), axis=1)\n","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:39:43.692619Z","start_time":"2022-06-25T01:39:10.954045Z"},"papermill":{"duration":2.779754,"end_time":"2022-07-07T13:08:39.129804","exception":false,"start_time":"2022-07-07T13:08:36.350050","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:57:13.576515Z","iopub.execute_input":"2022-07-08T02:57:13.576925Z","iopub.status.idle":"2022-07-08T02:57:15.077596Z","shell.execute_reply.started":"2022-07-08T02:57:13.576895Z","shell.execute_reply":"2022-07-08T02:57:15.076816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 1024\neval_ds = tf.data.Dataset.from_tensor_slices(test_ixs)\\\n                .batch(10000)\nhkmeans_labels = []\nname_kmeans_labels = []\naddress_kmeans_labels = []\nix_values_kmeans_labels = []\nmix_kmeans_labels = []\ncategory_kmeans_labels = []\nbrand_kmeans_labels = []\nbrand_pid_kmeans_labels = []\n\n\nhkmeans_scores = []\nname_kmeans_scores = []\naddress_kmeans_scores = []\nix_values_kmeans_scores = []\nmix_kmeans_scores = []\ncategory_kmeans_scores = []\nbrand_kmeans_scores = []\nbrand_pid_kmeans_scores = []\n\n\nfor ix in tqdm(eval_ds):\n    res = hkmeans.transform(tf.gather(test_container.position[0], ix), return_score=True)\n    hkmeans_labels.append(res[0])\n    hkmeans_scores.append(res[1])\n    \n    res = name_kmeans.transform(tf.gather(name_embeddings, ix), return_score=True)\n    name_kmeans_labels.append(res[0])\n    name_kmeans_scores.append(res[1])\n\n    res = address_kmeans.transform(tf.gather(address_embeddings, ix), return_score=True)\n    address_kmeans_labels.append(res[0])\n    address_kmeans_scores.append(res[1])\n\n    res = ix_values_kmeans.transform(tf.gather(ix_values_embeddings, ix), return_score=True)\n    ix_values_kmeans_labels.append(res[0])\n    ix_values_kmeans_scores.append(res[1])\n\n    res = mix_kmeans.transform(tf.gather(mix_embeddings, ix), return_score=True)\n    mix_kmeans_labels.append(res[0])\n    mix_kmeans_scores.append(res[1])\n    \n    res = category_kmeans.transform(tf.gather(category_embeddings, ix), return_score=True)\n    category_kmeans_labels.append(res[0])\n    category_kmeans_scores.append(res[1])\n    \n    res = brand_kmeans.transform(tf.gather(name_embeddings, ix), return_score=True)\n    brand_kmeans_labels.append(res[0])\n    brand_kmeans_scores.append(res[1]) \n\n    \ntest_df[\"hkmeans_labels\"] = tf.concat(hkmeans_labels, axis=0).numpy().astype(np.int16)\ntest_df[\"name_kmeans_labels\"] = tf.concat(name_kmeans_labels, axis=0).numpy().astype(np.int16)\ntest_df[\"address_kmeans_labels\"] = tf.concat(address_kmeans_labels, axis=0).numpy().astype(np.int16)\ntest_df[\"ix_values_kmeans_labels\"] = tf.concat(ix_values_kmeans_labels, axis=0).numpy().astype(np.int16)\ntest_df[\"mix_kmeans_labels\"] = tf.concat(mix_kmeans_labels, axis=0).numpy().astype(np.int16)\ntest_df[\"category_kmeans_labels\"] = tf.concat(category_kmeans_labels, axis=0).numpy().astype(np.int16)\ntest_df[\"brand_kmeans_labels\"] = tf.concat(brand_kmeans_labels, axis=0).numpy().astype(np.int16)\n\ntest_df[\"hkmeans_scores\"] = tf.concat(hkmeans_scores, axis=0).numpy().astype(np.float32)\ntest_df[\"name_kmeans_scores\"] = tf.concat(name_kmeans_scores, axis=0).numpy().astype(np.float32)\ntest_df[\"address_kmeans_scores\"] = tf.concat(address_kmeans_scores, axis=0).numpy().astype(np.float32)\ntest_df[\"ix_values_kmeans_scores\"] = tf.concat(ix_values_kmeans_scores, axis=0).numpy().astype(np.float32)\ntest_df[\"mix_kmeans_scores\"] = tf.concat(mix_kmeans_scores, axis=0).numpy().astype(np.float32)\ntest_df[\"category_kmeans_scores\"] = tf.concat(category_kmeans_scores, axis=0).numpy().astype(np.float32)\ntest_df[\"brand_kmeans_scores\"] = tf.concat(brand_kmeans_scores, axis=0).numpy().astype(np.float32)\n\n\nhkmeans_labels = []\nname_kmeans_labels = []\naddress_kmeans_labels = []\nix_values_kmeans_labels = []\nmix_kmeans_labels = []\ncategory_kmeans_labels = []\nbrand_kmeans_labels = []\nbrand_pid_kmeans_labels = []\n\n\nhkmeans_scores = []\nname_kmeans_scores = []\naddress_kmeans_scores = []\nix_values_kmeans_scores = []\nmix_kmeans_scores = []\ncategory_kmeans_scores = []\nbrand_kmeans_scores = []\nbrand_pid_kmeans_scores = []\n\nimport gc\ngc.collect()","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:39:47.503932Z","start_time":"2022-06-25T01:39:46.619447Z"},"papermill":{"duration":1.473259,"end_time":"2022-07-07T13:08:40.634284","exception":false,"start_time":"2022-07-07T13:08:39.161025","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:58:52.904863Z","iopub.execute_input":"2022-07-08T02:58:52.905413Z","iopub.status.idle":"2022-07-08T02:58:54.233065Z","shell.execute_reply.started":"2022-07-08T02:58:52.905378Z","shell.execute_reply":"2022-07-08T02:58:54.232218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# simple version\n\nimport tensorflow_ranking as tfr\n\nclass LogisticModel(tf.keras.Model):\n    def __init__(self, dim):\n        super(LogisticModel, self).__init__()\n        self.scale = tf.Variable(tf.zeros([dim], dtype=\"float32\"), trainable=True)\n        self.loss_func = tfr.keras.losses.ApproxNDCGLoss()\n        \n    def call(self, X, Y):\n        logit = tf.einsum(\"nmd,d->nm\", X, self.scale)# + self.bias\n        loss = self.loss_func(Y, logit)\n        return loss    \n    \n","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:40:52.23617Z","start_time":"2022-06-25T01:40:52.212171Z"},"papermill":{"duration":0.07546,"end_time":"2022-07-07T13:08:40.809466","exception":false,"start_time":"2022-07-07T13:08:40.734006","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:58:55.726473Z","iopub.execute_input":"2022-07-08T02:58:55.727052Z","iopub.status.idle":"2022-07-08T02:58:55.780305Z","shell.execute_reply.started":"2022-07-08T02:58:55.727011Z","shell.execute_reply":"2022-07-08T02:58:55.779551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"simple_logistic_model = LogisticModel(5)\nsimple_logistic_model.load_weights(nadare_feature_dir + \"logistic_model_full\")","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:41:47.319437Z","start_time":"2022-06-25T01:41:47.300505Z"},"papermill":{"duration":0.063962,"end_time":"2022-07-07T13:08:40.903276","exception":false,"start_time":"2022-07-07T13:08:40.839314","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:58:57.402877Z","iopub.execute_input":"2022-07-08T02:58:57.403237Z","iopub.status.idle":"2022-07-08T02:58:57.430269Z","shell.execute_reply.started":"2022-07-08T02:58:57.403206Z","shell.execute_reply":"2022-07-08T02:58:57.429459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\ndef count_unique_bigram(x):\n    split = x.split()\n    vocab = set()\n    for i in range(max(len(split)-1, 0)):\n        vocab.add(\"|\".join(split[i:i+2]))\n    return len(vocab)\n\nsp_name_unique = np.vectorize(lambda x: len(set(x.split())))(test_df[\"sp_name\"].str.lower())\nsp_name_word_vectorizer = TfidfVectorizer(min_df=2, binary=True, use_idf=False, norm=None, dtype=np.float32, ngram_range=(1, 1), token_pattern=r'[a-zA-Z0-9&]+')\nsp_name_word_bow = sp_name_word_vectorizer.fit_transform(test_df[\"sp_name\"]).tocoo()\nsp_name_cnt = tf.convert_to_tensor(sp_name_unique.astype(np.float32))\nsp_name_r_norm = 1 / (tf.math.sqrt(sp_name_cnt) + tf.keras.backend.epsilon())\nsp_name_word_bow_rag_data = tf.RaggedTensor.from_value_rowids(sp_name_word_bow.data, sp_name_word_bow.row, nrows=sp_name_word_bow.shape[0])\nsp_name_word_bow_rag_col = tf.RaggedTensor.from_value_rowids(sp_name_word_bow.col, sp_name_word_bow.row, nrows=sp_name_word_bow.shape[0])\nsp_name_word_bow_coo = tf.SparseTensor(tf.cast(tf.stack([sp_name_word_bow.row, sp_name_word_bow.col], axis=-1), \"int64\"), sp_name_word_bow.data, sp_name_word_bow.shape)\ndel sp_name_word_bow, sp_name_word_vectorizer\n\nn_name_split_unique = np.vectorize(lambda x: len(set(x.split())))(test_df[\"name_normalized\"].str.lower())\nn_name_word_vectorizer = TfidfVectorizer(min_df=2, binary=True, use_idf=False, norm=None, dtype=np.float32, ngram_range=(1, 1), token_pattern=r'[a-zA-Z0-9&\\-]+')\nn_name_word_bow = n_name_word_vectorizer.fit_transform(test_df[\"name_normalized\"]).tocoo()\nn_name_word_bow.data = np.where(n_name_split_unique[n_name_word_bow.row] > 0, np.sqrt(1/n_name_split_unique[n_name_word_bow.row]), 0.).astype(np.float32)\nn_name_word_bow_rag_data = tf.RaggedTensor.from_value_rowids(n_name_word_bow.data, n_name_word_bow.row, nrows=n_name_word_bow.shape[0])\nn_name_word_bow_rag_col = tf.RaggedTensor.from_value_rowids(n_name_word_bow.col, n_name_word_bow.row, nrows=n_name_word_bow.shape[0])\nn_name_word_bow_coo = tf.SparseTensor(tf.cast(tf.stack([n_name_word_bow.row, n_name_word_bow.col], axis=-1), \"int64\"), n_name_word_bow.data, n_name_word_bow.shape)\ndel n_name_word_bow, n_name_word_vectorizer\n\nname_bichar_vectorizer = TfidfVectorizer(min_df=1, binary=True, use_idf=False, norm=None, dtype=np.float32, ngram_range=(2, 2), lowercase=False, analyzer=\"char_wb\")\nname_bichar_bow = name_bichar_vectorizer.fit_transform(test_df[\"name\"].fillna(\"\"))\nname_bichar_cnt = tf.convert_to_tensor(np.array(name_bichar_bow.sum(axis=1)).T[0]) + tf.keras.backend.epsilon()\nname_bichar_r_norm = tf.convert_to_tensor(1 / (np.sqrt(np.array(name_bichar_bow.power(2).sum(axis=1))).T[0] + tf.keras.backend.epsilon()))\nname_bichar_bow = name_bichar_bow.tocoo()\nname_bichar_bow_rag_data = tf.RaggedTensor.from_value_rowids(name_bichar_bow.data, name_bichar_bow.row, nrows=name_bichar_bow.shape[0])\nname_bichar_bow_rag_col = tf.RaggedTensor.from_value_rowids(name_bichar_bow.col, name_bichar_bow.row, nrows=name_bichar_bow.shape[0])\nname_bichar_bow_coo = tf.SparseTensor(tf.cast(tf.stack([name_bichar_bow.row, name_bichar_bow.col], axis=-1), \"int64\"), name_bichar_bow.data, name_bichar_bow.shape)\ndel name_bichar_bow, name_bichar_vectorizer","metadata":{"ExecuteTime":{"end_time":"2022-06-25T01:44:05.957055Z","start_time":"2022-06-25T01:44:02.874451Z"},"papermill":{"duration":0.545307,"end_time":"2022-07-07T13:08:41.477926","exception":false,"start_time":"2022-07-07T13:08:40.932619","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:58:59.704246Z","iopub.execute_input":"2022-07-08T02:58:59.704668Z","iopub.status.idle":"2022-07-08T02:59:00.179587Z","shell.execute_reply.started":"2022-07-08T02:58:59.704636Z","shell.execute_reply":"2022-07-08T02:59:00.178777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"monogram_word_vectorizer = TfidfVectorizer(min_df=1, binary=True, use_idf=False, norm=None, dtype=np.float32, ngram_range=(1, 1), token_pattern=r'[a-zA-Z0-9&]+')\nmono_sp_name_bow = monogram_word_vectorizer.fit_transform(test_df[\"sp_name\"]).tocoo()\nmono_sp_name_bow_rag = tf.RaggedTensor.from_value_rowids(mono_sp_name_bow.col, mono_sp_name_bow.row, nrows=mono_sp_name_bow.shape[0])\ndel mono_sp_name_bow, monogram_word_vectorizer\n\nbigram_word_vectorizer = TfidfVectorizer(min_df=1, binary=True, use_idf=False, norm=None, dtype=np.float32, ngram_range=(2, 2), token_pattern=r'[a-zA-Z0-9&]+')\nbi_sp_name_bow = bigram_word_vectorizer.fit_transform(test_df[\"sp_name\"]).tocoo()\nbi_sp_name_bow_rag = tf.RaggedTensor.from_value_rowids(bi_sp_name_bow.col, bi_sp_name_bow.row, nrows=bi_sp_name_bow.shape[0])\ndel bi_sp_name_bow, bigram_word_vectorizer\n\nmonogram_char_vectorizer = TfidfVectorizer(min_df=1, binary=True, use_idf=False, norm=None, dtype=np.float32, ngram_range=(1, 1), lowercase=False, analyzer=\"char_wb\")\nmono_name_boc = monogram_char_vectorizer.fit_transform(test_df[\"name\"].fillna(\"\")).tocoo()\nmono_name_boc_rag = tf.RaggedTensor.from_value_rowids(mono_name_boc.col, mono_name_boc.row, nrows=mono_name_boc.shape[0])\ndel mono_name_boc, monogram_char_vectorizer\n\nbigram_char_vectorizer = TfidfVectorizer(min_df=1, binary=True, use_idf=False, norm=None, dtype=np.float32, ngram_range=(2, 2), lowercase=False, analyzer=\"char_wb\")\nbi_name_boc = bigram_char_vectorizer.fit_transform(test_df[\"name\"].fillna(\"\")).tocoo()\nbi_name_boc_rag = tf.RaggedTensor.from_value_rowids(bi_name_boc.col, bi_name_boc.row, nrows=bi_name_boc.shape[0])\ndel bi_name_boc, bigram_char_vectorizer\n","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:59:00.247313Z","iopub.execute_input":"2022-07-08T02:59:00.247663Z","iopub.status.idle":"2022-07-08T02:59:00.905429Z","shell.execute_reply.started":"2022-07-08T02:59:00.247632Z","shell.execute_reply":"2022-07-08T02:59:00.904611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import phonenumbers\nfrom urllib.parse import urlparse\n\ndef get_phene_info(phone, country):\n    if phone == \"\":\n        return {\"phone_country_code\": 0, \"national_number\": 0}\n    if phonenumbers.country_code_for_region(country) == 0:\n        country = None\n    try:\n        res = phonenumbers.parse(phone, country)\n    except:\n        return {\"phone_country_code\": 0, \"national_number\": 0}\n    return {\"phone_country_code\": res.country_code, \"national_number\": res.national_number}\nres = []\nfor phone, country in test_df[[\"phone\", \"country\"]].values:\n    res.append(get_phene_info(phone, country))\nphone_df = pd.DataFrame(res)\n\ntest_df[\"url\"] = test_df[\"url\"].fillna(\"\")\ntest_df[\"phone_national_number\"] = phone_df[\"national_number\"].astype(str)\ntest_df[\"url_netloc\"] = test_df[\"url\"].fillna(\"\").apply(lambda x: urlparse(x)[1])\ntest_df[\"url_path\"] = test_df[\"url\"].fillna(\"\").apply(lambda x: urlparse(x)[2])\n","metadata":{"papermill":{"duration":0.502919,"end_time":"2022-07-07T13:08:42.010820","exception":false,"start_time":"2022-07-07T13:08:41.507901","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:59:02.395741Z","iopub.execute_input":"2022-07-08T02:59:02.396300Z","iopub.status.idle":"2022-07-08T02:59:02.862138Z","shell.execute_reply.started":"2022-07-08T02:59:02.396265Z","shell.execute_reply":"2022-07-08T02:59:02.860954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def change_number(text):\n    res = []\n    for word in text.split():\n        if word.isnumeric():\n            res.append(word)\n        else:\n            for r in re.findall(r\"\\d+\", str(word)):\n                res.append(r)\n    return \" \".join(res)\n\ntest_df[\"name_numbers\"] = np.vectorize(change_number)(test_df[\"name_normalized\"].fillna(\"\").astype(str))\ntest_df[\"address_numbers\"] = np.vectorize(change_number)(test_df[\"address_normalized\"].fillna(\"\").astype(str))","metadata":{"papermill":{"duration":0.140781,"end_time":"2022-07-07T13:08:42.181539","exception":false,"start_time":"2022-07-07T13:08:42.040758","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:59:02.866006Z","iopub.execute_input":"2022-07-08T02:59:02.868694Z","iopub.status.idle":"2022-07-08T02:59:02.987426Z","shell.execute_reply.started":"2022-07-08T02:59:02.868663Z","shell.execute_reply":"2022-07-08T02:59:02.986381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[\"name_len\"] = test_df[\"name\"].fillna(\"\").str.len().astype(np.int16)\ntest_df[\"n_name_len\"] = test_df[\"name_normalized\"].fillna(\"\").str.len().astype(np.int16)\ntest_df[\"sp_name_split\"] = test_df[\"sp_name\"].fillna(\"\").apply(lambda x: len(x.split())).astype(np.int16)\n","metadata":{"papermill":{"duration":0.069786,"end_time":"2022-07-07T13:08:42.280250","exception":false,"start_time":"2022-07-07T13:08:42.210464","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:59:04.658249Z","iopub.execute_input":"2022-07-08T02:59:04.658974Z","iopub.status.idle":"2022-07-08T02:59:04.701067Z","shell.execute_reply.started":"2022-07-08T02:59:04.658934Z","shell.execute_reply":"2022-07-08T02:59:04.700147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.notebook import tqdm\ncategory_col_names = [\"category_frequent_min\", \"category_frequent_min_tsp\", \"category_frequent_max\", \"category_frequent_max_tsp\", \"category_count\", \"category_tsp_delta\", \"category_tsp_mean\"]\n\n\ncategory_features = []\ncategory_ix_dict = {k: v for k, v in category_index_df[[\"category\", \"category_ix\"]].values}\ncategory_tsp_ix_dict = {k: v for k, v in category_index_df[[\"category\", \"category_tsp_ix\"]].values}\n\n# caution: if category is frequent in train, cix is near to 0\nfor categories in tqdm(test_df[\"categories\"]):\n    frequent_min = - 10 ** 9\n    frequent_min_tsp = -1\n    frequent_max = + (10**9)\n    frequent_max_tsp = -1\n    category_count = 0\n    for category in categories.split(\", \"):\n        cix = category_ix_dict.get(category, 0)\n        if cix == 0:\n            category_count -= 1\n            continue\n        ctix = category_tsp_ix_dict.get(category, 0)\n        if cix > frequent_min:\n            frequent_min = cix\n            frequent_min_tsp = ctix\n        if cix < frequent_max:\n            frequent_max = cix\n            frequent_max_tsp = ctix\n        category_count += 1\n    \n    category_delta = 0\n    category_mean = frequent_min_tsp\n    \n    if frequent_max_tsp > frequent_min_tsp:\n        category_delta = min(frequent_max_tsp - frequent_min_tsp, frequent_min_tsp + len(category_ix_dict) - frequent_max_tsp)\n        category_mean = (frequent_min_tsp + category_delta/2)%len(category_ix_dict)\n            \n    elif frequent_max_tsp < frequent_min_tsp:\n        category_delta = min(frequent_min_tsp - frequent_max_tsp, frequent_max_tsp + len(category_ix_dict) - frequent_min_tsp)\n        category_mean = (frequent_max_tsp + category_delta/2)%len(category_ix_dict)\n        \n    category_features.append([frequent_min, frequent_min_tsp, frequent_max, frequent_max_tsp, category_count, category_delta, category_mean])\n\ntest_df[category_col_names] = category_features\ntest_df[category_col_names[:-1]] = test_df[category_col_names[:-1]].astype(np.int16)\ntest_df[category_col_names[-1:]] = test_df[category_col_names[-1:]].astype(np.float16)\n\ndel category_features\ngc.collect()","metadata":{"papermill":{"duration":1.348836,"end_time":"2022-07-07T13:08:43.658433","exception":false,"start_time":"2022-07-07T13:08:42.309597","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:59:04.946421Z","iopub.execute_input":"2022-07-08T02:59:04.947097Z","iopub.status.idle":"2022-07-08T02:59:06.150257Z","shell.execute_reply.started":"2022-07-08T02:59:04.947059Z","shell.execute_reply":"2022-07-08T02:59:06.149422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from joblib import Parallel, delayed\nimport Levenshtein\nfrom difflib import SequenceMatcher\nfrom polyleven import levenshtein\nfrom cdifflib import CSequenceMatcher\nfrom jarowinkler import jarowinkler_similarity, jaro_similarity\n\ntest_df[\"name\"] = test_df[\"name\"].fillna(\"\")\ndef make_data(query_ix, candidate_ix):\n    query_info = {}\n    candidate_info = {}\n    query_info[\"ix\"] = query_ix\n    candidate_info[\"ix\"] = candidate_ix\n    \n    for col in [\"name\", \"sp_name\", \"name_normalized\", \"sp_address_raw\", \"address_normalized\", \"phone_national_number\", \"url\", \"phone_national_number\",\n                \"url_netloc\", \"name_numbers\", \"address_numbers\"]:\n        query_info[col] = test_df.at[query_ix, col]\n        candidate_info[col] = test_df.at[candidate_ix, col]\n    query_info[\"name_numbers_split_size\"] = len(query_info[\"name_numbers\"].split())\n    candidate_info[\"name_numbers_split_size\"] = len(candidate_info[\"name_numbers\"].split())\n    query_info[\"address_numbers_split_size\"] = len(query_info[\"address_numbers\"].split())\n    candidate_info[\"address_numbers_split_size\"] = len(candidate_info[\"address_numbers\"].split())    \n        \n    return query_info, candidate_info\n\ndef add_features(query_info, candidate_info):\n    res = []\n    res.append(query_info[\"ix\"])\n    res.append(candidate_info[\"ix\"])\n    \n    for col in [\"name\", \"sp_name\", \"name_normalized\", \"sp_address_raw\", \"address_normalized\"]:\n        s = query_info[col]\n        match_s = candidate_info[col]\n        if len(s) and len(match_s):\n            res.append(abs(len(match_s) - len(s)))\n            res.append(abs(len(match_s.split()) - len(s.split())))\n            res.append(CSequenceMatcher(None, s, match_s).ratio())\n            res.append(levenshtein(s, match_s))\n            res.append(res[-1] / (max(len(s), len(match_s)) + 1e-9))\n            res.append(Levenshtein.jaro_winkler(s, match_s))\n#             res.append(jarowinkler_similarity(s, match_s))\n#             res.append(jaro_similarity(s, match_s))\n        else:\n            res.append(abs(len(match_s) - len(s)))\n            res.append(abs(len(match_s.split()) - len(s.split())))\n            res.append(np.nan)\n            res.append(np.nan)\n            res.append(res[-1] / (max(len(s), len(match_s)) + 1e-9))\n            res.append(np.nan)\n    \n    for col in [\"phone_national_number\", \"url\", \"url_netloc\"]:\n        r = 0\n        if min(len(query_info[col]), len(candidate_info[col])) > 0:\n            r = -1 + 2 * (query_info[col] == candidate_info[col])\n        res.append(r)\n    for col in [\"name_numbers\", \"address_numbers\"]:\n        res.append(max(query_info[col + \"_split_size\"], candidate_info[col + \"_split_size\"]))\n        res.append(min(query_info[col + \"_split_size\"], candidate_info[col + \"_split_size\"]))\n        s = query_info[col]\n        match_s = candidate_info[col]\n        if s != '' and match_s != '':\n            res.append(SequenceMatcher(None, query_info[col], candidate_info[col]).ratio())\n        else:\n            res.append(np.nan)\n    \n    return res","metadata":{"papermill":{"duration":0.064197,"end_time":"2022-07-07T13:08:43.753428","exception":false,"start_time":"2022-07-07T13:08:43.689231","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:59:07.832887Z","iopub.execute_input":"2022-07-08T02:59:07.833534Z","iopub.status.idle":"2022-07-08T02:59:07.863008Z","shell.execute_reply.started":"2022-07-08T02:59:07.833496Z","shell.execute_reply":"2022-07-08T02:59:07.862120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_cols = ['name_delta_len',\n 'name_delta_words',\n 'name_gesh',\n 'name_leven',\n 'name_nleven',\n 'name_jaro',\n 'sp_name_delta_len',\n 'sp_name_delta_words',\n 'sp_name_gesh',\n 'sp_name_leven',\n 'sp_name_nleven',\n 'sp_name_jaro',\n 'name_normalized_delta_len',\n 'name_normalized_delta_words',\n 'name_normalized_gesh',\n 'name_normalized_leven',\n 'name_normalized_nleven',\n 'name_normalized_jaro',\n 'address_delta_len',\n 'address_delta_words',\n 'address_gesh',\n 'address_leven',\n 'address_nleven',\n 'address_jaro',\n 'address_normalized_delta_len',\n 'address_normalized_delta_words',\n 'address_normalized_gesh',\n 'address_normalized_leven',\n 'address_normalized_nleven',\n 'address_normalized_jaro',\n 'phone_national_number_match',\n 'url_match',\n 'url_netloc_match',\n 'name_numbers_max',\n 'name_numbers_min',\n 'name_numbers_gesh',\n 'address_numbers_max',\n 'address_numbers_min',\n 'address_numbers_gesh',\n 'rouge_1_name_word_f',\n 'rouge_1_name_word_p_rate',\n 'rouge_1_name_word_r_rate',\n 'rouge_1_name_word_p_cnt',\n 'rouge_1_name_word_r_cnt',\n 'rouge_2_name_word_f',\n 'rouge_2_name_word_p_rate',\n 'rouge_2_name_word_r_rate',\n 'rouge_2_name_word_p_cnt',\n 'rouge_2_name_word_r_cnt',\n 'rouge_1_name_char_f',\n 'rouge_1_name_char_p_rate',\n 'rouge_1_name_char_r_rate',\n 'rouge_1_name_char_p_cnt',\n 'rouge_1_name_char_r_cnt',\n 'rouge_2_name_char_f',\n 'rouge_2_name_char_p_rate',\n 'rouge_2_name_char_r_rate',\n 'rouge_2_name_char_p_cnt',\n 'rouge_2_name_char_r_cnt',\n 'rouge_l_name_f',\n 'rouge_l_name_p_rate',\n 'rouge_l_name_r_rate',\n 'rouge_l_name_p_cnt',\n 'rouge_l_name_r_cnt',\n 'rouge_l_n_name_f',\n 'rouge_l_n_name_p_rate',\n 'rouge_l_n_name_r_rate',\n 'rouge_l_n_name_p_cnt',\n 'rouge_l_n_name_r_cnt',\n 'rouge_l_address_f',\n 'rouge_l_address_p_rate',\n 'rouge_l_address_r_rate',\n 'rouge_l_address_p_cnt',\n 'rouge_l_address_r_cnt',\n 'rouge_l_n_address_f',\n 'rouge_l_n_address_p_rate',\n 'rouge_l_n_address_r_rate',\n 'rouge_l_n_address_p_cnt',\n 'rouge_l_n_address_r_cnt',\n 'distance',\n 'sp_name_word_sparse_sim',\n 'n_name_word_sparse_sim',\n 'name_char_sparse_sim',\n 'name_cossim',\n 'address_cossim',\n 'category_cossim',\n 'mix_cossim',\n 'query_city_index',\n 'query_geo_name_index',\n 'query_country_index',\n 'query_state_index',\n 'query_city_pos_tsp_index',\n 'query_geo_name_pos_tsp_index',\n 'query_state_pos_tsp_index',\n 'query_hkmeans_labels',\n 'query_name_kmeans_labels',\n 'query_address_kmeans_labels',\n 'query_ix_values_kmeans_labels',\n 'query_mix_kmeans_labels',\n 'query_category_kmeans_labels',\n 'query_brand_kmeans_labels',\n 'query_hkmeans_scores',\n 'query_name_kmeans_scores',\n 'query_address_kmeans_scores',\n 'query_ix_values_kmeans_scores',\n 'query_mix_kmeans_scores',\n 'query_category_kmeans_scores',\n 'query_brand_kmeans_scores',\n 'query_category_frequent_min',\n 'query_category_frequent_min_tsp',\n 'query_category_frequent_max',\n 'query_category_frequent_max_tsp',\n 'query_category_count',\n 'query_category_tsp_delta',\n 'query_category_tsp_mean',\n 'query_name_len',\n 'query_n_name_len',\n 'query_sp_name_split',\n 'candidate_city_index',\n 'candidate_geo_name_index',\n 'candidate_country_index',\n 'candidate_state_index',\n 'candidate_city_pos_tsp_index',\n 'candidate_geo_name_pos_tsp_index',\n 'candidate_state_pos_tsp_index',\n 'candidate_hkmeans_labels',\n 'candidate_name_kmeans_labels',\n 'candidate_address_kmeans_labels',\n 'candidate_ix_values_kmeans_labels',\n 'candidate_mix_kmeans_labels',\n 'candidate_category_kmeans_labels',\n 'candidate_brand_kmeans_labels',\n 'candidate_hkmeans_scores',\n 'candidate_name_kmeans_scores',\n 'candidate_address_kmeans_scores',\n 'candidate_ix_values_kmeans_scores',\n 'candidate_mix_kmeans_scores',\n 'candidate_category_kmeans_scores',\n 'candidate_brand_kmeans_scores',\n 'candidate_category_frequent_min',\n 'candidate_category_frequent_min_tsp',\n 'candidate_category_frequent_max',\n 'candidate_category_frequent_max_tsp',\n 'candidate_category_count',\n 'candidate_category_tsp_delta',\n 'candidate_category_tsp_mean',\n 'candidate_name_len',\n 'candidate_n_name_len',\n 'candidate_sp_name_split',\n 'hkmeans_labels_1d_dist',\n 'name_kmeans_labels_1d_dist',\n 'address_kmeans_labels_1d_dist',\n 'ix_values_kmeans_labels_1d_dist',\n 'mix_kmeans_labels_1d_dist',\n 'category_kmeans_labels_1d_dist',\n 'brand_kmeans_labels_1d_dist',\n 'city_pos_tsp_index_1d_dist',\n 'state_pos_tsp_index_1d_dist',\n 'geo_name_pos_tsp_index_1d_dist']\n\nmax_bin_by_feature = []\nfor col in feature_cols:\n    if (col.endswith(\"kmeans_labels\") or col.endswith(\"pos_tsp_index\")) and col.startswith(\"query\"):\n        print(col)\n        max_bin_by_feature.append(255*4)\n    elif (col.endswith(\"kmeans_labels\") or col.endswith(\"pos_tsp_index\")) and col.startswith(\"candidate\"):\n        print(col)\n        max_bin_by_feature.append(255*2) \n    elif col.endswith(\"_category_frequent_min_tsp\") or col.endswith(\"_category_frequent_max_tsp\"):\n        print(col)\n        max_bin_by_feature.append(255*2)\n    else:\n        max_bin_by_feature.append(255)","metadata":{"papermill":{"duration":0.0489,"end_time":"2022-07-07T13:08:43.831694","exception":false,"start_time":"2022-07-07T13:08:43.782794","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:59:09.422809Z","iopub.execute_input":"2022-07-08T02:59:09.423165Z","iopub.status.idle":"2022-07-08T02:59:09.441788Z","shell.execute_reply.started":"2022-07-08T02:59:09.423134Z","shell.execute_reply":"2022-07-08T02:59:09.440491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"papermill":{"duration":1.19167,"end_time":"2022-07-07T13:08:45.053041","exception":false,"start_time":"2022-07-07T13:08:43.861371","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T02:59:12.184446Z","iopub.execute_input":"2022-07-08T02:59:12.185128Z","iopub.status.idle":"2022-07-08T02:59:13.325155Z","shell.execute_reply.started":"2022-07-08T02:59:12.185090Z","shell.execute_reply":"2022-07-08T02:59:13.324095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_text as tf_text\nimport lightgbm as lgb\nfrom cuml import ForestInference\n\nlgb_model = ForestInference.load(filename = nadare_feature_dir + f\"lgb_model_pred34_mod1_fold5_epoch2000.lgb\", model_type='lightgbm')\n\n\nNUM_CANDIDATE = 32\nNEAREST = 8\nNAME_DUPLICATE = 3000\nNAME_THRETHOLD = 0.99\nNAME_SIM = 2\nNUM_FEAT = 8\n\ndef update_selected(selected, score, sub_score, threthold, size):\n    score, topk = tf.math.top_k(score, NAME_DUPLICATE)\n    update_pos = tf.where(score < threthold)\n    score = tf.tensor_scatter_nd_update(tf.gather(sub_score, topk, batch_dims=1), \n                                        update_pos,\n                                        tf.zeros(tf.shape(update_pos)[0]))\n    update_pos = tf.where(tf.reduce_any(tf.expand_dims(topk, axis=2) == tf.expand_dims(selected, axis=1), axis=2))\n    score = tf.tensor_scatter_nd_update(score, \n                                        update_pos,\n                                        tf.ones(tf.shape(update_pos)[0]) * tf.float32.min / 10)\n    topk = tf.gather(topk, tf.argsort(-score, axis=1), batch_dims=1)\n    selected = tf.concat([selected, tf.slice(topk, [0, 0], [-1, size])], axis=1) \n    return selected\n\n@tf.function(experimental_relax_shapes=True)\ndef make_pred_feat(ix):\n    selected = tf.expand_dims(ix, axis=1)\n    \n    query_position = tf.expand_dims(test_container.get_position(ix), axis=1)\n    dist = test_container.log_haversine(query_position, test_container.position)    \n    \n    indices = tf.stack([tf.repeat(tf.range(len(ix)), tf.gather(sp_name_word_bow_rag_data, ix).row_lengths()),\n                        tf.gather(sp_name_word_bow_rag_col, ix).values], axis=-1)\n    values = tf.gather(sp_name_word_bow_rag_data, ix).values\n    shape = [len(ix), sp_name_word_bow_coo.shape[1]]\n    query = tf.scatter_nd(indices, values, shape)\n    sp_name_word_sparse_match = tf.sparse.sparse_dense_matmul(query, sp_name_word_bow_coo, adjoint_b=True)\n    sp_name_word_sparse_sim = sp_name_word_sparse_match * tf.expand_dims(tf.gather(sp_name_r_norm, ix), axis=1) * tf.expand_dims(sp_name_r_norm, axis=0)\n    sp_name_word_sparse_p = sp_name_word_sparse_match / tf.expand_dims(tf.gather(sp_name_cnt, ix), axis=1)\n    sp_name_word_sparse_r = sp_name_word_sparse_match / tf.expand_dims(sp_name_cnt, axis=0)\n    \n    indices = tf.stack([tf.repeat(tf.range(len(ix)), tf.gather(n_name_word_bow_rag_data, ix).row_lengths()),\n                        tf.gather(n_name_word_bow_rag_col, ix).values], axis=-1)\n    values = tf.gather(n_name_word_bow_rag_data, ix).values\n    shape = [len(ix), n_name_word_bow_coo.shape[1]]\n    query = tf.scatter_nd(indices, values, shape)\n    n_name_word_sparse_sim = tf.sparse.sparse_dense_matmul(query, n_name_word_bow_coo, adjoint_b=True)\n    \n    indices = tf.stack([tf.repeat(tf.range(len(ix)), tf.gather(name_bichar_bow_rag_data, ix).row_lengths()),\n                        tf.gather(name_bichar_bow_rag_col, ix).values], axis=-1)\n    values = tf.gather(name_bichar_bow_rag_data, ix).values\n    shape = [len(ix), name_bichar_bow_coo.shape[1]]\n    query = tf.scatter_nd(indices, values, shape)\n    name_bichar_sparse_match = tf.sparse.sparse_dense_matmul(query, name_bichar_bow_coo, adjoint_b=True)\n    name_bichar_sparse_sim = name_bichar_sparse_match * tf.expand_dims(tf.gather(name_bichar_r_norm, ix), axis=1) * tf.expand_dims(name_bichar_r_norm, axis=0)\n    name_bichar_sparse_p = name_bichar_sparse_match / tf.expand_dims(tf.gather(name_bichar_cnt, ix), axis=1)\n    name_bichar_sparse_r = name_bichar_sparse_match / tf.expand_dims(name_bichar_cnt, axis=0)\n       \n    name_cossim = tf.einsum(\"nd,md->nm\", tf.gather(name_embeddings, ix), name_embeddings)\n    address_cossim = tf.einsum(\"nd,md->nm\", tf.gather(address_embeddings, ix), address_embeddings)\n    category_cossim = tf.einsum(\"nd,md->nm\", tf.gather(category_embeddings, ix), category_embeddings)   \n    mix_cossim = tf.einsum(\"nd,md->nm\", tf.gather(mix_embeddings, ix), mix_embeddings)   \n    \n    logistic_score = tf.einsum(\"d,nmd->nm\",\n                               simple_logistic_model.scale,\n                               tf.stack([dist,\n                                         name_cossim, \n                                         address_cossim, \n                                         category_cossim,\n                                         mix_cossim], axis=-1))\n    \n    # logistic_score: 4\n    selected = update_selected(selected, -dist, -dist, tf.float32.min / 10, 4)\n    \n    # logistic_score: 12\n    selected = update_selected(selected, logistic_score, logistic_score, tf.float32.min / 10, 12)\n    \n    # sp_name cossim, precision, 2recall: 4\n    for score, size in zip([sp_name_word_sparse_sim, sp_name_word_sparse_p, sp_name_word_sparse_r], [1, 1, 2]):\n        selected = update_selected(selected, score, logistic_score,  NAME_THRETHOLD, size)\n    \n    #name char cossim, precision, 2recall: 8\n    for score, size in zip([name_bichar_sparse_sim, name_bichar_sparse_p, name_bichar_sparse_r], [2, 2, 4]):\n        selected = update_selected(selected, score, logistic_score, NAME_THRETHOLD, size)\n        \n    #name cossim: 4\n    selected = update_selected(selected, name_cossim, logistic_score, NAME_THRETHOLD * .95 , 4)\n\n   \n    # remove myself\n    candidate_ixs  = tf.cast(tf.slice(selected, [0, 1], [-1, -1]), \"int32\")\n    candidate_dist = tf.gather(dist, candidate_ixs, batch_dims=1)\n    \n    candidate_sp_name_word_sparse_sim = tf.gather(sp_name_word_sparse_sim, candidate_ixs, batch_dims=1)\n    candidate_n_name_word_sparse_sim = tf.gather(n_name_word_sparse_sim, candidate_ixs, batch_dims=1)\n    candidate_name_bichar_sparse_sim = tf.gather(name_bichar_sparse_sim, candidate_ixs, batch_dims=1)\n \n    candidate_name_cossim = tf.gather(name_cossim, candidate_ixs, batch_dims=1)\n    candidate_address_cossim = tf.gather(address_cossim, candidate_ixs, batch_dims=1)\n    candidate_category_cossim = tf.gather(category_cossim, candidate_ixs, batch_dims=1)\n    candidate_mix_cossim = tf.gather(mix_cossim, candidate_ixs, batch_dims=1)\n    \n    \n    cross_feat = tf.stack([candidate_dist,\n                           candidate_sp_name_word_sparse_sim,\n                           candidate_n_name_word_sparse_sim,\n                           candidate_name_bichar_sparse_sim,\n                           candidate_name_cossim,\n                           candidate_address_cossim,\n                           candidate_category_cossim,\n                           candidate_mix_cossim], axis=-1)\n    \n    candidate_ixs = tf.reshape(candidate_ixs, [-1])\n    cross_feat = tf.reshape(cross_feat, [-1, NUM_FEAT])\n    \n    return candidate_ixs, cross_feat\n\nwsp_tokenizer = tf_text.WhitespaceTokenizer()\nrouge_l = tf_text.metrics.rouge_l\n\npredict_per_step = 10000\nstep_size = -(-len(test_ixs)//predict_per_step)\ntest_ixs_list = [test_ixs.numpy()[i*predict_per_step:(i+1)*predict_per_step] for i in range(step_size)]\n\nsp_names = tf.constant(test_df[\"sp_name\"].values)\nnormalized_names = tf.constant(test_df[\"name_normalized\"].values)\nsp_address = tf.constant(test_df[\"sp_address_raw\"].values)\nnormalized_address = tf.constant(test_df[\"address_normalized\"].values)\n\ndef update_selected(selected, score, sub_score, threthold, size):\n    score, topk = tf.math.top_k(score, NAME_DUPLICATE)\n    update_pos = tf.where(score < threthold)\n    score = tf.tensor_scatter_nd_update(tf.gather(sub_score, topk, batch_dims=1), \n                                        update_pos,\n                                        tf.zeros(tf.shape(update_pos)[0]))\n    update_pos = tf.where(tf.reduce_any(tf.expand_dims(topk, axis=2) == tf.expand_dims(selected, axis=1), axis=2))\n    score = tf.tensor_scatter_nd_update(score, \n                                        update_pos,\n                                        tf.ones(tf.shape(update_pos)[0]) * tf.float32.min / 10)\n    topk = tf.gather(topk, tf.argsort(-score, axis=1), batch_dims=1)\n    selected = tf.concat([selected, tf.slice(topk, [0, 0], [-1, size])], axis=1) \n    return selected\n\n@tf.function(experimental_relax_shapes=True)\ndef make_pred_feat(ix):\n    selected = tf.expand_dims(ix, axis=1)\n    \n    query_position = tf.expand_dims(test_container.get_position(ix), axis=1)\n    dist = test_container.log_haversine(query_position, test_container.position)    \n    \n    indices = tf.stack([tf.repeat(tf.range(len(ix)), tf.gather(sp_name_word_bow_rag_data, ix).row_lengths()),\n                        tf.gather(sp_name_word_bow_rag_col, ix).values], axis=-1)\n    values = tf.gather(sp_name_word_bow_rag_data, ix).values\n    shape = [len(ix), sp_name_word_bow_coo.shape[1]]\n    query = tf.scatter_nd(indices, values, shape)\n    sp_name_word_sparse_match = tf.sparse.sparse_dense_matmul(query, sp_name_word_bow_coo, adjoint_b=True)\n    sp_name_word_sparse_sim = sp_name_word_sparse_match * tf.expand_dims(tf.gather(sp_name_r_norm, ix), axis=1) * tf.expand_dims(sp_name_r_norm, axis=0)\n    sp_name_word_sparse_p = sp_name_word_sparse_match / tf.expand_dims(tf.gather(sp_name_cnt, ix), axis=1)\n    sp_name_word_sparse_r = sp_name_word_sparse_match / tf.expand_dims(sp_name_cnt, axis=0)\n    \n    indices = tf.stack([tf.repeat(tf.range(len(ix)), tf.gather(n_name_word_bow_rag_data, ix).row_lengths()),\n                        tf.gather(n_name_word_bow_rag_col, ix).values], axis=-1)\n    values = tf.gather(n_name_word_bow_rag_data, ix).values\n    shape = [len(ix), n_name_word_bow_coo.shape[1]]\n    query = tf.scatter_nd(indices, values, shape)\n    n_name_word_sparse_sim = tf.sparse.sparse_dense_matmul(query, n_name_word_bow_coo, adjoint_b=True)\n    \n    indices = tf.stack([tf.repeat(tf.range(len(ix)), tf.gather(name_bichar_bow_rag_data, ix).row_lengths()),\n                        tf.gather(name_bichar_bow_rag_col, ix).values], axis=-1)\n    values = tf.gather(name_bichar_bow_rag_data, ix).values\n    shape = [len(ix), name_bichar_bow_coo.shape[1]]\n    query = tf.scatter_nd(indices, values, shape)\n    name_bichar_sparse_match = tf.sparse.sparse_dense_matmul(query, name_bichar_bow_coo, adjoint_b=True)\n    name_bichar_sparse_sim = name_bichar_sparse_match * tf.expand_dims(tf.gather(name_bichar_r_norm, ix), axis=1) * tf.expand_dims(name_bichar_r_norm, axis=0)\n    name_bichar_sparse_p = name_bichar_sparse_match / tf.expand_dims(tf.gather(name_bichar_cnt, ix), axis=1)\n    name_bichar_sparse_r = name_bichar_sparse_match / tf.expand_dims(name_bichar_cnt, axis=0)\n       \n    name_cossim = tf.einsum(\"nd,md->nm\", tf.gather(name_embeddings, ix), name_embeddings)\n    address_cossim = tf.einsum(\"nd,md->nm\", tf.gather(address_embeddings, ix), address_embeddings)\n    category_cossim = tf.einsum(\"nd,md->nm\", tf.gather(category_embeddings, ix), category_embeddings)   \n    mix_cossim = tf.einsum(\"nd,md->nm\", tf.gather(mix_embeddings, ix), mix_embeddings)   \n    \n    logistic_score = tf.einsum(\"d,nmd->nm\",\n                               simple_logistic_model.scale,\n                               tf.stack([dist,\n                                         name_cossim, \n                                         address_cossim, \n                                         category_cossim,\n                                         mix_cossim], axis=-1))\n    \n    # logistic_score: 4\n    selected = update_selected(selected, -dist, -dist, tf.float32.min / 10, 4)\n    \n    # logistic_score: 12\n    selected = update_selected(selected, logistic_score, logistic_score, tf.float32.min / 10, 12)\n    \n    # sp_name cossim, precision, 2recall: 4\n    for score, size in zip([sp_name_word_sparse_sim, sp_name_word_sparse_p, sp_name_word_sparse_r], [1, 1, 2]):\n        selected = update_selected(selected, score, logistic_score,  NAME_THRETHOLD, size)\n    \n    #name char cossim, precision, 2recall: 8\n    for score, size in zip([name_bichar_sparse_sim, name_bichar_sparse_p, name_bichar_sparse_r], [2, 2, 4]):\n        selected = update_selected(selected, score, logistic_score, NAME_THRETHOLD, size)\n        \n    #name cossim: 4\n    selected = update_selected(selected, name_cossim, logistic_score, NAME_THRETHOLD * .95 , 4)\n\n   \n    # remove myself\n    candidate_ixs  = tf.cast(tf.slice(selected, [0, 1], [-1, -1]), \"int32\")\n    candidate_dist = tf.gather(dist, candidate_ixs, batch_dims=1)\n    \n    candidate_sp_name_word_sparse_sim = tf.gather(sp_name_word_sparse_sim, candidate_ixs, batch_dims=1)\n    candidate_n_name_word_sparse_sim = tf.gather(n_name_word_sparse_sim, candidate_ixs, batch_dims=1)\n    candidate_name_bichar_sparse_sim = tf.gather(name_bichar_sparse_sim, candidate_ixs, batch_dims=1)\n \n    candidate_name_cossim = tf.gather(name_cossim, candidate_ixs, batch_dims=1)\n    candidate_address_cossim = tf.gather(address_cossim, candidate_ixs, batch_dims=1)\n    candidate_category_cossim = tf.gather(category_cossim, candidate_ixs, batch_dims=1)\n    candidate_mix_cossim = tf.gather(mix_cossim, candidate_ixs, batch_dims=1)\n    \n    \n    cross_feat = tf.stack([candidate_dist,\n                           candidate_sp_name_word_sparse_sim,\n                           candidate_n_name_word_sparse_sim,\n                           candidate_name_bichar_sparse_sim,\n                           candidate_name_cossim,\n                           candidate_address_cossim,\n                           candidate_category_cossim,\n                           candidate_mix_cossim], axis=-1)\n    \n    candidate_ixs = tf.reshape(candidate_ixs, [-1])\n    cross_feat = tf.reshape(cross_feat, [-1, NUM_FEAT])\n    \n    return candidate_ixs, cross_feat\n\n\n@tf.function(experimental_relax_shapes=True)\ndef calc_rouge_1_2(q_ix, c_ix):\n    q_code = tf.gather(mono_sp_name_bow_rag, q_ix).to_tensor(default_value=-1)\n    c_code = tf.gather(mono_sp_name_bow_rag, c_ix).to_tensor(default_value=-2)\n    q_cnt = tf.cast(tf.gather(mono_sp_name_bow_rag.row_lengths(), q_ix), \"float32\")\n    c_cnt = tf.cast(tf.gather(mono_sp_name_bow_rag.row_lengths(), c_ix), \"float32\")\n\n    is_na = tf.minimum(q_cnt, c_cnt) == 0.\n    match = tf.expand_dims(q_code, axis=2) == tf.expand_dims(c_code, axis=1)\n    precision = tf.math.count_nonzero(tf.math.reduce_any(match, axis=2), axis=1, dtype=\"float32\")# / (q_cnt + tf.keras.backend.epsilon())\n    recall = tf.math.count_nonzero(tf.math.reduce_any(match, axis=1), axis=1, dtype=\"float32\")# / (c_cnt + tf.keras.backend.epsilon())\n    precision_rate = precision / (q_cnt + tf.keras.backend.epsilon())\n    recall_rate = recall / (c_cnt + tf.keras.backend.epsilon())\n    precision_cnt = q_cnt - precision\n    recall_cnt = c_cnt - recall    \n    harmony = 2. / ((1 / (precision_rate + tf.keras.backend.epsilon())) + (1 / (recall_rate + tf.keras.backend.epsilon())))\n    res_1 = tf.where(tf.expand_dims(is_na, axis=-1), np.nan, tf.stack([harmony, precision_rate, recall_rate, precision_cnt, recall_cnt], axis=1))\n    \n    q_code = tf.gather(bi_sp_name_bow_rag, q_ix).to_tensor(default_value=-1)\n    c_code = tf.gather(bi_sp_name_bow_rag, c_ix).to_tensor(default_value=-2)\n    q_cnt = tf.cast(tf.gather(bi_sp_name_bow_rag.row_lengths(), q_ix), \"float32\")\n    c_cnt = tf.cast(tf.gather(bi_sp_name_bow_rag.row_lengths(), c_ix), \"float32\")\n\n    is_na = tf.minimum(q_cnt, c_cnt) == 0.\n    match = tf.expand_dims(q_code, axis=2) == tf.expand_dims(c_code, axis=1)\n    precision = tf.math.count_nonzero(tf.math.reduce_any(match, axis=2), axis=1, dtype=\"float32\")# / (q_cnt + tf.keras.backend.epsilon())\n    recall = tf.math.count_nonzero(tf.math.reduce_any(match, axis=1), axis=1, dtype=\"float32\")# / (c_cnt + tf.keras.backend.epsilon())\n    precision_rate = precision / (q_cnt + tf.keras.backend.epsilon())\n    recall_rate = recall / (c_cnt + tf.keras.backend.epsilon())\n    precision_cnt = q_cnt - precision\n    recall_cnt = c_cnt - recall    \n    harmony = 2. / ((1 / (precision_rate + tf.keras.backend.epsilon())) + (1 / (recall_rate + tf.keras.backend.epsilon())))\n    res_2 = tf.where(tf.expand_dims(is_na, axis=-1), np.nan, tf.stack([harmony, precision_rate, recall_rate, precision_cnt, recall_cnt], axis=1))\n    \n    q_code = tf.gather(mono_name_boc_rag, q_ix).to_tensor(default_value=-1)\n    c_code = tf.gather(mono_name_boc_rag, c_ix).to_tensor(default_value=-2)\n    q_cnt = tf.cast(tf.gather(mono_name_boc_rag.row_lengths(), q_ix), \"float32\")\n    c_cnt = tf.cast(tf.gather(mono_name_boc_rag.row_lengths(), c_ix), \"float32\")\n\n    is_na = tf.minimum(q_cnt, c_cnt) == 0.\n    match = tf.expand_dims(q_code, axis=2) == tf.expand_dims(c_code, axis=1)\n    precision = tf.math.count_nonzero(tf.math.reduce_any(match, axis=2), axis=1, dtype=\"float32\")# / (q_cnt + tf.keras.backend.epsilon())\n    recall = tf.math.count_nonzero(tf.math.reduce_any(match, axis=1), axis=1, dtype=\"float32\")# / (c_cnt + tf.keras.backend.epsilon())\n    precision_rate = precision / (q_cnt + tf.keras.backend.epsilon())\n    recall_rate = recall / (c_cnt + tf.keras.backend.epsilon())\n    precision_cnt = q_cnt - precision\n    recall_cnt = c_cnt - recall    \n    harmony = 2. / ((1 / (precision_rate + tf.keras.backend.epsilon())) + (1 / (recall_rate + tf.keras.backend.epsilon())))\n    res_3 = tf.where(tf.expand_dims(is_na, axis=-1), np.nan, tf.stack([harmony, precision_rate, recall_rate, precision_cnt, recall_cnt], axis=1))\n    \n    q_code = tf.gather(bi_name_boc_rag, q_ix).to_tensor(default_value=-1)\n    c_code = tf.gather(bi_name_boc_rag, c_ix).to_tensor(default_value=-2)\n    q_cnt = tf.cast(tf.gather(bi_name_boc_rag.row_lengths(), q_ix), \"float32\")\n    c_cnt = tf.cast(tf.gather(bi_name_boc_rag.row_lengths(), c_ix), \"float32\")\n\n    is_na = tf.minimum(q_cnt, c_cnt) == 0.\n    match = tf.expand_dims(q_code, axis=2) == tf.expand_dims(c_code, axis=1)\n    precision = tf.math.count_nonzero(tf.math.reduce_any(match, axis=2), axis=1, dtype=\"float32\")# / (q_cnt + tf.keras.backend.epsilon())\n    recall = tf.math.count_nonzero(tf.math.reduce_any(match, axis=1), axis=1, dtype=\"float32\")# / (c_cnt + tf.keras.backend.epsilon())\n    precision_rate = precision / (q_cnt + tf.keras.backend.epsilon())\n    recall_rate = recall / (c_cnt + tf.keras.backend.epsilon())\n    precision_cnt = q_cnt - precision\n    recall_cnt = c_cnt - recall    \n    harmony = 2. / ((1 / (precision_rate + tf.keras.backend.epsilon())) + (1 / (recall_rate + tf.keras.backend.epsilon())))\n    res_4 = tf.where(tf.expand_dims(is_na, axis=-1), np.nan, tf.stack([harmony, precision_rate, recall_rate, precision_cnt, recall_cnt], axis=1))\n    \n    return tf.concat([res_1, res_2, res_3, res_4], axis=-1)\n\n@tf.function(experimental_relax_shapes=True)\ndef calc_rouge_l(q_ix, c_ix):\n    q_name = wsp_tokenizer.tokenize(tf.gather(sp_names, q_ix))\n    c_name = wsp_tokenizer.tokenize(tf.gather(sp_names, c_ix))\n    name_score = tf.stack(tf_text.metrics.rouge_l(q_name, c_name), axis=-1)\n    name_p_count = (1. - tf.gather(name_score, 1, axis=1)) * tf.cast(q_name.row_lengths(), \"float32\")\n    name_r_count = (1. - tf.gather(name_score, 2, axis=1)) * tf.cast(c_name.row_lengths(), \"float32\")\n    name_score = tf.concat([name_score, tf.stack([name_p_count, name_r_count], axis=1)], axis=1)    \n    \n    q_name = uch_tokenizer.tokenize(tf.gather(normalized_names, q_ix))\n    c_name = uch_tokenizer.tokenize(tf.gather(normalized_names, c_ix))\n    n_name_score = tf.stack(tf_text.metrics.rouge_l(q_name, c_name), axis=-1)\n    name_p_count = (1. - tf.gather(n_name_score, 1, axis=1)) * tf.cast(q_name.row_lengths(), \"float32\")\n    name_r_count = (1. - tf.gather(n_name_score, 2, axis=1)) * tf.cast(c_name.row_lengths(), \"float32\")\n    n_name_score = tf.concat([n_name_score, tf.stack([name_p_count, name_r_count], axis=1)], axis=1)\n    \n    q_address = wsp_tokenizer.tokenize(tf.gather(sp_address, q_ix))\n    c_address = wsp_tokenizer.tokenize(tf.gather(sp_address, c_ix))\n    address_score = tf.stack(tf_text.metrics.rouge_l(q_address, c_address), axis=-1)\n    address_p_count = (1. - tf.gather(address_score, 1, axis=1)) * tf.cast(q_address.row_lengths(), \"float32\")\n    address_r_count = (1. - tf.gather(address_score, 2, axis=1)) * tf.cast(c_address.row_lengths(), \"float32\")\n    address_score = tf.concat([address_score, tf.stack([address_p_count, address_r_count], axis=1)], axis=1)    \n    \n    q_address = uch_tokenizer.tokenize(tf.gather(normalized_address, q_ix))\n    c_address = uch_tokenizer.tokenize(tf.gather(normalized_address, c_ix))\n    n_address_score = tf.stack(tf_text.metrics.rouge_l(q_address, c_address), axis=-1)\n    address_p_count = (1. - tf.gather(n_address_score, 1, axis=1)) * tf.cast(q_address.row_lengths(), \"float32\")\n    address_r_count = (1. - tf.gather(n_address_score, 2, axis=1)) * tf.cast(c_address.row_lengths(), \"float32\")\n    n_address_score = tf.concat([n_address_score, tf.stack([address_p_count, address_r_count], axis=1)], axis=1)\n    \n    score = tf.concat([name_score, n_name_score, address_score, n_address_score], axis=-1)\n    return score","metadata":{"papermill":{"duration":25.122078,"end_time":"2022-07-07T13:09:10.971244","exception":false,"start_time":"2022-07-07T13:08:45.849166","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T03:00:21.364913Z","iopub.execute_input":"2022-07-08T03:00:21.365304Z","iopub.status.idle":"2022-07-08T03:00:47.928906Z","shell.execute_reply.started":"2022-07-08T03:00:21.365272Z","shell.execute_reply":"2022-07-08T03:00:47.928125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nimport tensorflow_text as tf_text\nwsp_tokenizer = tf_text.WhitespaceTokenizer()\nuch_tokenizer = tf_text.UnicodeCharTokenizer()\nrouge_l = tf_text.metrics.rouge_l\n\npredictions = []\nfor mini_test_ixs in test_ixs_list:\n    test_ds = tf.data.Dataset.from_tensor_slices(tf.constant(mini_test_ixs))\\\n                    .batch(100)\n\n    query_ixs = []\n    candidate_ixs = []\n    cross_feats = []\n\n    for query_ix in test_ds:\n        query_ix = tf.reshape(query_ix, [-1])\n        candidate_ix, cross_feat = make_pred_feat(query_ix)\n\n        query_ixs.append(query_ix)\n        candidate_ixs.append(candidate_ix)\n        cross_feats.append(cross_feat)\n\n    query_ixs = tf.repeat(tf.concat(query_ixs, axis=0), NUM_CANDIDATE, axis=0)\n    candidate_ixs = tf.concat(candidate_ixs, axis=0)\n    cross_feats = tf.concat(cross_feats, axis=0)\n    \n    results = Parallel(n_jobs=-1, verbose=1)(delayed(add_features)(*make_data(q_ix, c_ix)) for q_ix, c_ix in zip(query_ixs.numpy(), candidate_ixs.numpy()))\n    candidate_data_df =  pd.DataFrame(results, \n                       columns = [\"query_ix\",\n                                                         \"candidate_ix\",\n                                                         \"name_delta_len\",\n                                                         \"name_delta_words\",\n                                                         \"name_gesh\", \n                                                         \"name_leven\", \n                                                         \"name_nleven\", \n                                                         \"name_jaro\",               \n                                                         \"sp_name_delta_len\",\n                                                         \"sp_name_delta_words\",\n                                                         \"sp_name_gesh\", \n                                                         \"sp_name_leven\", \n                                                         \"sp_name_nleven\", \n                                                         \"sp_name_jaro\",                                                  \n                                                         \"name_normalized_delta_len\",\n                                                         \"name_normalized_delta_words\",\n                                                         \"name_normalized_gesh\", \n                                                         \"name_normalized_leven\", \n                                                         \"name_normalized_nleven\", \n                                                         \"name_normalized_jaro\", \n                                                         \"address_delta_len\", \n                                                         \"address_delta_words\", \n                                                         \"address_gesh\", \n                                                         \"address_leven\", \n                                                         \"address_nleven\", \n                                                         \"address_jaro\",\n                                                         \"address_normalized_delta_len\", \n                                                         \"address_normalized_delta_words\", \n                                                         \"address_normalized_gesh\", \n                                                         \"address_normalized_leven\", \n                                                         \"address_normalized_nleven\", \n                                                         \"address_normalized_jaro\",\n                                                         \"phone_national_number_match\",\n                                                         \"url_match\",\n                                                         \"url_netloc_match\",\n                                                         \"name_numbers_max\",\n                                                         \"name_numbers_min\",\n                                                         \"name_numbers_gesh\",\n                                                         \"address_numbers_max\",\n                                                         \"address_numbers_min\",\n                                                         \"address_numbers_gesh\",])\n    \n    del results\n    gc.collect()\n\n    int64_cols = ['name_delta_len', 'name_delta_words',\n                  'sp_name_delta_len', 'sp_name_delta_words',\n                  'name_normalized_delta_len', 'name_normalized_delta_words',\n                  'address_delta_len','address_delta_words', \n                  'address_normalized_delta_len','address_normalized_delta_words', \n                  'phone_national_number_match', 'url_match', 'url_netloc_match',\n                  'name_numbers_max', 'name_numbers_min',\n                  'address_numbers_max', 'address_numbers_min']\n    float64_cols = ['name_gesh', 'name_nleven', 'name_jaro', 'name_leven',\n                    'sp_name_gesh', 'sp_name_nleven', 'sp_name_jaro', 'sp_name_leven', \n                    'name_normalized_gesh', 'name_normalized_nleven', 'name_normalized_jaro', 'name_normalized_leven',\n                    'address_gesh','address_nleven', 'address_jaro', 'address_leven',\n                    'address_normalized_gesh','address_normalized_nleven', 'address_normalized_jaro', 'address_normalized_leven',\n                    'name_numbers_gesh','address_numbers_gesh']\n    candidate_data_df[int64_cols] = candidate_data_df[int64_cols].astype(np.int16)\n    candidate_data_df[float64_cols] = candidate_data_df[float64_cols].astype(np.float32)\n\n    pairix_ds = tf.data.Dataset.zip((tf.data.Dataset.from_tensor_slices(query_ixs), tf.data.Dataset.from_tensor_slices(candidate_ixs)))\\\n                             .batch(1000)\n\n    n_scores = []\n    l_scores = []\n    for q_ix, c_ix in tqdm(pairix_ds):\n        n_scores.append(calc_rouge_1_2(q_ix, c_ix))\n        l_scores.append(calc_rouge_l(q_ix, c_ix))\n    n_scores = tf.concat(n_scores, axis=0)\n    l_scores = tf.concat(l_scores, axis=0)\n    \n    candidate_data_df[[\"rouge_1_name_word_f\", \"rouge_1_name_word_p_rate\", \"rouge_1_name_word_r_rate\", \"rouge_1_name_word_p_cnt\", \"rouge_1_name_word_r_cnt\",\n                   \"rouge_2_name_word_f\", \"rouge_2_name_word_p_rate\", \"rouge_2_name_word_r_rate\", \"rouge_2_name_word_p_cnt\", \"rouge_2_name_word_r_cnt\",\n                   \"rouge_1_name_char_f\", \"rouge_1_name_char_p_rate\", \"rouge_1_name_char_r_rate\", \"rouge_1_name_char_p_cnt\", \"rouge_1_name_char_r_cnt\",\n                   \"rouge_2_name_char_f\", \"rouge_2_name_char_p_rate\", \"rouge_2_name_char_r_rate\", \"rouge_2_name_char_p_cnt\", \"rouge_2_name_char_r_cnt\"]] = n_scores.numpy().astype(np.float32)\n    candidate_data_df[[\"rouge_l_name_f\", \"rouge_l_name_p_rate\", \"rouge_l_name_r_rate\", \"rouge_l_name_p_cnt\", \"rouge_l_name_r_cnt\",\n                   \"rouge_l_n_name_f\", \"rouge_l_n_name_p_rate\", \"rouge_l_n_name_r_rate\", \"rouge_l_n_name_p_cnt\", \"rouge_l_n_name_r_cnt\",\n                   \"rouge_l_address_f\", \"rouge_l_address_p_rate\", \"rouge_l_address_r_rate\", \"rouge_l_address_p_cnt\", \"rouge_l_address_r_cnt\",\n                   \"rouge_l_n_address_f\", \"rouge_l_n_address_p_rate\", \"rouge_l_n_address_r_rate\", \"rouge_l_n_address_p_cnt\", \"rouge_l_n_address_r_cnt\"]] = l_scores.numpy().astype(np.float32)\n    \n    for i, col in enumerate([\"distance\",\n                               \"sp_name_word_sparse_sim\",\n                               \"n_name_word_sparse_sim\",\n                               \"name_char_sparse_sim\",\n                               \"name_cossim\",\n                               \"address_cossim\",\n                               \"category_cossim\",\n                               \"mix_cossim\"]):\n        candidate_data_df[col] = tf.reshape(tf.gather(cross_feats, i, axis=1), [-1]).numpy().astype(np.float32)\n\n    len_col = [\"name_len\", \"n_name_len\", \"sp_name_split\"]\n    category_tsp_cols = [\"city_pos_tsp_index\", \"state_pos_tsp_index\", \"geo_name_pos_tsp_index\"]\n    category_col_names\n    kmeans_labels_cols = ['hkmeans_labels',\n                          'name_kmeans_labels',\n                          'address_kmeans_labels', \n                          'ix_values_kmeans_labels',\n                          \"mix_kmeans_labels\",\n                          \"category_kmeans_labels\",\n                          'brand_kmeans_labels']\n    kmeans_scores_cols = ['hkmeans_scores',\n                          'name_kmeans_scores', \n                          'address_kmeans_scores', \n                          'ix_values_kmeans_scores',\n                          \"mix_kmeans_scores\", \n                          \"category_kmeans_scores\",\n                          'brand_kmeans_scores']\n    \n    candidate_data_df[\"query_city_index\"] = test_df[\"pseudo_city_ix\"].values.astype(np.int16)[candidate_data_df[\"query_ix\"]]\n    candidate_data_df[\"query_geo_name_index\"] = test_df[\"pseudo_geo_name_ix\"].values.astype(np.int16)[candidate_data_df[\"query_ix\"]]\n    candidate_data_df[\"query_country_index\"] = test_df[\"country_ix\"].values.astype(np.int16)[candidate_data_df[\"query_ix\"]]\n    candidate_data_df[\"query_state_index\"] = test_df[\"pseudo_state_ix\"].values.astype(np.int16)[candidate_data_df[\"query_ix\"]]\n\n    #candidate_data_df[\"query_city_emb_tsp_index\"] = city_index_df[\"city_emb_tsp_index\"].values.astype(np.int16)[candidate_data_df[\"query_city_index\"]]\n    #candidate_data_df[\"query_state_emb_tsp_index\"] = state_index_df[\"state_emb_tsp_index\"].values.astype(np.int16)[candidate_data_df[\"query_state_index\"]]\n    candidate_data_df[\"query_city_pos_tsp_index\"] = city_index_df[\"city_pos_tsp_index\"].values.astype(np.int16)[candidate_data_df[\"query_city_index\"]]\n    candidate_data_df[\"query_geo_name_pos_tsp_index\"] = geo_name_index_df[\"geo_name_pos_tsp_index\"].values.astype(np.int16)[candidate_data_df[\"query_geo_name_index\"]]\n    candidate_data_df[\"query_state_pos_tsp_index\"] = state_index_df[\"state_pos_tsp_index\"].values.astype(np.int16)[candidate_data_df[\"query_state_index\"]]\n\n    for col in kmeans_labels_cols:\n        candidate_data_df[f\"query_{col}\"] = test_df[col].values.astype(np.int16)[candidate_data_df[\"query_ix\"]]\n    for col in kmeans_scores_cols:\n        candidate_data_df[f\"query_{col}\"] = test_df[col].values.astype(np.float32)[candidate_data_df[\"query_ix\"]]\n    for col in category_col_names:\n        candidate_data_df[f\"query_{col}\"] = test_df[col].values[candidate_data_df[\"query_ix\"]]\n    for col in len_col:\n        candidate_data_df[f\"query_{col}\"] = test_df[col].values[candidate_data_df[\"query_ix\"]]\n\n\n    candidate_data_df[\"candidate_city_index\"] = test_df[\"pseudo_city_ix\"].values.astype(np.int16)[candidate_data_df[\"candidate_ix\"]]\n    candidate_data_df[\"candidate_geo_name_index\"] = test_df[\"pseudo_geo_name_ix\"].values.astype(np.int16)[candidate_data_df[\"candidate_ix\"]]\n\n    candidate_data_df[\"candidate_country_index\"] = test_df[\"country_ix\"].values.astype(np.int16)[candidate_data_df[\"candidate_ix\"]]\n    candidate_data_df[\"candidate_state_index\"] = test_df[\"pseudo_state_ix\"].values.astype(np.int16)[candidate_data_df[\"candidate_ix\"]]\n\n    #andidate_data_df[\"candidate_city_emb_tsp_index\"] = city_index_df[\"city_emb_tsp_index\"].values.astype(np.int16)[candidate_data_df[\"candidate_city_index\"]]\n    #candidate_data_df[\"candidate_state_emb_tsp_index\"] = state_index_df[\"state_emb_tsp_index\"].values.astype(np.int16)[candidate_data_df[\"candidate_state_index\"]]\n    candidate_data_df[\"candidate_city_pos_tsp_index\"] = city_index_df[\"city_pos_tsp_index\"].values.astype(np.int16)[candidate_data_df[\"candidate_city_index\"]]\n    candidate_data_df[\"candidate_geo_name_pos_tsp_index\"] = geo_name_index_df[\"geo_name_pos_tsp_index\"].values.astype(np.int16)[candidate_data_df[\"candidate_geo_name_index\"]]\n    candidate_data_df[\"candidate_state_pos_tsp_index\"] = state_index_df[\"state_pos_tsp_index\"].values.astype(np.int16)[candidate_data_df[\"candidate_state_index\"]]\n\n    for col in kmeans_labels_cols:\n        candidate_data_df[f\"candidate_{col}\"] = test_df[col].values.astype(np.int16)[candidate_data_df[\"candidate_ix\"]]\n    for col in kmeans_scores_cols:\n        candidate_data_df[f\"candidate_{col}\"] = test_df[col].values.astype(np.float32)[candidate_data_df[\"candidate_ix\"]]\n    for col in category_col_names:\n        candidate_data_df[f\"candidate_{col}\"] = test_df[col].values[candidate_data_df[\"candidate_ix\"]]\n    for col in len_col:\n        candidate_data_df[f\"candidate_{col}\"] = test_df[col].values[candidate_data_df[\"candidate_ix\"]]\n\n    for col in kmeans_labels_cols + category_tsp_cols:\n        candidate_data_df[f\"{col}_1d_dist\"] = np.abs(candidate_data_df[f\"query_{col}\"] - candidate_data_df[f\"candidate_{col}\"])\n    \n    candidate_data_df[\"pred\"] = lgb_model.predict(candidate_data_df[feature_cols])\n    predictions.append(candidate_data_df[[\"query_ix\", \"candidate_ix\", \"pred\"]].query(\"pred > 0.90\"))\n    gc.collect()\n","metadata":{"ExecuteTime":{"end_time":"2022-06-25T10:58:53.832032Z","start_time":"2022-06-25T10:35:49.814013Z"},"papermill":{"duration":246.700745,"end_time":"2022-07-07T13:13:17.701762","exception":false,"start_time":"2022-07-07T13:09:11.001017","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T03:02:54.079695Z","iopub.execute_input":"2022-07-08T03:02:54.080091Z","iopub.status.idle":"2022-07-08T03:06:52.099262Z","shell.execute_reply.started":"2022-07-08T03:02:54.080059Z","shell.execute_reply":"2022-07-08T03:06:52.098393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.concat(predictions, axis=0).reset_index(drop=True)\ndel predictions\ngc.collect()\n\nquery_col = \"query_ix\"\ncandidate_col = \"candidate_ix\"\nscore_col = \"pred\"","metadata":{"ExecuteTime":{"end_time":"2022-06-25T10:58:53.894273Z","start_time":"2022-06-25T10:58:53.833034Z"},"papermill":{"duration":1.307934,"end_time":"2022-07-07T13:13:19.041988","exception":false,"start_time":"2022-07-07T13:13:17.734054","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T03:07:22.333385Z","iopub.execute_input":"2022-07-08T03:07:22.334236Z","iopub.status.idle":"2022-07-08T03:07:23.575996Z","shell.execute_reply.started":"2022-07-08T03:07:22.334175Z","shell.execute_reply":"2022-07-08T03:07:23.574829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from itertools import groupby\nixs2idx = {tuple(x):i for i, x in enumerate(pred_df[[\"query_ix\", \"candidate_ix\"]].values)}\n\ndef using_edge_betweenness_centrality(df, remove_edge_threshold = 0.2):\n    G = nx.Graph()\n    for i, j, _ in df[['query_ix', 'candidate_ix', 'pred']].values:\n        G.add_edge(i, j, weight=1)\n\n    def split_graph(G):\n        list_remove_edges = []\n        list_comp = list(nx.connected_components(G))\n        n = len(G.nodes)\n        map_bet = nx.edge_betweenness_centrality(G, normalized=True)\n        vals = []\n        for edge, val in map_bet.items():\n            if val > remove_edge_threshold:\n                list_remove_edges.append(edge)\n            vals.append(val)\n        return list_remove_edges, vals\n            \n                            \n    list_remove_edges, vals = split_graph(G)\n    return list_remove_edges, vals\n\nclass UnionFind():\n    \n    def __init__(self, N):\n        self.parent = [-1] * N\n        self.size = [1] * N\n        \n    def find(self, x):\n        p = self.parent[x]\n        if p == -1:\n            return x\n        p = self.find(p)\n        self.parent[x] = p\n        return p\n    \n    def unite(self, x, y):\n        px = self.find(x)\n        py = self.find(y)\n        if px == py:\n            return\n        if self.size[px] < self.size[py]:\n            px, py = py, px\n        self.size[px] += self.size[py]\n        self.parent[py] = px","metadata":{"papermill":{"duration":0.044926,"end_time":"2022-07-07T13:13:20.010218","exception":false,"start_time":"2022-07-07T13:13:19.965292","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T03:08:39.465162Z","iopub.execute_input":"2022-07-08T03:08:39.465606Z","iopub.status.idle":"2022-07-08T03:08:39.482094Z","shell.execute_reply.started":"2022-07-08T03:08:39.465538Z","shell.execute_reply":"2022-07-08T03:08:39.481063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport networkx as nx\nimport warnings\nfrom heapq import heappop, heappush\nfrom collections import defaultdict\nfrom tqdm.notebook import tqdm\nwarnings.simplefilter('ignore')\n\nmax_dist = 2\nmax_N = 256\nremove_edge_threshold = 0.405\nremove_edge_graphsize = 7\nshortest_paths_threshold = 1000\neval_points = list(range(len(test_df)))\n\nsubmission_df = test_df[[\"id\"]]\nsubmission_df[\"matches\"] = test_df[\"id\"]\n\n####################################\nuft = UnionFind(len(test_df))\n\nfor ix, nix in pred_df[[\"query_ix\", \"candidate_ix\"]].values:\n    uft.unite(ix, nix)\ngroup_members = defaultdict(list)\ngroup_size = defaultdict(int)\ngroup_map = {}\nfor i in range(len(test_df)):\n    group_members[uft.find(i)].append(i)\n    group_size[uft.find(i)] = uft.size[uft.find(i)]\n    group_map[i] = uft.find(i)\n\nlarge_groups = set([k for k, v in group_size.items() if v > min(max_N, max_dist+1)])\n\npred_df[\"group\"] = pred_df[\"query_ix\"].map(group_map)\npred_df[\"size\"] = pred_df[\"group\"].map(group_size)\npred_df[\"left\"] = np.minimum(pred_df[\"query_ix\"], pred_df[\"candidate_ix\"])\npred_df[\"right\"] = np.maximum(pred_df[\"query_ix\"], pred_df[\"candidate_ix\"])\n\nlist_remove_edges = []\nvals = []\nfor x, df in tqdm(pred_df.groupby('group')[['query_ix', 'candidate_ix']]):\n    if df.shape[0] <= remove_edge_graphsize:\n        continue\n    list_remove_edges_, vals_ = using_edge_betweenness_centrality(df[['query_ix', 'candidate_ix', 'pred']], remove_edge_threshold)\n    list_remove_edges += list_remove_edges_\n    vals += vals_\n    \nremove_pairs = []\nfor x in list_remove_edges:\n    if x in ixs2idx:\n        remove_pairs.append(ixs2idx[x])\n    x = (x[1], x[0])\n    if x in ixs2idx:\n        remove_pairs.append(ixs2idx[x])\nremove_pairs.sort()\n    \npred_df = pred_df.reset_index(drop=True)\n\nprint(pred_df.shape)\npred_df_ = pred_df[~pred_df.index.isin(remove_pairs)]\nprint(pred_df.shape)\n############################################\n\nuft = UnionFind(len(test_df))\n\nfor ix, nix in pred_df[[\"query_ix\", \"candidate_ix\"]].values:\n    uft.unite(ix, nix)\ngroup_members = defaultdict(list)\ngroup_size = defaultdict(int)\ngroup_map = {}\nfor i in range(len(test_df)):\n    group_members[uft.find(i)].append(i)\n    group_size[uft.find(i)] = uft.size[uft.find(i)]\n    group_map[i] = uft.find(i)\n\nlarge_groups = set([k for k, v in group_size.items() if v > min(max_N, max_dist+1)])\n\npred_df[\"group\"] = pred_df[\"query_ix\"].map(group_map)\npred_df[\"size\"] = pred_df[\"group\"].map(group_size)\npred_df[\"left\"] = np.minimum(pred_df[\"query_ix\"], pred_df[\"candidate_ix\"])\npred_df[\"right\"] = np.maximum(pred_df[\"query_ix\"], pred_df[\"candidate_ix\"])\n\npred_df = pred_df.sort_values(by=\"pred\", ascending=False).drop_duplicates([\"left\", \"right\"])\n\ngraphs = {k: nx.Graph() for k in large_groups}\nfor large_group in large_groups:\n    for member in group_members[large_group]:\n        graphs[large_group].add_node(member)\n\nneighbors = defaultdict(list)\nfor l, r, s in pred_df[pred_df[\"group\"].isin(large_groups)][[\"left\", \"right\", \"pred\"]].values:\n    l, r = int(l), int(r)\n    g = uft.find(l)\n    neighbors[l].append((r, s))\n    neighbors[r].append((l, s))\n    graphs[g].add_edge(l, r)\n\nshortest_paths = {g: {k: d for k, d in nx.all_pairs_shortest_path_length(graphs[g])} for g in large_groups if len(group_members[g]) < shortest_paths_threshold}\n\nmatches = []\nfor i in eval_points:\n    g = uft.find(i)\n    preds = []\n    if g in large_groups:\n        if g in shortest_paths.keys():\n            for n, d in shortest_paths[g][i].items():\n                if d <= max_dist:\n                    preds.append(n)\n        if (len(preds) > max_N) or (not g in shortest_paths.keys()):\n            searched = set()\n            heapq = [(-1., 0, i)]\n            while len(heapq) and (len(searched) < max_N):\n                _, step, x = heappop(heapq)\n                if x in searched:\n                    continue\n                searched.add(x)\n                if step >= max_dist:\n                    continue\n                for n, s in neighbors[x]:\n                    if n in searched:\n                        continue\n                    heappush(heapq, (-s, step+1, n))\n            preds = list(searched)\n    else:\n        preds = group_members[g]\n    matches.append(\" \".join([test_df.at[p, \"id\"] for p in preds]))\nsubmission_df[\"matches\"] = matches","metadata":{"papermill":{"duration":0.284687,"end_time":"2022-07-07T13:13:20.326292","exception":false,"start_time":"2022-07-07T13:13:20.041605","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T03:09:14.954741Z","iopub.execute_input":"2022-07-08T03:09:14.955111Z","iopub.status.idle":"2022-07-08T03:09:15.421850Z","shell.execute_reply.started":"2022-07-08T03:09:14.955072Z","shell.execute_reply":"2022-07-08T03:09:15.420730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if DEBUG:\n    pd.read_csv(\"../input/foursquare-location-matching/sample_submission.csv\").to_csv(\"submission.csv\", index=None, mode=\"w\")\nelse:\n    submission_df.to_csv(\"submission.csv\", index=None, mode=\"w\")","metadata":{"ExecuteTime":{"start_time":"2022-06-25T11:05:00.938Z"},"papermill":{"duration":0.043047,"end_time":"2022-07-07T13:13:20.401654","exception":false,"start_time":"2022-07-07T13:13:20.358607","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T03:09:18.900027Z","iopub.execute_input":"2022-07-08T03:09:18.900919Z","iopub.status.idle":"2022-07-08T03:09:18.922588Z","shell.execute_reply.started":"2022-07-08T03:09:18.900882Z","shell.execute_reply":"2022-07-08T03:09:18.921750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import Image\nImage(nadare_feature_dir + \"feature_importances_gain_pred34_mod1_fold5_epoch2000.png\")","metadata":{"papermill":{"duration":0.081325,"end_time":"2022-07-07T13:13:20.671014","exception":false,"start_time":"2022-07-07T13:13:20.589689","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T03:09:39.774380Z","iopub.execute_input":"2022-07-08T03:09:39.774895Z","iopub.status.idle":"2022-07-08T03:09:39.803909Z","shell.execute_reply.started":"2022-07-08T03:09:39.774848Z","shell.execute_reply":"2022-07-08T03:09:39.803078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import Image\nImage(nadare_feature_dir + \"feature_importances_split_pred34_mod1_fold5_epoch2000.png\")","metadata":{"papermill":{"duration":0.062263,"end_time":"2022-07-07T13:13:20.770599","exception":false,"start_time":"2022-07-07T13:13:20.708336","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-08T03:09:46.719183Z","iopub.execute_input":"2022-07-08T03:09:46.719661Z","iopub.status.idle":"2022-07-08T03:09:46.748461Z","shell.execute_reply.started":"2022-07-08T03:09:46.719609Z","shell.execute_reply":"2022-07-08T03:09:46.747541Z"},"trusted":true},"execution_count":null,"outputs":[]}]}