{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"DEBUG = False\nDEBUG_COUNTRY = [\"GB\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:27:52.285875Z","iopub.execute_input":"2022-07-03T05:27:52.286347Z","iopub.status.idle":"2022-07-03T05:27:52.310405Z","shell.execute_reply.started":"2022-07-03T05:27:52.286263Z","shell.execute_reply":"2022-07-03T05:27:52.309789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# =========================\n# Library\n# =========================\nimport pandas as pd\nimport numpy as np\nfrom tqdm.auto import tqdm\nimport os\nimport gc\nimport random\nfrom glob import glob\nimport warnings\nimport pickle\nimport json\nimport re\nimport time\nimport sys\nfrom requests import get\nimport multiprocessing\nimport joblib\nfrom joblib import Parallel, delayed\nimport Levenshtein\nimport difflib\nfrom contextlib import contextmanager\nfrom sklearn.neighbors import KNeighborsRegressor\nimport unicodedata\nfrom transformers import AutoModel,AutoTokenizer\nfrom cuml import ForestInference\nfrom cuml.neighbors import NearestNeighbors\n%env TOKENIZERS_PARALLELISM=true\nimport torch.nn.functional as F\nimport torch.nn as nn\nimport torch\nfrom torch.utils.data import DataLoader, Dataset\nfrom torch.cuda.amp import autocast\nfrom catboost import CatBoostClassifier\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:27:52.311938Z","iopub.execute_input":"2022-07-03T05:27:52.312399Z","iopub.status.idle":"2022-07-03T05:28:03.664848Z","shell.execute_reply.started":"2022-07-03T05:27:52.312362Z","shell.execute_reply":"2022-07-03T05:28:03.664132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# =========================\n# Constant\n# =========================\nTRAIN_PATH = \"../input/foursquare-fold/fold_train.csv\"\nTRAIN_RAW_PATH = \"../input/foursquare-location-matching/train.csv\"\nTEST_PATH = \"../input/foursquare-location-matching/test.csv\"\nSUB_PATH = \"../input/foursquare-location-matching/sample_submission.csv\"\nTARGET = \"point_of_interest\"","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:28:03.666591Z","iopub.execute_input":"2022-07-03T05:28:03.666911Z","iopub.status.idle":"2022-07-03T05:28:03.671532Z","shell.execute_reply.started":"2022-07-03T05:28:03.666876Z","shell.execute_reply":"2022-07-03T05:28:03.670719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# =========================\n# Settings\n# =========================\nn_neighbors_first_stage = 100\nsecond_stage_rank = 20\nthrid_stage_blocking = 0.01\nfourth_stage_blocking = 0.02\n\n# ================================\n# model\n# ================================\n# place\nfirst_stage_place_lgb_path = [\"../input/foursquare-ex073/lgb_fold0.txt\"]\n\n# name\nfirst_stage_name_lgb_path = [\"../input/foursquare-ex074/lgb_fold0.txt\"]\n\nsecond_stage_lgb_path = [\"../input/foursquare-ex075/lgb_fold0.txt\"]\nthird_stage_cat_path = [\"../input/foursquare-ex090/model\"]\n\n\n# ================================\n# fe \n# ================================\ncategory_encoder = \"045\"\nfe045_categories_path = f\"../input/foursquare-fe{category_encoder}/fe{category_encoder}_categories.pkl\"\nfe045_city_path = f\"../input/foursquare-fe{category_encoder}/fe{category_encoder}_city.pkl\"\nfe045_country_path = f\"../input/foursquare-fe{category_encoder}/fe{category_encoder}_country.pkl\"\nfe045_state_path = f\"../input/foursquare-fe{category_encoder}/fe{category_encoder}_state.pkl\"\nfe046_svd_path = f\"../input/foursquare-fe046/fe046_svd.pkl\"\n\nfirst_stage_place_features = ['latitude', 'longitude', 'rank', 'd_near', \n                             'near_latitude', 'near_longitude', 'name_jaro', \n                             'categories_jaro']\n\nfirst_stage_name_features = ['latitude', 'longitude', 'rank', 'd_near', 'near_latitude', 'near_longitude', 'name_jaro', 'distance']\n\n\nsecond_stage_features= ['latitude',\n 'longitude',\n 'rank',\n 'd_near',\n 'near_latitude',\n 'near_longitude',\n 'name_gesh',\n 'name_leven',\n 'name_jaro',\n 'address_gesh',\n 'address_leven',\n 'address_jaro',\n 'city_gesh',\n 'city_leven',\n 'city_jaro',\n 'state_gesh',\n 'state_leven',\n 'state_jaro',\n 'zip_gesh',\n 'zip_leven',\n 'zip_jaro',\n 'url_gesh',\n 'url_leven',\n 'url_jaro',\n 'phone_gesh',\n 'phone_leven',\n 'phone_jaro',\n 'categories_gesh',\n 'categories_leven',\n 'categories_jaro',\n 'distance',\n 'distance_rank',\n 'name_gesh_mean',\n 'name_gesh_max',\n 'near_name_gesh_mean',\n 'near_name_gesh_max',\n 'name_gesh_mean_rate',\n 'name_gesh_max_rate',\n 'near_name_gesh_mean_rate',\n 'near_name_gesh_max_rate',\n 'name_leven_mean',\n 'name_leven_min',\n 'near_name_leven_mean',\n 'near_name_leven_min',\n 'name_leven_mean_rate',\n 'name_leven_min_rate',\n 'near_name_leven_mean_rate',\n 'near_name_leven_min_rate',\n 'name_jaro_mean',\n 'name_jaro_max',\n 'near_name_jaro_mean',\n 'near_name_jaro_max',\n 'name_jaro_mean_rate',\n 'name_jaro_max_rate',\n 'near_name_jaro_mean_rate',\n 'near_name_jaro_max_rate',\n 'categories_gesh_mean',\n 'categories_gesh_max',\n 'near_categories_gesh_mean',\n 'near_categories_gesh_max',\n 'categories_gesh_mean_rate',\n 'categories_gesh_max_rate',\n 'near_categories_gesh_mean_rate',\n 'near_categories_gesh_max_rate',\n 'categories_leven_mean',\n 'categories_leven_min',\n 'near_categories_leven_mean',\n 'near_categories_leven_min',\n 'categories_leven_mean_rate',\n 'categories_leven_min_rate',\n 'near_categories_leven_mean_rate',\n 'near_categories_leven_min_rate',\n 'categories_jaro_mean',\n 'categories_jaro_max',\n 'near_categories_jaro_mean',\n 'near_categories_jaro_max',\n 'categories_jaro_mean_rate',\n 'categories_jaro_max_rate',\n 'near_categories_jaro_mean_rate',\n 'near_categories_jaro_max_rate',\n 'd_near_mean',\n 'd_near_min',\n 'near_d_near_mean',\n 'near_d_near_min',\n 'd_near_mean_rate',\n 'd_near_min_rate',\n 'near_d_near_mean_rate',\n 'near_d_near_min_rate',\n 'distance_mean',\n 'distance_min',\n 'near_distance_mean',\n 'near_distance_min',\n 'distance_mean_rate',\n 'distance_min_rate',\n 'near_distance_mean_rate',\n 'near_distance_min_rate',\n 'city_label',\n 'near_city_label',\n 'state_label',\n 'near_state_label',\n 'country_label',\n 'near_country_label',\n 'categories_label',\n 'near_categories_label',\n 'name_emb_svd0',\n 'name_emb_svd1',\n 'name_emb_svd2',\n 'name_emb_svd3',\n 'name_emb_svd4',\n 'name_emb_svd5',\n 'name_emb_svd6',\n 'name_emb_svd7',\n 'name_emb_svd8',\n 'name_emb_svd9',\n 'near_name_emb_svd0',\n 'near_name_emb_svd1',\n 'near_name_emb_svd2',\n 'near_name_emb_svd3',\n 'near_name_emb_svd4',\n 'near_name_emb_svd5',\n 'near_name_emb_svd6',\n 'near_name_emb_svd7',\n 'near_name_emb_svd8',\n 'near_name_emb_svd9']\n\n# ================================\n# bert\n# ================================\nbert_num_cols1 = [\"latitude\", \"longitude\", \"near_latitude\", \"near_longitude\",\n           \"latdiff\", \"londiff\", \"manhattan\", \"euclidean\", \"haversine\",\n           \"x\", \"y\", \"z\", \"near_x\", \"near_y\", \"near_z\", \"dot\",\n           'name_gesh', 'name_leven', 'name_jaro',\n           'address_gesh', 'address_leven', 'address_jaro', 'city_gesh',\n           'city_leven', 'city_jaro', 'state_gesh', 'state_leven', 'state_jaro',\n           'zip_gesh', 'zip_leven', 'zip_jaro', 'url_gesh', 'url_leven',\n           'url_jaro', 'phone_gesh', 'phone_leven','phone_jaro', 'categories_gesh', 'categories_leven', 'categories_jaro',\n           'distance', 'distance_rank', 'name_gesh_mean', 'name_gesh_max',\n           'near_name_gesh_mean', 'near_name_gesh_max', 'name_gesh_mean_rate',\n           'name_gesh_max_rate', 'near_name_gesh_mean_rate',\n           'near_name_gesh_max_rate', 'name_leven_mean', 'name_leven_min',\n           'near_name_leven_mean', 'near_name_leven_min', 'name_leven_mean_rate',\n           'name_leven_min_rate', 'near_name_leven_mean_rate',\n           'near_name_leven_min_rate', 'name_jaro_mean', 'name_jaro_max',\n           'near_name_jaro_mean', 'near_name_jaro_max', 'name_jaro_mean_rate',\n           'name_jaro_max_rate', 'near_name_jaro_mean_rate',\n           'near_name_jaro_max_rate', 'categories_gesh_mean',\n           'categories_gesh_max', 'near_categories_gesh_mean',\n           'near_categories_gesh_max', 'categories_gesh_mean_rate',\n           'categories_gesh_max_rate', 'near_categories_gesh_mean_rate',\n           'near_categories_gesh_max_rate', 'categories_leven_mean',\n           'categories_leven_min', 'near_categories_leven_mean',\n           'near_categories_leven_min', 'categories_leven_mean_rate',\n           'categories_leven_min_rate', 'near_categories_leven_mean_rate',\n           'near_categories_leven_min_rate', 'categories_jaro_mean',\n           'categories_jaro_max', 'near_categories_jaro_mean',\n           'near_categories_jaro_max','categories_jaro_mean_rate', 'categories_jaro_max_rate',\n           'near_categories_jaro_mean_rate', 'near_categories_jaro_max_rate']\n\n\nbert_num_cols2 = [\"latitude\",\"longitude\",'name_gesh', 'name_leven', 'name_jaro',\n       'address_gesh', 'address_leven', 'address_jaro', 'city_gesh',\n       'city_leven', 'city_jaro', 'state_gesh', 'state_leven', 'state_jaro',\n       'zip_gesh', 'zip_leven', 'zip_jaro', 'url_gesh', 'url_leven',\n       'url_jaro', 'phone_gesh', 'phone_leven','phone_jaro', 'categories_gesh', 'categories_leven', 'categories_jaro',\n       'distance', 'distance_rank', 'name_gesh_mean', 'name_gesh_max',\n       'near_name_gesh_mean', 'near_name_gesh_max', 'name_gesh_mean_rate',\n       'name_gesh_max_rate', 'near_name_gesh_mean_rate',\n       'near_name_gesh_max_rate', 'name_leven_mean', 'name_leven_min',\n       'near_name_leven_mean', 'near_name_leven_min', 'name_leven_mean_rate',\n       'name_leven_min_rate', 'near_name_leven_mean_rate',\n       'near_name_leven_min_rate', 'name_jaro_mean', 'name_jaro_max',\n       'near_name_jaro_mean', 'near_name_jaro_max', 'name_jaro_mean_rate',\n       'name_jaro_max_rate', 'near_name_jaro_mean_rate',\n       'near_name_jaro_max_rate', 'categories_gesh_mean',\n       'categories_gesh_max', 'near_categories_gesh_mean',\n       'near_categories_gesh_max', 'categories_gesh_mean_rate',\n       'categories_gesh_max_rate', 'near_categories_gesh_mean_rate',\n       'near_categories_gesh_max_rate', 'categories_leven_mean',\n       'categories_leven_min', 'near_categories_leven_mean',\n       'near_categories_leven_min', 'categories_leven_mean_rate',\n       'categories_leven_min_rate', 'near_categories_leven_mean_rate',\n       'near_categories_leven_min_rate', 'categories_jaro_mean',\n       'categories_jaro_max', 'near_categories_jaro_mean',\n       'near_categories_jaro_max','categories_jaro_mean_rate', 'categories_jaro_max_rate',\n       'near_categories_jaro_mean_rate', 'near_categories_jaro_max_rate']\n\nsc_dict = {'latitude': [26.87459868745177, 23.144740576788625],\n 'longitude': [20.70497351331466, 82.6778436146614],\n 'near_latitude': [22.377329, 23.80125],\n 'near_longitude': [47.604324, 72.81156],\n 'latdiff': [0.0026245795, 0.60088676],\n 'londiff': [-0.004257245, 1.7946571],\n 'manhattan': [0.21769507, 2.1573222],\n 'euclidean': [0.17776342, 1.8842192],\n 'haversine': [0.002634386, 0.023196151],\n 'x': [0, 1],\n 'y': [0, 1],\n 'z': [0, 1],\n 'near_x': [0, 1],\n 'near_y': [0, 1],\n 'near_z': [0, 1],\n 'dot': [0, 1],\n 'name_gesh': [0.535247, 0.28312334],\n 'name_leven': [12.289453, 8.717725],\n 'name_jaro': [0.6814999, 0.278118],\n 'address_gesh': [0.55750847, 0.32904968],\n 'address_leven': [11.519652, 11.330601],\n 'address_jaro': [0.68748957, 0.28175715],\n 'city_gesh': [0.78845024, 0.33762273],\n 'city_leven': [2.639476, 4.342476],\n 'city_jaro': [0.8420279, 0.29122102],\n 'state_gesh': [0.7989922, 0.3356451],\n 'state_leven': [2.618497, 4.6914506],\n 'state_jaro': [0.8407704, 0.2916656],\n 'zip_gesh': [0.90403444, 0.18740492],\n 'zip_leven': [0.6094352, 1.2436553],\n 'zip_jaro': [0.9485774, 0.11972204],\n 'url_gesh': [0.8163289, 0.22502218],\n 'url_leven': [15.599948, 23.641768],\n 'url_jaro': [0.9568382, 0.07457281],\n 'phone_gesh': [0.7717348, 0.24688902],\n 'phone_leven': [3.4015062, 3.4060066],\n 'phone_jaro': [0.85937643, 0.16305733],\n 'categories_gesh': [0.5930225, 0.32405168],\n 'categories_leven': [10.481613, 10.402995],\n 'categories_jaro': [0.72280174, 0.2595957],\n 'distance': [3.5818813, 170.23347],\n 'distance_rank': [11.811793, 10.082427],\n 'name_gesh_mean': [0.40428534, 0.14291741],\n 'name_gesh_max': [0.8063714, 0.17690912],\n 'near_name_gesh_mean': [0.40961915, 0.14795418],\n 'near_name_gesh_max': [0.8105544, 0.17704241],\n 'name_gesh_mean_rate': [1.3989464, 1.0484291],\n 'name_gesh_max_rate': [0.6530004, 0.29792878],\n 'near_name_gesh_mean_rate': [1.3906603, 1.0571884],\n 'near_name_gesh_max_rate': [0.65022796, 0.298543],\n 'name_leven_mean': [14.080113, 5.990362],\n 'name_leven_min': [5.0102377, 5.371386],\n 'near_name_leven_mean': [13.955088, 6.017419],\n 'near_name_leven_min': [4.898045, 5.3486223],\n 'name_leven_mean_rate': [0.8712333, 0.6081466],\n 'name_leven_min_rate': [3.0417712, 3.5432434],\n 'near_name_leven_mean_rate': [0.88712335, 0.6753299],\n 'near_name_leven_min_rate': [3.103235, 3.6311984],\n 'name_jaro_mean': [0.59393793, 0.14204046],\n 'name_jaro_max': [0.922182, 0.113545366],\n 'near_name_jaro_mean': [0.5990131, 0.14566767],\n 'near_name_jaro_max': [0.9246227, 0.112311274],\n 'name_jaro_mean_rate': [1.17415, 0.75928885],\n 'name_jaro_max_rate': [0.73229, 0.27749673],\n 'near_name_jaro_mean_rate': [1.1688758, 0.76395464],\n 'near_name_jaro_max_rate': [0.73066276, 0.2782531],\n 'categories_gesh_mean': [0.45021054, 0.16041645],\n 'categories_gesh_max': [0.8934652, 0.18464169],\n 'near_categories_gesh_mean': [0.45295003, 0.1621628],\n 'near_categories_gesh_max': [0.89459765, 0.18372972],\n 'categories_gesh_mean_rate': [1.3444637, 0.71897376],\n 'categories_gesh_max_rate': [0.65741974, 0.31307998],\n 'near_categories_gesh_mean_rate': [1.3444862, 0.7269207],\n 'near_categories_gesh_max_rate': [0.65774196, 0.31624898],\n 'categories_leven_mean': [13.444308, 6.6503487],\n 'categories_leven_min': [2.9128094, 5.704073],\n 'near_categories_leven_mean': [13.410719, 6.695578],\n 'near_categories_leven_min': [2.9045722, 5.7120137],\n 'categories_leven_mean_rate': [0.7467077, 0.7704559],\n 'categories_leven_min_rate': [2.024397, 1.5368462],\n 'near_categories_leven_mean_rate': [0.75811625, 0.9181195],\n 'near_categories_leven_min_rate': [2.0135462, 1.5323894],\n 'categories_jaro_mean': [0.6192549, 0.1293145],\n 'categories_jaro_max': [0.94835114, 0.116735004],\n 'near_categories_jaro_mean': [0.6217563, 0.13072947],\n 'near_categories_jaro_max': [0.94931644, 0.11579985],\n 'categories_jaro_mean_rate': [1.1727746, 0.40448013],\n 'categories_jaro_max_rate': [0.7595728, 0.24773462],\n 'near_categories_jaro_mean_rate': [1.1711832, 0.4081076],\n 'near_categories_jaro_max_rate': [0.75936586, 0.24919231]}\n\nMAX_LEN = 32\nBS = 64\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nBERT_MODEL = \"../input/bert-base-multilingual/bert-base-multilingual-uncased\"\n\n# ================================\n# third bert\n# ================================\nTHIRD_MAX_LEN = 128\nTHIRD_BS = 32\nTHIRD_BERT_MODEL1 = \"../input/xlm-roberta-large/xlm-roberta-large\"\n# third stage\nthird_stage_model1_path = \"../input/foursquare-ex104/ex104_2.pth\"\n\n# ================================\n# third bert2\n# ================================\nTHIRD_MAX_LEN2 = 128\nTHIRD_BS2 = 48\nTHIRD_BERT_MODEL2 = \"../input/mdeberta-base/mdeberta-v3-base\"\n# third stage\nthird_stage_model2_path = \"../input/foursquare-ex115/ex115_3_ema.pth\"\n\n# ================================\n# ensemble\n# ================================\nw1 = 0.013225176172449034\nw2 = 0.28751060139700985\nw3 = 0.3813813593361608\nw4 = 0.31788286309438024\n\n# ================================\n# fourth bert\n# ================================\nFOURTH_MAX_LEN = 128\nFOURTH_BS = 32\nFOURTH_BERT_MODEL1 = \"../input/xlm-roberta-large/xlm-roberta-large\"\n# fourth stage\nfourth_stage_model1_path = \"../input/foursquare-ex101/ex101_2.pth\"","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:28:03.673193Z","iopub.execute_input":"2022-07-03T05:28:03.67366Z","iopub.status.idle":"2022-07-03T05:28:03.718739Z","shell.execute_reply.started":"2022-07-03T05:28:03.673623Z","shell.execute_reply":"2022-07-03T05:28:03.71801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# colsのdictへの変換\nfirst_stage_place_cols2num_dict = {}\nfor n,c in enumerate(first_stage_place_features):\n    first_stage_place_cols2num_dict[c] = n\n    \nfirst_stage_name_cols2num_dict = {}\nfor n,c in enumerate(first_stage_name_features):\n    first_stage_name_cols2num_dict[c] = n\n    \nsecond_stage_cols2num_dict = {}\nfor n,c in enumerate(second_stage_features):\n    second_stage_cols2num_dict[c] = n","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:31:48.144997Z","iopub.execute_input":"2022-07-03T05:31:48.145303Z","iopub.status.idle":"2022-07-03T05:31:48.154012Z","shell.execute_reply.started":"2022-07-03T05:31:48.14527Z","shell.execute_reply":"2022-07-03T05:31:48.153113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ================================\n# functions utils\n# ================================\ndef reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.\n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024 ** 2\n    #print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    #print(\"column = \", len(df.columns))\n    for i, col in enumerate(df.columns):\n       #if i % 50 == 0:\n       #     print(i)\n        try:\n            col_type = df[col].dtype\n\n            if col_type != object:\n                c_min = df[col].min()\n                c_max = df[col].max()\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int32)\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float32)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        df[col] = df[col].astype(np.float32)\n        except:\n            continue\n\n    end_mem = df.memory_usage().sum() / 1024 ** 2\n    #print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    #print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n\n    return df\n\n\ndef join(df):\n    x = [str(e) for e in list(df)]\n    return \" \".join(x)\n\n# https://www.kaggle.com/code/columbia2131/foursquare-iou-metrics\ndef get_id2poi(input_df: pd.DataFrame) -> dict:\n    return dict(zip(input_df['id'], input_df['point_of_interest']))\n\ndef get_poi2ids(input_df: pd.DataFrame) -> dict:\n    return input_df.groupby('point_of_interest')['id'].apply(set).to_dict()\n\ndef get_score(input_df: pd.DataFrame):\n    scores = []\n    for id_str, matches in zip(input_df['id'].to_numpy(), input_df['matches'].to_numpy()):\n        targets = poi2ids[id2poi[id_str]]\n        preds = set(matches.split())\n        score = len((targets & preds)) / len((targets | preds))\n        scores.append(score)\n    scores = np.array(scores)\n    return scores.mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:31:48.868107Z","iopub.execute_input":"2022-07-03T05:31:48.868409Z","iopub.status.idle":"2022-07-03T05:31:48.896747Z","shell.execute_reply.started":"2022-07-03T05:31:48.868374Z","shell.execute_reply":"2022-07-03T05:31:48.895853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ================================\n# functions first stage\n# ================================\ndef make_candidate_first_stage(country_df, n_neighbors,columns):\n    concat_df = []\n    knn = KNeighborsRegressor(n_neighbors=min(len(country_df), n_neighbors), \n                              metric=\"euclidean\", n_jobs=-1)\n    knn.fit(country_df[['latitude','longitude']], country_df.index)\n    dists, nears = knn.kneighbors(country_df[['latitude','longitude']], return_distance=True)\n\n    for i in range(min(len(country_df), n_neighbors)):\n        country_df_ = country_df[columns].copy()\n        country_df_[\"rank\"] = i\n        country_df_[\"d_near\"] = dists[:, i]\n        for c in columns:\n            country_df_[f\"near_{c}\"] = country_df_[c].values[nears[:, i]]\n        concat_df.append(country_df_)\n    concat_df = pd.concat(concat_df).reset_index(drop=True)\n    return concat_df\n\n\ndef delete_match_id(concat_df):\n    concat_df[\"id_match\"] = concat_df[\"id\"] == concat_df[\"near_id\"]\n    concat_df[\"id_match\"] = concat_df[\"id_match\"].astype(int)\n    concat_df = concat_df[concat_df[\"id_match\"] == 0].reset_index(drop=True)\n    concat_df = reduce_mem_usage(concat_df)\n    return concat_df\n\n\ndef df2numpy(concat_npy,concat_df,columns,cols2num_dict):\n    for c in columns:\n        concat_npy[:,cols2num_dict[c]] = concat_df[c].values.astype(np.float32)\n        concat_df = concat_df.drop(columns = c)\n    return concat_npy, concat_df\n\ndef text_preprocess(text):\n    text = str(text)\n    text = text.replace(\" \",\"\")\n    text = text.lower()\n    text = unicodedata.normalize(\"NFKC\",text)\n    return text\n\n# 特徴量エンジニアリング\ndef calc_distance_first_stage(c1_array,c2_array):\n    distance = np.zeros(len(c1_array),dtype=np.float32)\n    for n,(c1,c2) in enumerate(zip(c1_array,c2_array)):\n        c1 = text_preprocess(c1)\n        c2 = text_preprocess(c2)\n        if (str(c1) != \"nan\") and (str(c2) != \"nan\"):\n            distance[n] = Levenshtein.jaro_winkler(str(c1), str(c2))\n        else:\n            distance[n] = np.nan\n    return distance.reshape([-1,1])\n\n# 予測値作成\ndef make_pred(npy,model_list):\n    len_npy = len(npy)\n    batch = 50000\n    if len_npy > batch:\n        pred = np.zeros(len_npy)\n        all_batch = int(len_npy //  batch + 1)\n        for n,m in enumerate(model_list):\n            for i in range(all_batch):\n                if i < all_batch - 1:\n                    pred[i *  batch : (i + 1) * batch] += m.predict(\n                        npy[i *  batch : (i + 1) * batch].astype(np.float32)) / len(model_list)\n                else:\n                    pred[i *  batch : ] += m.predict(\n                        npy[i *  batch : ].astype(np.float32)) / len(model_list)\n    else:         \n        for n,m in enumerate(model_list):\n            if n == 0:\n                pred = m.predict(npy.astype(np.float32)) / len(model_list)\n            else:\n                pred += m.predict(npy.astype(np.float32)) / len(model_list)\n    return pred\n\n\ndef remove_low_rank_place(concat_df, concat_npy, second_stage_rank, first_stage_place_cols2num_dict):\n    concat_df[\"id\"] = concat_df[\"id\"].astype(\"category\")\n    concat_df[\"pred_rank\"] = concat_df.groupby(by=\"id\")[\"pred\"].rank(ascending=False)\n    concat_npy = concat_npy[concat_df[\"pred_rank\"] <= second_stage_rank]\n    concat_df = concat_df[concat_df[\"pred_rank\"] <= second_stage_rank].reset_index(drop=True)\n    for c in [\"longitude\",\"near_longitude\",\"latitude\",\"near_latitude\"]:\n        concat_df[c] = concat_npy[:,first_stage_place_cols2num_dict[c]]\n    return concat_df, concat_npy","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:31:49.091416Z","iopub.execute_input":"2022-07-03T05:31:49.091663Z","iopub.status.idle":"2022-07-03T05:31:49.112172Z","shell.execute_reply.started":"2022-07-03T05:31:49.091636Z","shell.execute_reply":"2022-07-03T05:31:49.111301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ================================\n# functions first stage name\n# ================================\ndef make_candidate_name_first_stage(country_df, country_name_emb, n_neighbors,columns):\n    concat_df = []\n    knn = NearestNeighbors(n_neighbors=min(len(country_df), n_neighbors),metric=\"cosine\")\n    knn.fit(country_name_emb)\n    dists, nears = knn.kneighbors(country_name_emb)\n    del knn\n    for i in range(min(len(country_df), n_neighbors)):\n        country_df_ = country_df[columns].copy()\n        country_df_[\"rank\"] = i\n        country_df_[\"d_near\"] = dists[:, i]\n        for c in columns:\n            country_df_[f\"near_{c}\"] = country_df_[c].values[nears[:, i]]\n        concat_df.append(country_df_)\n    concat_df = pd.concat(concat_df).reset_index(drop=True)\n    return concat_df\n\ndef make_place_distance(concat_npy,first_stage_name_cols2num_dict):\n    c = \"distance\"\n    c1 = \"latitude\"\n    c2 = \"near_latitude\"\n    c3 = \"longitude\"\n    c4 = \"near_longitude\"\n    concat_npy[:,first_stage_name_cols2num_dict[c]] = (concat_npy[:,first_stage_name_cols2num_dict[c1]] - concat_npy[:,first_stage_name_cols2num_dict[c2]])**2 + \\\n    (concat_npy[:,first_stage_name_cols2num_dict[c3]] - concat_npy[:,first_stage_name_cols2num_dict[c4]])**2\n    return concat_npy\n\ndef remove_low_rank_name(concat_df, concat_npy, second_stage_rank, first_stage_name_cols2num_dict):\n    concat_df[\"id\"] = concat_df[\"id\"].astype(\"category\")\n    concat_df[\"pred_rank\"] = concat_df.groupby(by=\"id\")[\"pred\"].rank(ascending=False)\n    concat_npy = concat_npy[concat_df[\"pred_rank\"] <= second_stage_rank]\n    concat_df = concat_df[concat_df[\"pred_rank\"] <= second_stage_rank].reset_index(drop=True)\n    for c in [\"longitude\",\"near_longitude\",\"latitude\",\"near_latitude\",\"rank\",\"d_near\"]:\n        concat_df[c] = concat_npy[:,first_stage_name_cols2num_dict[c]]\n    return concat_df, concat_npy","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:31:49.247219Z","iopub.execute_input":"2022-07-03T05:31:49.247696Z","iopub.status.idle":"2022-07-03T05:31:49.258804Z","shell.execute_reply.started":"2022-07-03T05:31:49.247661Z","shell.execute_reply":"2022-07-03T05:31:49.258135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## ================================\n# functions second stage\n# ================================\ndef move_features(concat_npy2, concat_df, move_features):\n    for c in move_features:\n        concat_npy2[:,second_stage_cols2num_dict[c]] = concat_df[c].values.astype(np.float32)\n    concat_df.drop(columns = move_features,inplace=True)\n    return concat_npy2, concat_df\n        \ndef merge_raw_data(concat_df, country_df,use_cols):\n    country_df_ = country_df[use_cols].copy()\n    country_df_[\"id\"] = country_df_[\"id\"].astype(\"category\")\n    concat_df = concat_df.merge(country_df_,how=\"left\",on=\"id\")\n    country_df_.columns = [f\"near_{i}\" for i in country_df_.columns]\n    concat_df = concat_df.merge(country_df_,how=\"left\",on=\"near_id\")\n    del country_df_\n    gc.collect()\n    return concat_df\n\ndef calc_distance_second_stage(c1_array,c2_array,col):\n    distance = np.zeros([len(c1_array),3],dtype=np.float32)\n    for n,(c1,c2) in enumerate(zip(c1_array,c2_array)):\n        c1 = text_preprocess(c1)\n        c2 = text_preprocess(c2)\n        if (str(c1) != \"nan\") and (str(c2) != \"nan\"):\n            distance[n,:] = np.array([difflib.SequenceMatcher(None, str(c1), str(c2)).ratio(),\n                    Levenshtein.distance(str(c1), str(c2)),\n                    Levenshtein.jaro_winkler(str(c1), str(c2))])\n        else:\n            distance[n,:] = np.array([np.nan,np.nan,np.nan])\n    return distance\n\n        \ndef make_distance_second_stage(train,concat_npy,distance_columns,second_stage_cols2num_dict):\n    for c in distance_columns:\n        distance = calc_distance_second_stage(train[c].values,train[f\"near_{c}\"].values,c)\n        for n,f_c in enumerate([f\"{c}_gesh\",f\"{c}_leven\",f\"{c}_jaro\"]):\n            concat_npy[:,second_stage_cols2num_dict[f_c]] = distance[:,n].astype(np.float32)\n        if c not in [\"categories\",\"city\",\"country\",\"state\"]:\n            train = train.drop(columns = [c,f\"near_{c}\"])\n        del distance\n        gc.collect()\n    # 位置の距離\n    c = \"distance\"\n    c1 = \"latitude\"\n    c2 = \"near_latitude\"\n    c3 = \"longitude\"\n    c4 = \"near_longitude\"\n    concat_npy[:,second_stage_cols2num_dict[c]] = \\\n    (concat_npy[:,second_stage_cols2num_dict[c1]] - concat_npy[:,second_stage_cols2num_dict[c2]])**2 \\\n    + (concat_npy[:,second_stage_cols2num_dict[c3]] - concat_npy[:,second_stage_cols2num_dict[c4]])**2\n    return train, concat_npy\n    \n\ndef distance_agg(concat_npy,train,cols2num_dict):\n    for c in [\"name\",\"categories\"]:\n        for d in [\"gesh\",\"leven\",\"jaro\"]:\n            train[f\"{c}_{d}\"] = concat_npy[:,cols2num_dict[f\"{c}_{d}\"]]\n            if d == \"leven\":\n                tmp_mean = train.groupby(by=\"id\")[f\"{c}_{d}\"].mean().to_dict()\n                tmp_min = train.groupby(by=\"id\")[f\"{c}_{d}\"].min().to_dict()\n                concat_npy[:,cols2num_dict[f\"{c}_{d}_mean\"]] = train[\"id\"].map(tmp_mean)\n                concat_npy[:,cols2num_dict[f\"{c}_{d}_min\"]] = train[\"id\"].map(tmp_min)\n                concat_npy[:,cols2num_dict[f\"near_{c}_{d}_mean\"]] = train[\"near_id\"].map(tmp_mean)\n                concat_npy[:,cols2num_dict[f\"near_{c}_{d}_min\"]] = train[\"near_id\"].map(tmp_min)\n                concat_npy[:,cols2num_dict[f\"{c}_{d}_mean_rate\"]] = concat_npy[:,cols2num_dict[f\"{c}_{d}\"]] / concat_npy[:,cols2num_dict[f\"{c}_{d}_mean\"]]\n                concat_npy[:,cols2num_dict[f\"{c}_{d}_min_rate\"]] = concat_npy[:,cols2num_dict[f\"{c}_{d}\"]] / concat_npy[:,cols2num_dict[f\"{c}_{d}_min\"]]\n                concat_npy[:,cols2num_dict[f\"near_{c}_{d}_mean_rate\"]] = concat_npy[:,cols2num_dict[f\"{c}_{d}\"]] / concat_npy[:,cols2num_dict[f\"near_{c}_{d}_mean\"]]\n                concat_npy[:,cols2num_dict[f\"near_{c}_{d}_min_rate\"]] = concat_npy[:,cols2num_dict[f\"{c}_{d}\"]] / concat_npy[:,cols2num_dict[f\"near_{c}_{d}_min\"]]\n            else:\n                tmp_mean = train.groupby(by=\"id\")[f\"{c}_{d}\"].mean().to_dict()\n                tmp_max = train.groupby(by=\"id\")[f\"{c}_{d}\"].max().to_dict()\n                concat_npy[:,cols2num_dict[f\"{c}_{d}_mean\"]] = train[\"id\"].map(tmp_mean)\n                concat_npy[:,cols2num_dict[f\"{c}_{d}_max\"]] = train[\"id\"].map(tmp_max)\n                concat_npy[:,cols2num_dict[f\"near_{c}_{d}_mean\"]] = train[\"near_id\"].map(tmp_mean)\n                concat_npy[:,cols2num_dict[f\"near_{c}_{d}_max\"]] = train[\"near_id\"].map(tmp_max)\n                concat_npy[:,cols2num_dict[f\"{c}_{d}_mean_rate\"]] = concat_npy[:,cols2num_dict[f\"{c}_{d}\"]] / concat_npy[:,cols2num_dict[f\"{c}_{d}_mean\"]]\n                concat_npy[:,cols2num_dict[f\"{c}_{d}_max_rate\"]] = concat_npy[:,cols2num_dict[f\"{c}_{d}\"]] / concat_npy[:,cols2num_dict[f\"{c}_{d}_max\"]]\n                concat_npy[:,cols2num_dict[f\"near_{c}_{d}_mean_rate\"]] = concat_npy[:,cols2num_dict[f\"{c}_{d}\"]] / concat_npy[:,cols2num_dict[f\"near_{c}_{d}_mean\"]]\n                concat_npy[:,cols2num_dict[f\"near_{c}_{d}_max_rate\"]] = concat_npy[:,cols2num_dict[f\"{c}_{d}\"]] / concat_npy[:,cols2num_dict[f\"near_{c}_{d}_max\"]]\n            train = train.drop(f\"{c}_{d}\",axis=1)\n\n    for c in [\"d_near\",\"distance\"]:\n        train[f\"{c}\"] = concat_npy[:,cols2num_dict[f\"{c}\"]]\n        tmp_mean = train.groupby(by=\"id\")[c].mean().to_dict()\n        tmp_min = train.groupby(by=\"id\")[c].min().to_dict()\n        concat_npy[:,cols2num_dict[f\"{c}_mean\"]] = train[\"id\"].map(tmp_mean)\n        concat_npy[:,cols2num_dict[f\"{c}_min\"]] = train[\"id\"].map(tmp_min)\n        concat_npy[:,cols2num_dict[f\"near_{c}_mean\"]] = train[\"near_id\"].map(tmp_mean)\n        concat_npy[:,cols2num_dict[f\"near_{c}_min\"]] = train[\"near_id\"].map(tmp_min)\n\n        concat_npy[:,cols2num_dict[f\"{c}_mean_rate\"]] = concat_npy[:,cols2num_dict[f\"{c}\"]] / concat_npy[:,cols2num_dict[f\"{c}_mean\"]]\n        concat_npy[:,cols2num_dict[f\"{c}_min_rate\"]] = concat_npy[:,cols2num_dict[f\"{c}\"]] / concat_npy[:,cols2num_dict[f\"{c}_min\"]]\n        concat_npy[:,cols2num_dict[f\"near_{c}_mean_rate\"]] = concat_npy[:,cols2num_dict[f\"{c}\"]] / concat_npy[:,cols2num_dict[f\"near_{c}_mean\"]]\n        concat_npy[:,cols2num_dict[f\"near_{c}_min_rate\"]] = concat_npy[:,cols2num_dict[f\"{c}\"]] / concat_npy[:,cols2num_dict[f\"near_{c}_min\"]]\n        if c == \"distance\":\n            concat_npy[:,cols2num_dict[f\"distance_rank\"]] = train.groupby(by=\"id\")[c].rank()\n    return concat_npy\n\ndef make_cat_features(concat_npy,concat_df,cat_dict_list,cols2num_dict):\n    for n,c in enumerate([\"categories\",\"city\",\"country\",\"state\"]):\n        concat_npy[:,cols2num_dict[f\"{c}_label\"]] = concat_df[c].map(cat_dict_list[n])\n        concat_npy[:,cols2num_dict[f\"near_{c}_label\"]] = concat_df[f\"near_{c}\"].map(cat_dict_list[n])\n        concat_df = concat_df.drop(columns = [c,f\"near_{c}\"])\n        gc.collect()\n    return concat_npy,concat_df\n\n\ndef concat_name_emb(concat_npy,concat_df,id2num_dict,bert_emb,cols2num_dict):\n    # nameのemb\n    name_svd = np.zeros([len(concat_npy),10],np.float32)\n    near_name_svd = np.zeros([len(concat_npy),10],np.float32)\n    for n,i in enumerate(concat_df[\"id\"].values):\n        name_svd[n,] = bert_emb[id2num_dict[i]]\n    for n,i in enumerate(concat_df[\"near_id\"].values):\n        near_name_svd[n,] = bert_emb[id2num_dict[i]]\n    concat_npy[:,cols2num_dict['name_emb_svd0']:cols2num_dict['name_emb_svd9'] + 1] = name_svd\n    del name_svd\n    concat_npy[:,cols2num_dict['near_name_emb_svd0']:cols2num_dict['near_name_emb_svd9'] + 1] = near_name_svd\n    del near_name_svd\n    return concat_npy\n\n\n    \n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:31:49.402496Z","iopub.execute_input":"2022-07-03T05:31:49.402933Z","iopub.status.idle":"2022-07-03T05:31:49.443623Z","shell.execute_reply.started":"2022-07-03T05:31:49.402901Z","shell.execute_reply":"2022-07-03T05:31:49.442875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ================================\n# functions bert\n# ================================\ndef text_preprocess_bert(text):\n    text = str(text)\n    text = text.lower()\n    return text\n\n\nclass BertDataset(Dataset):\n    def __init__(self, text, tokenizer, max_len,preprocess=None):\n        self.text = text\n        self.tokenizer = tokenizer\n        self.max_len = max_len\n        self.preprocess = preprocess\n\n    def __len__(self):\n        return len(self.text)\n\n    def __getitem__(self, item):\n        if self.preprocess:\n            text = text_preprocess_bert(self.text[item])\n        else:\n            text = str(self.text[item])\n        inputs = self.tokenizer(\n            text,\n            max_length=self.max_len,\n            padding=\"max_length\",\n            truncation=True,\n            return_attention_mask=True,\n            return_token_type_ids=True\n        )\n        ids = inputs[\"input_ids\"]\n        mask = inputs[\"attention_mask\"]\n        token_type_ids = inputs[\"token_type_ids\"]\n        \n        return {\n            \"input_ids\": torch.tensor(ids, dtype=torch.long),\n            \"attention_mask\": torch.tensor(mask, dtype=torch.long),\n            \"token_type_ids\": torch.tensor(token_type_ids, dtype=torch.long)\n        }\n    \nclass bert_model(nn.Module):\n    def __init__(self):\n        super(bert_model, self).__init__()\n        self.model = AutoModel.from_pretrained(BERT_MODEL)\n\n    def forward(self, ids, mask):\n        # pooler\n        bert_out = self.model(ids, attention_mask=mask)[0]\n        x = F.normalize((bert_out[:, 1:, :]*mask[:, 1:, None]).mean(axis=1))\n        return x\n    \ndef make_emb(model,train_loader,svd=None):\n    bert_emb = []\n    with torch.no_grad():\n        for d in tqdm(train_loader,total=len(train_loader)):\n            input_ids = d['input_ids']\n            mask = d['attention_mask']\n            token_type_ids = d[\"token_type_ids\"]\n            input_ids = input_ids.to(device)\n            mask = mask.to(device)\n            output = model(input_ids, mask)\n            output = output.detach().cpu().numpy().astype(np.float32)\n            if svd is not None:\n                output = svd.transform(output)\n            bert_emb.append(output)\n    torch.cuda.empty_cache()\n    bert_emb = np.concatenate(bert_emb)\n    return bert_emb","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:31:49.557036Z","iopub.execute_input":"2022-07-03T05:31:49.557536Z","iopub.status.idle":"2022-07-03T05:31:49.571584Z","shell.execute_reply.started":"2022-07-03T05:31:49.557501Z","shell.execute_reply":"2022-07-03T05:31:49.570844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ===============================\n# third stage\n# ===============================\n# ===============================================================================\n# Get manhattan distance\n# ===============================================================================\ndef manhattan(lat1, long1, lat2, long2):\n    return np.abs(lat2 - lat1) + np.abs(long2 - long1)\n\n# ===============================================================================\n# Get haversine distance\n# ===============================================================================\ndef vectorized_haversine(lats1, lats2, longs1, longs2):\n    # radius = 6371\n    radius = 1\n    dlat=np.radians(lats2 - lats1)\n    dlon=np.radians(longs2 - longs1)\n    a = np.sin(dlat/2) * np.sin(dlat/2) + np.cos(np.radians(lats1)) \\\n        * np.cos(np.radians(lats2)) * np.sin(dlon/2) * np.sin(dlon/2)\n    c = 2 * np.arctan2(np.sqrt(a), np.sqrt(1-a))\n    d = radius * c\n    return d\n\ndef add_lat_lon_distance_features(df):\n    lat1 = df['latitude']\n    lat2 = df['near_latitude']\n    lon1 = df['longitude']\n    lon2 = df['near_longitude']\n    df['latdiff'] = (lat1 - lat2)\n    df['londiff'] = (lon1 - lon2)\n    df['manhattan'] = manhattan(lat1, lon1, lat2, lon2)\n    df['euclidean'] = (df['latdiff'] ** 2 + df['londiff'] ** 2) ** 0.5\n    df['haversine'] = vectorized_haversine(lat1, lat2, lon1, lon2)\n    df[\"x\"] = np.cos(np.radians(df[\"latitude\"]))*np.cos(np.radians(df[\"longitude\"]))\n    df[\"y\"] = np.sin(np.radians(df[\"latitude\"]))*np.cos(np.radians(df[\"longitude\"]))\n    df[\"z\"] = np.sin(np.radians(df[\"longitude\"]))\n    df[\"near_x\"] = np.cos(np.radians(df[\"near_latitude\"]))*np.cos(np.radians(df[\"near_longitude\"]))\n    df[\"near_y\"] = np.sin(np.radians(df[\"near_latitude\"]))*np.cos(np.radians(df[\"near_longitude\"]))\n    df[\"near_z\"] = np.sin(np.radians(df[\"near_longitude\"]))\n    df[\"dot\"] = df[\"x\"]*df[\"near_x\"]+df[\"y\"]*df[\"near_y\"]+df[\"z\"]*df[\"near_z\"]\n\n\n    col_64 = list(df.dtypes[df.dtypes == np.float64].index)\n    for col in col_64:\n        df[col] = df[col].astype(np.float32)\n    return df\n\ndef blocking_and_cat_pred(concat_df,concat_npy2,pred,country_df,thrid_stage_blocking,cat_model):\n    concat_df[\"pred\"] = pred\n    remain_cols = [\"id\",\"near_id\",\"pred\"]\n    drop_cols = [i for i in concat_df.columns if i not in remain_cols]\n    concat_df.drop(columns=drop_cols,inplace=True)\n    concat_df[\"latitude\"] = concat_npy2[:,second_stage_cols2num_dict[\"latitude\"]]\n    concat_df[\"longitude\"] = concat_npy2[:,second_stage_cols2num_dict[\"longitude\"]]\n    concat_df[\"near_latitude\"] = concat_npy2[:,second_stage_cols2num_dict[\"near_latitude\"]]\n    concat_df[\"near_longitude\"] = concat_npy2[:,second_stage_cols2num_dict[\"near_longitude\"]]\n    concat_npy2 = concat_npy2[concat_df[\"pred\"] >= thrid_stage_blocking]\n    concat_df = concat_df[concat_df[\"pred\"] >= thrid_stage_blocking].reset_index(drop=True)\n    cat_pred = cat_model.predict_proba(concat_npy2.astype(np.float32))[:,1]\n    concat_df[\"cat_pred\"] = cat_pred\n    concat_df = concat_df.merge(country_df[[\"id\",\"name\",\"categories\",'address','city','state']],how=\"left\",on=\"id\")\n    country_df = country_df.rename(columns = {\"id\":\"near_id\",\"name\":\"near_name\",\"categories\":\"near_categories\",\n                                             'address':'near_address',\"city\":\"near_city\",'state':'near_state'})\n    concat_df = concat_df.merge(country_df[[\"near_id\",\"near_name\",\"near_categories\",\n                                           'near_address',\"near_city\",'near_state']],how=\"left\",on=\"near_id\")\n    concat_df = add_lat_lon_distance_features(concat_df)\n    return concat_df,concat_npy2\n\n\ndef move_and_sc_num_features(concat_df,concat_npy2,bert_num_cols1,second_stage_cols2num_dict,sc_dict):\n    concat_npy3 = np.zeros([len(concat_npy2),len(bert_num_cols1)]).astype(np.float32)\n    for n,c in enumerate(bert_num_cols1):\n        if c in second_stage_cols2num_dict.keys():\n            concat_npy3[:,n] = concat_npy2[:,second_stage_cols2num_dict[c]]\n            # infの処理\n            concat_npy3[concat_npy3[:,n] == np.inf,n] = np.nan\n            concat_npy3[concat_npy3[:,n] == -np.inf,n] = np.nan\n            # scaling\n            concat_npy3[:,n] = (concat_npy3[:,n] - sc_dict[c][0]) / (sc_dict[c][1]) \n        else:\n            concat_npy3[:,n] = concat_df[c].values.astype(np.float32)\n            # infの処理\n            concat_npy3[concat_npy3[:,n] == np.inf,n] = np.nan\n            concat_npy3[concat_npy3[:,n] == -np.inf,n] = np.nan\n            # scaling\n            concat_npy3[:,n] = (concat_npy3[:,n] - sc_dict[c][0]) / (sc_dict[c][1]) \n    # nanの処理\n    concat_npy3 = np.nan_to_num(concat_npy3)\n    return concat_npy3\n\n\ndef token_sort(concat_df,tokenizer,concat_npy3):\n    token_len = []\n    for t,n_t in zip(concat_df[\"text\"].values, concat_df[\"near_text\"].values):\n        inputs = tokenizer.encode_plus(t, n_t, \n                                       add_special_tokens=True,\n                                      return_offsets_mapping=False)\n        token_len.append(len(inputs[\"input_ids\"]))\n    concat_df[\"token_len\"] = token_len\n    concat_df[\"num_index\"] = np.arange(len(concat_df))\n    concat_df = concat_df.sort_values(by=\"token_len\").reset_index(drop=True)\n    concat_npy3 = concat_npy3[concat_df[\"num_index\"].values,:]\n    concat_df.drop(columns = [\"num_index\"],inplace=True)\n    return concat_df,concat_npy3\n\n\nclass FourSquareDataset(Dataset):\n    def __init__(self, text, near_text,num_features, tokenizer, max_len):\n        self.text = text\n        self.near_text = near_text\n        self.num_features = num_features\n        self.tokenizer = tokenizer\n        self.max_len = max_len\n\n    def __len__(self):\n        return len(self.text)\n\n    def __getitem__(self, item):\n        text = self.text[item]\n        near_text = self.near_text[item]\n        inputs = self.tokenizer(\n            text,near_text,\n            max_length=self.max_len,\n            padding=\"max_length\",\n            truncation=True,\n            return_attention_mask=True,\n            return_token_type_ids=True\n        )\n        ids = inputs[\"input_ids\"]\n        mask = inputs[\"attention_mask\"]\n        token_type_ids = inputs[\"token_type_ids\"]\n        num_feature = self.num_features[item]\n        return {\n                \"input_ids\": torch.tensor(ids, dtype=torch.long),\n                \"attention_mask\": torch.tensor(mask, dtype=torch.long),\n                \"token_type_ids\": torch.tensor(token_type_ids, dtype=torch.long),\n                \"num_feature\" : torch.tensor(num_feature, dtype=torch.float32),\n            }\n    \nclass TransformerHead(nn.Module):\n    def __init__(self, in_features, max_length=128, num_layers=1, nhead=8):\n        super().__init__()\n\n        self.transformer = nn.TransformerEncoder(\n            encoder_layer=nn.TransformerEncoderLayer(d_model=in_features, nhead=nhead),\n            num_layers=num_layers,\n        )\n        self.row_fc = nn.Linear(in_features, 1)\n        self.out_features = max_length\n\n    def forward(self, x):\n        out = self.transformer(x)\n        out = self.row_fc(out).squeeze(-1)\n        p1d = (0, self.out_features - out.shape[-1])\n        out = F.pad(out, p1d, \"constant\", 0)\n        return out\n\nclass FourSquare_model2(nn.Module):\n    def __init__(self):\n        super(FourSquare_model2, self).__init__()\n        self.model = AutoModel.from_pretrained(THIRD_BERT_MODEL2)\n        self.head_type = \"linear\"\n        encoder_feature_size = 768\n        if self.head_type == \"transformer\":\n            self.transformer_head = TransformerHead(\n                    in_features=768,\n                    max_length=128,\n                    num_layers=1,\n                    nhead=8,\n                )\n            encoder_feature_size = self.transformer_head.out_features\n        self.ln1 = nn.LayerNorm(encoder_feature_size)\n        self.linear1 = nn.Sequential(\n            nn.Linear(encoder_feature_size,128),\n            nn.LayerNorm(128),\n            nn.ReLU(),\n            nn.Dropout(0.2))\n        \n        self.linear2 = nn.Sequential(\n            nn.Linear(90,128),\n            nn.LayerNorm(128),\n            nn.ReLU(),\n            nn.Dropout(0.2))\n        \n        self.linear3 = nn.Sequential(\n            nn.Linear(128 + 128,64),\n            nn.LayerNorm(64),\n            nn.ReLU(),\n            nn.Dropout(0.2),\n            nn.Linear(64,1),\n           )\n        \n\n    \n\n    def forward(self, ids, mask, token_type_ids,num_features):\n        if self.head_type == \"transformer\":\n            out = self.model(ids, attention_mask=mask,token_type_ids=token_type_ids)['last_hidden_state']\n            out = self.transformer_head(out)\n        else:\n            out = self.model(ids, attention_mask=mask,token_type_ids=token_type_ids)['last_hidden_state'][:,0,:]\n        out =  self.ln1(out)\n        out = self.linear1(out)\n        out2 = self.linear2(num_features)\n        out = torch.cat([out,out2],axis=-1)\n        out = self.linear3(out)\n        return out\n    \n    \n\n\n    \nclass FourSquare_model(nn.Module):\n    def __init__(self):\n        super(FourSquare_model, self).__init__()\n        self.model = AutoModel.from_pretrained(THIRD_BERT_MODEL1)\n        self.ln1 = nn.LayerNorm(1024)\n        self.linear1 = nn.Sequential(\n            nn.Linear(1024,128),\n            nn.LayerNorm(128),\n            nn.ReLU(),\n            nn.Dropout(0.2))\n        \n        self.linear2 = nn.Sequential(\n            nn.Linear(76,128),\n            nn.LayerNorm(128),\n            nn.ReLU(),\n            nn.Dropout(0.2))\n        \n        self.linear3 = nn.Sequential(\n            nn.Linear(128 + 128,64),\n            nn.LayerNorm(64),\n            nn.ReLU(),\n            nn.Dropout(0.2),\n            nn.Linear(64,1),\n           )\n        \n    \n\n    def forward(self, ids, mask, token_type_ids,num_features):\n        # pooler\n        out = self.model(ids, attention_mask=mask,token_type_ids=token_type_ids)['last_hidden_state'][:,0,:]\n        out =  self.ln1(out)\n        out = self.linear1(out)\n        out2 = self.linear2(num_features)\n        out = torch.cat([out,out2],axis=-1)\n        out = self.linear3(out)\n        return out\n    \n    \ndef collate(d):\n    mask_len = int(d[\"attention_mask\"].sum(axis=1).max())\n    return {\"input_ids\" : d['input_ids'][:,:mask_len],\n                \"attention_mask\" : d['attention_mask'][:,:mask_len],\n                \"token_type_ids\" : d[\"token_type_ids\"][:,:mask_len],\n                 \"num_feature\" : d[\"num_feature\"]}\n\n\ndef make_thrid_pred_amp(model,test_loader):\n    test_preds = []\n    with torch.no_grad():\n        for d in tqdm(test_loader,total=len(test_loader)):\n            d = collate(d)\n            ids = d[\"input_ids\"].to(device)\n            mask = d['attention_mask'].to(device)\n            token_type_ids = d[\"token_type_ids\"].to(device)\n            num_features = d[\"num_feature\"].to(device)\n            with autocast():\n                outputs = model(ids,mask,token_type_ids,num_features)\n            test_preds.append(outputs.sigmoid().detach().cpu().numpy())\n    torch.cuda.empty_cache()\n    test_preds = np.concatenate(test_preds,axis=0)\n    return test_preds\n\ndef make_thrid_pred(model,test_loader):\n    test_preds = []\n    with torch.no_grad():\n        for d in tqdm(test_loader,total=len(test_loader)):\n            d = collate(d)\n            ids = d[\"input_ids\"].to(device)\n            mask = d['attention_mask'].to(device)\n            token_type_ids = d[\"token_type_ids\"].to(device)\n            num_features = d[\"num_feature\"].to(device)\n            outputs = model(ids,mask,token_type_ids,num_features)\n            test_preds.append(outputs.sigmoid().detach().cpu().numpy())\n    torch.cuda.empty_cache()\n    test_preds = np.concatenate(test_preds,axis=0)\n    return test_preds\n\ndef pp(concat_df,pred):\n    sub_ = pd.DataFrame()\n    sub_[\"id\"] = concat_df[\"id\"].values\n    sub_[\"near_id\"] = concat_df[\"near_id\"].values\n    sub_[\"pred\"] = concat_df[pred]\n    sub_ = sub_[sub_[\"pred\"] > 0.5].reset_index(drop=True)\n    # PP\n    # idとnear_idを交換したものの作成\n    sub__ = sub_.copy()\n    sub__.columns = [\"near_id\",\"id\",\"pred\"]\n    sub_ = pd.concat([sub_,sub__]).reset_index(drop=True)\n    sub_ = sub_.drop_duplicates(subset=[\"id\",\"near_id\"]).reset_index(drop=True)\n    sub2_ = pd.DataFrame()\n    sub2_[\"id\"] = id_unique\n    sub2_[\"near_id\"] = id_unique\n    # id == id2を入れる\n    sub_ = pd.concat([sub_,sub2_]).reset_index(drop=True)\n    del sub2_,sub__\n    gc.collect()\n    return sub_","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:31:49.931259Z","iopub.execute_input":"2022-07-03T05:31:49.931781Z","iopub.status.idle":"2022-07-03T05:31:49.996608Z","shell.execute_reply.started":"2022-07-03T05:31:49.931744Z","shell.execute_reply":"2022-07-03T05:31:49.995796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# =================================\n# 4th stage\n# =================================\nclass FourSquare_model3(nn.Module):\n    def __init__(self):\n        super(FourSquare_model3, self).__init__()\n        self.model = AutoModel.from_pretrained(FOURTH_BERT_MODEL1)\n        self.ln1 = nn.LayerNorm(1024)\n        self.linear1 = nn.Sequential(\n            nn.Linear(1024,128),\n            nn.LayerNorm(128),\n            nn.ReLU(),\n            nn.Dropout(0.2))\n        \n        self.linear2 = nn.Sequential(\n            nn.Linear(2,32),\n            nn.LayerNorm(32),\n            nn.ReLU(),\n            nn.Dropout(0.2))\n        \n        self.linear3 = nn.Sequential(\n            nn.Linear(128 + 32,64),\n            nn.LayerNorm(64),\n            nn.ReLU(),\n            nn.Dropout(0.2),\n            nn.Linear(64,1),\n           )\n        \n    def forward(self, ids, mask, token_type_ids,num_features):\n        # pooler\n        out = self.model(ids, attention_mask=mask,token_type_ids=token_type_ids)['last_hidden_state'][:,0,:]\n        out =  self.ln1(out)\n        out = self.linear1(out)\n        out2 = self.linear2(num_features)\n        out = torch.cat([out,out2],axis=-1)\n        out = self.linear3(out)\n        return out\n    \ndef id_sort(id_near_id):\n    id_near_id = \"-\".join(sorted(id_near_id.split(\"-\")))\n    return id_near_id\n\ndef make_pair_4th(sub_3,sub_pair):\n    new_pair = []\n    id_array = sub_3[\"id\"].values\n    match_array = sub_3[\"matches\"].values\n    for i in tqdm(range(len(sub_3))):\n        id_ = id_array[i]\n        match_ = match_array[i]\n        df_ = pd.DataFrame()\n        df_[\"near_id\"] = match_.split(\" \")\n        df_[\"id\"] = id_\n        new_pair.append(df_)\n    new_pair = pd.concat(new_pair).reset_index(drop=True)\n    sub_pair[\"id_near_id\"] = sub_pair[\"id\"].astype(str) + \"-\" + sub_pair[\"near_id\"].astype(str)\n    new_pair[\"id_near_id\"] = new_pair[\"id\"].astype(str) + \"-\" + new_pair[\"near_id\"].astype(str)\n    # ~3rdまで出たpairを削除\n    new_pair = new_pair[~new_pair[\"id_near_id\"].isin(sub_pair[\"id_near_id\"])].reset_index(drop=True)\n    # 重複の削除\n    new_pair[\"id_near_id_sort\"] = new_pair[\"id_near_id\"].map(id_sort)\n    new_pair = new_pair.drop_duplicates(subset = \"id_near_id_sort\").reset_index(drop=True)\n    return new_pair\n    \ndef merge_raw_data_4th(new_pair,test):\n    use_cols = [\"id\",\"name\",\"categories\",'latitude', 'longitude','address','city','state']\n    test_ = test[use_cols].copy()\n    new_pair = new_pair.merge(test_,how=\"left\",on=\"id\")\n    test_.columns = [f\"near_{c}\" for c in use_cols]\n    new_pair = new_pair.merge(test_,how=\"left\",on=\"near_id\")\n    return new_pair\n\ndef token_sort_4th(new_pair,tokenizer):\n    token_len = []\n    for t,n_t in zip(new_pair[\"text\"].values,new_pair[\"near_text\"].values):\n        inputs = tokenizer.encode_plus(t, n_t, \n                                       add_special_tokens=True,\n                                      return_offsets_mapping=False)\n        token_len.append(len(inputs[\"input_ids\"]))\n    new_pair[\"token_len\"] = token_len\n    new_pair = new_pair.sort_values(by=\"token_len\").reset_index(drop=True)\n    return new_pair\n\ndef blocking_4th(new_pair,th):\n    new_pair = new_pair[new_pair[\"pred\"] > th].reset_index(drop=True)\n    new_pair = new_pair[[\"id\",\"near_id\"]].reset_index(drop=True)\n    new_pair_ = new_pair.copy()\n    new_pair_.columns = [\"near_id\",\"id\"]\n    new_pair = pd.concat([new_pair,new_pair_]).reset_index(drop=True)\n    return new_pair","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:31:50.049172Z","iopub.execute_input":"2022-07-03T05:31:50.049572Z","iopub.status.idle":"2022-07-03T05:31:50.070467Z","shell.execute_reply.started":"2022-07-03T05:31:50.04954Z","shell.execute_reply":"2022-07-03T05:31:50.068526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ================================\n# Main\n# ================================\nif DEBUG:\n    test = pd.read_csv(TRAIN_PATH)\n    test = test[test[\"set\"] == 0].reset_index(drop=True)\nelse:\n    test = pd.read_csv(TEST_PATH)\n\nsub = pd.read_csv(SUB_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:31:50.183108Z","iopub.execute_input":"2022-07-03T05:31:50.185024Z","iopub.status.idle":"2022-07-03T05:31:57.944386Z","shell.execute_reply.started":"2022-07-03T05:31:50.184985Z","shell.execute_reply":"2022-07-03T05:31:57.943653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ================================\n# model load\n# ================================\nfirst_stage_place_lgb = [ForestInference.load(i, output_class=False, model_type=\"lightgbm\") for i in first_stage_place_lgb_path]\n\nfirst_stage_name_lgb = [ForestInference.load(i, output_class=False, model_type=\"lightgbm\") for i in first_stage_name_lgb_path]\n\nsecond_stage_lgb = [ForestInference.load(i, output_class=False, model_type=\"lightgbm\") for i in second_stage_lgb_path]\nthird_stage_cat = CatBoostClassifier()\nthird_stage_cat.load_model(third_stage_cat_path[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:31:57.946077Z","iopub.execute_input":"2022-07-03T05:31:57.946349Z","iopub.status.idle":"2022-07-03T05:32:07.721616Z","shell.execute_reply.started":"2022-07-03T05:31:57.946313Z","shell.execute_reply":"2022-07-03T05:32:07.720879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ================\n# feのload\n# ================\nwith open(fe045_categories_path, 'rb') as f:\n    categories_dict = pickle.load(f)\n\nwith open(fe045_city_path, 'rb') as f:\n    city_dict = pickle.load(f)\n\nwith open(fe045_country_path, 'rb') as f:\n    country_dict = pickle.load(f)\n    \nwith open(fe045_state_path , 'rb') as f:\n    state_dict = pickle.load(f)\n\ncat_dict_list = [categories_dict,\n                 city_dict,\n                 country_dict,\n                 state_dict]\n\n# ================\n# fe46\n# ================\nwith open(fe046_svd_path, 'rb') as f:\n    svd = pickle.load(f)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:32:07.722952Z","iopub.execute_input":"2022-07-03T05:32:07.723328Z","iopub.status.idle":"2022-07-03T05:32:07.76271Z","shell.execute_reply.started":"2022-07-03T05:32:07.723294Z","shell.execute_reply":"2022-07-03T05:32:07.762049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# bert-base-multilingualのload\ntest[\"name_bert\"] = test[\"name\"].copy()\ntest[\"name_bert\"] = test[\"name_bert\"].astype(str)\ntest[\"name_bert\"] = test[\"name_bert\"].str.lower()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:32:07.764744Z","iopub.execute_input":"2022-07-03T05:32:07.765153Z","iopub.status.idle":"2022-07-03T05:32:08.214365Z","shell.execute_reply.started":"2022-07-03T05:32:07.765117Z","shell.execute_reply":"2022-07-03T05:32:08.213627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# embの作成 + svd\ntokenizer = AutoTokenizer.from_pretrained(BERT_MODEL)\n\nname = test[\"name_bert\"].unique()\nname2num_dict = {}\nfor n,i in enumerate(name):\n    name2num_dict[i] = n  \n    \nid2num_dict = {}\nfor i,n in zip(test[\"id\"].values,test[\"name_bert\"].values):\n    id2num_dict[i] = name2num_dict[n]\n    \ntest = test.drop(columns = [\"name_bert\"])\ndel name2num_dict\ngc.collect()\n\ntrain_ = BertDataset(name, tokenizer, MAX_LEN)\ntrain_loader = DataLoader(\n        dataset=train_, batch_size=BS, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:32:08.215638Z","iopub.execute_input":"2022-07-03T05:32:08.215897Z","iopub.status.idle":"2022-07-03T05:32:09.910579Z","shell.execute_reply.started":"2022-07-03T05:32:08.215864Z","shell.execute_reply":"2022-07-03T05:32:09.909428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = bert_model()\nmodel = model.to(device)\nmodel.eval()\nif DEBUG:\n    bert_emb = np.load(\"../input/exp038-ex059-sub-fold2-all-distance-miss-gb-debug/bert_emb.npy\")\nelse:\n    bert_emb = make_emb(model,train_loader,svd=svd)\n#np.save(\"bert_emb.npy\",bert_emb)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:32:09.91507Z","iopub.execute_input":"2022-07-03T05:32:09.915277Z","iopub.status.idle":"2022-07-03T05:32:17.595368Z","shell.execute_reply.started":"2022-07-03T05:32:09.915251Z","shell.execute_reply":"2022-07-03T05:32:17.594645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_list = []\npair_list = []\nif DEBUG:\n    if DEBUG_COUNTRY[0] == \"ALL\":\n        country_unique = test[\"country\"].value_counts().index\n    else:\n        country_unique = DEBUG_COUNTRY\nelse:\n    country_unique = test[\"country\"].value_counts().index","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:32:17.596664Z","iopub.execute_input":"2022-07-03T05:32:17.596914Z","iopub.status.idle":"2022-07-03T05:32:17.601586Z","shell.execute_reply.started":"2022-07-03T05:32:17.596887Z","shell.execute_reply":"2022-07-03T05:32:17.600829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"third_model = None\nthird_tokenizer = AutoTokenizer.from_pretrained(THIRD_BERT_MODEL1)\nthird_model2 = None\nthird_tokenizer2 = AutoTokenizer.from_pretrained(THIRD_BERT_MODEL2)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:32:17.603745Z","iopub.execute_input":"2022-07-03T05:32:17.604468Z","iopub.status.idle":"2022-07-03T05:32:19.777705Z","shell.execute_reply.started":"2022-07-03T05:32:17.604433Z","shell.execute_reply":"2022-07-03T05:32:19.776871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for n,country in enumerate(country_unique):\n    country_df = test[test[\"country\"] == country].reset_index(drop=True)\n    id_unique = country_df[\"id\"].unique()\n    # feature enginnering -> pred\n    if len(country_df ) >= 2:\n        # ==================================\n        # first stage\n        # ==================================\n        print(f\"{n},{country}\")\n        # place\n        first_stage_columns = ['id','name','categories','latitude','longitude']\n        concat_df = make_candidate_first_stage(country_df, \n                                               n_neighbors_first_stage,\n                                               first_stage_columns)\n        # id == near_idの削除\n        concat_df = delete_match_id(concat_df)\n        \n        # numpyへの変換\n        concat_npy = np.zeros([len(concat_df),\n                               len(first_stage_place_features)],\n                              dtype=np.float32)\n        cols =  ['latitude','longitude','rank','d_near','near_latitude', 'near_longitude']\n        concat_npy, concat_df = df2numpy(concat_npy,\n                                         concat_df,cols,\n                                         first_stage_place_cols2num_dict)\n        \n        gc.collect()\n        \n        # 特徴量エンジニアリング\n        distance_columns = ['name','categories']\n        # 特徴量エンジニアリング\n\n        # 距離特徴量の作成\n        for c in distance_columns:\n            distance = calc_distance_first_stage(concat_df[c].values,\n                                                 concat_df[f\"near_{c}\"].values)\n            for n,f_c in enumerate([f\"{c}_jaro\"]):\n                concat_npy[:,first_stage_place_cols2num_dict[f_c]] = distance[:,n].astype(np.float32)\n            del distance\n            gc.collect()  \n        \n        # 予測 + rank\n        pred = make_pred(concat_npy,first_stage_place_lgb )\n                \n        concat_df[\"pred\"] = pred\n        if (DEBUG) & (DEBUG_COUNTRY[0] != \"ALL\"):\n            concat_df.to_csv(f\"1st_stage_place_{country}.csv\",index=False)\n            np.save(f\"1st_stage_place_npy_{country}.npy\",concat_npy)\n        concat_df, concat_npy = remove_low_rank_place(concat_df, \n                                                concat_npy, \n                                                second_stage_rank,\n                                                first_stage_place_cols2num_dict)\n        del pred,concat_npy\n        gc.collect()\n        \n        # ====================\n        # name emb\n        # ====================\n        train_ = BertDataset(country_df[\"name\"], tokenizer, MAX_LEN,True)\n        train_loader = DataLoader(\n                dataset=train_, batch_size=BS, shuffle=False)\n        country_name_emb = make_emb(model,train_loader,svd=None)\n        \n        first_stage_columns = ['id','name','latitude','longitude']\n        concat_name_df = make_candidate_name_first_stage(country_df, \n                                               country_name_emb,\n                                               n_neighbors_first_stage,\n                                               first_stage_columns)\n        del country_name_emb\n        gc.collect()\n        # id == near_idの削除\n        concat_name_df = delete_match_id(concat_name_df)\n        \n        # numpyへの変換\n        concat_name_npy = np.zeros([len(concat_name_df),\n                               len(first_stage_name_features)],\n                              dtype=np.float32)\n        cols =  ['latitude','longitude','rank','d_near','near_latitude', 'near_longitude']\n        concat_name_npy, concat_name_df = df2numpy(concat_name_npy,\n                                                   concat_name_df,\n                                                   cols,\n                                                   first_stage_name_cols2num_dict)\n        gc.collect()\n        \n        # 特徴量エンジニアリング\n        distance_columns = ['name']\n        # 特徴量エンジニアリング\n\n        # 距離特徴量の作成\n        for c in distance_columns:\n            distance = calc_distance_first_stage(concat_name_df[c].values,\n                                                 concat_name_df[f\"near_{c}\"].values)\n            for n,f_c in enumerate([f\"{c}_jaro\"]):\n                concat_name_npy[:,first_stage_name_cols2num_dict[f_c]] = distance[:,n].astype(np.float32)\n            del distance\n            gc.collect() \n\n        # 位置のdistanceの作成\n        concat_name_npy = make_place_distance(concat_name_npy,first_stage_name_cols2num_dict)\n        \n        # 予測 + rank\n        pred = make_pred(concat_name_npy,first_stage_name_lgb )\n        \n        concat_name_df[\"pred\"] = pred\n        if (DEBUG) & (DEBUG_COUNTRY[0] != \"ALL\"):\n            concat_name_df.to_csv(f\"1st_stage_name_{country}.csv\",index=False)\n            np.save(f\"1st_stage_name_npy_{country}.npy\",concat_name_npy)\n        #np.save(\"1st_stage\")\n        concat_name_df, concat_name_npy = remove_low_rank_name(concat_name_df, \n                                                concat_name_npy, \n                                                second_stage_rank,\n                                                first_stage_name_cols2num_dict)\n        del pred,concat_name_npy\n        # remove\n        remove_cols = [\"categories\",\"near_categories\"]\n        concat_df.drop(columns = remove_cols,inplace=True)\n        concat_df = pd.concat([concat_name_df,concat_df]).reset_index(drop=True)\n        del concat_name_df\n        concat_df = concat_df.drop_duplicates(subset=[\"id\",\"near_id\"]).reset_index(drop=True)\n        \n        \n        \n        # ==================================\n        # second stage\n        # ==================================\n        concat_npy2 = np.zeros([len(concat_df),len(second_stage_features)],dtype=np.float32)\n        # 特徴量を移す\n        features = ['latitude','longitude','rank','d_near','near_latitude', 'near_longitude']\n        concat_npy2,concat_df = move_features(concat_npy2, concat_df, features)\n        gc.collect()\n        # merge\n        concat_df[\"near_id\"] = concat_df[\"near_id\"].astype(\"category\")\n        use_cols = [\"id\",\"address\",\"city\",\"state\",\"zip\",\"country\",\"url\",\"phone\",'categories']\n        concat_df = merge_raw_data(concat_df, country_df,use_cols)\n        remain_cols = [\"id\",\"name\",\"categories\",'address', 'city', 'state']\n        country_df.drop(columns = [c for c in country_df.columns if c not in remain_cols],inplace=True)\n        gc.collect()\n        # 特徴量エンジニアリング\n        distance_columns = ['name', 'address', 'city', 'state',\n           'zip', 'url', 'phone', 'categories']\n        concat_df, concat_npy2 = make_distance_second_stage(concat_df,\n                                                           concat_npy2,\n                                                           distance_columns,\n                                                           second_stage_cols2num_dict)\n        # 集約\n        concat_npy2 = distance_agg(concat_npy2,concat_df,second_stage_cols2num_dict)\n        cat_cols = [\"categories\",\"city\",\"country\",\"state\"]\n        concat_npy2, concat_df = make_cat_features(concat_npy2,concat_df,cat_dict_list,second_stage_cols2num_dict)\n        concat_npy2 = concat_name_emb(concat_npy2,concat_df,id2num_dict,bert_emb,second_stage_cols2num_dict)\n        \n        gc.collect() \n        # predict\n        pred = make_pred(concat_npy2,second_stage_lgb)\n        if (DEBUG) & (DEBUG_COUNTRY[0] != \"ALL\"):\n            #sub_.to_csv(f\"second_{country}_sub.csv\",index=False)\n            np.save(f\"second_{country}.npy\",concat_npy2)\n            concat_df.to_csv(f\"second_{country}.csv\",index=False)\n            \n        # ==================================\n        # third stage\n        # ==================================\n        concat_df,concat_npy2 = blocking_and_cat_pred(concat_df,concat_npy2,pred,country_df,thrid_stage_blocking,third_stage_cat)\n        concat_npy3 = move_and_sc_num_features(concat_df,concat_npy2,bert_num_cols1,second_stage_cols2num_dict,sc_dict)\n                                 \n        del pred,concat_npy2,country_df\n        concat_df[\"text\"] = concat_df[\"name\"].astype(str).str.lower() + \" \" + concat_df[\"categories\"].astype(str).str.lower()+\\\n                            concat_df['address'].astype(str).str.lower() + \" \" + concat_df['city'].astype(str).str.lower() + \" \" + concat_df['state'].astype(str).str.lower() \n        concat_df[\"near_text\"] = concat_df[\"near_name\"].astype(str).str.lower() + \" \" + concat_df[\"near_categories\"].astype(str).str.lower()+\\\n                             concat_df['near_address'].astype(str).str.lower() + \" \" + concat_df['near_city'].astype(str).str.lower() + \" \" + concat_df['near_state'].astype(str).str.lower()\n        concat_df,concat_npy3 = token_sort(concat_df,third_tokenizer,concat_npy3)  \n        \n        # 必要な特徴量のみ\n        cols2num_bert_cols1 = {}\n        for n,c in enumerate(bert_num_cols1):\n            cols2num_bert_cols1[c] = n\n        use_cols_index = []\n        for i in bert_num_cols2:\n            use_cols_index.append(cols2num_bert_cols1[i])\n            \n            \n        test_ = FourSquareDataset(concat_df[\"text\"].values,\n                                  concat_df[\"near_text\"].values,\n                                  concat_npy3[:,use_cols_index],\n                                  third_tokenizer, THIRD_MAX_LEN)\n        \n            \n        test2_ = FourSquareDataset(concat_df[\"text\"].values,\n                                  concat_df[\"near_text\"].values,\n                                  concat_npy3,\n                                  third_tokenizer2, THIRD_MAX_LEN2)\n        \n        test_loader = DataLoader(\n            dataset=test_, batch_size=THIRD_BS, shuffle=False)\n        test_loader2 = DataLoader(\n            dataset=test2_, batch_size=THIRD_BS2, shuffle=False)\n        # fgmはamp解除で学習\n        if third_model is None:\n            third_model =  FourSquare_model()\n            third_model.load_state_dict(torch.load(third_stage_model1_path))\n            third_model = third_model.to(device)\n            third_model.eval()\n        pred1 = make_thrid_pred_amp(third_model,test_loader)\n        \n        if third_model2 is None:\n            third_model2 =  FourSquare_model2()\n            third_model2.load_state_dict(torch.load(third_stage_model2_path))\n            third_model2 = third_model2.to(device)\n            third_model2.eval() \n        pred2 = make_thrid_pred(third_model2,test_loader2)\n        \n        concat_df[\"pred\"] = concat_df[\"pred\"]*w1 + pred1.reshape(-1)*w2 + pred2.reshape(-1)*w3 + concat_df[\"cat_pred\"]*w4\n        gc.collect()\n        #print(pred)\n        sub_ = pp(concat_df, \"pred\")\n        if (DEBUG) & (DEBUG_COUNTRY[0] != \"ALL\"):\n            sub_.to_csv(f\"second_{country}_sub.csv\",index=False)\n            concat_df.to_csv(f\"second_{country}.csv\",index=False)\n        del concat_df,concat_npy3,test_,test_loader\n        gc.collect()\n        # make sub\n        pair_list.append(sub_[[\"id\",\"near_id\"]])\n        sub_ = sub_.groupby(\"id\")[\"near_id\"].apply(join)\n        sub_ = sub_.reset_index()\n        sub_.columns = [\"id\",\"matches\"]\n        \n        sub_list.append(sub_)\n    else:\n        sub_ = pd.DataFrame()\n        sub_[\"id\"] = id_unique\n        sub_[\"matches\"] = id_unique\n        sub_list.append(sub_)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:32:19.779231Z","iopub.execute_input":"2022-07-03T05:32:19.779618Z","iopub.status.idle":"2022-07-03T05:39:27.488653Z","shell.execute_reply.started":"2022-07-03T05:32:19.77958Z","shell.execute_reply":"2022-07-03T05:39:27.487942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# modelの削除\ndel third_model,third_model2,model","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:39:27.491259Z","iopub.execute_input":"2022-07-03T05:39:27.491524Z","iopub.status.idle":"2022-07-03T05:39:27.498086Z","shell.execute_reply.started":"2022-07-03T05:39:27.491489Z","shell.execute_reply":"2022-07-03T05:39:27.497219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.concat(sub_list).reset_index(drop=True)\nsub_pair = pd.concat(pair_list).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:39:27.499926Z","iopub.execute_input":"2022-07-03T05:39:27.50026Z","iopub.status.idle":"2022-07-03T05:39:27.513706Z","shell.execute_reply.started":"2022-07-03T05:39:27.500224Z","shell.execute_reply":"2022-07-03T05:39:27.512988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ===================================================\n# PP\n# ===================================================\nsub[\"matches_len\"] = sub[\"matches\"].map(lambda x:len(x.split(\" \")))\nsub_3 = sub[sub[\"matches_len\"] > 3].reset_index(drop=True)\nnear_id_value = sub_3[\"matches\"].values\nid_list = sub_3[\"id\"].values\nkey_id_list = []\nfor n,key_id in tqdm(enumerate(id_list),total=len(id_list)):\n    if key_id in(key_id_list):\n        pass\n    else:\n        len_list = []\n        key_near_id = near_id_value[n]\n        a = len(key_near_id.split(\" \"))\n        for near_id in near_id_value:\n            b = len(near_id.split(\" \"))\n            c = len(set(key_near_id.split(\" \")) & set(near_id.split(\" \")))\n            len_list.append([a,b,c])\n        df = pd.DataFrame(len_list)\n        df.columns = [\"id_len\",\"near_id_len\",\"common_len\"]\n        df[\"id_rate\"] =  df[\"common_len\"] / df[\"id_len\"]\n        df[\"near_id_rate\"] = df[\"common_len\"] / df[\"near_id_len\"]\n        df[\"id\"] = id_list\n        df = df[df[\"common_len\"] != 0].reset_index(drop=True)\n        df = df[(df[\"id_rate\"] >= 0.5) | (df[\"near_id_rate\"] >= 0.5)].reset_index(drop=True)\n        if len(df) > 1:\n            for k in df[\"id\"]:\n                key_id_list.append(k)\n            all_id = near_id_value[sub_3[\"id\"].isin(df[\"id\"])]\n            all_id_concat = []\n            for i in all_id:\n                all_id_concat += i.split(\" \")\n            all_id_unique = list(set(all_id_concat))\n            near_id_value[sub_3[\"id\"].isin(df[\"id\"])] = \" \".join(all_id_unique)\nsub_3[\"matches\"] = near_id_value\nsub_under_3 = sub[sub[\"matches_len\"] <= 3].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:39:27.515376Z","iopub.execute_input":"2022-07-03T05:39:27.515788Z","iopub.status.idle":"2022-07-03T05:39:27.779128Z","shell.execute_reply.started":"2022-07-03T05:39:27.515746Z","shell.execute_reply":"2022-07-03T05:39:27.778366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ===============================================================\n# 4th stage\n# ===============================================================\nif len(sub_3) > 0:\n    fourth_tokenizer = AutoTokenizer.from_pretrained(FOURTH_BERT_MODEL1)\n    new_pair = make_pair_4th(sub_3,sub_pair)\n    new_pair = merge_raw_data_4th(new_pair,test)\n    new_pair[\"text\"] = new_pair[\"name\"].astype(str).str.lower() + \" \" + new_pair[\"categories\"].astype(str).str.lower()+\\\n                            new_pair['address'].astype(str).str.lower() + \" \" + new_pair['city'].astype(str).str.lower() + \" \" + new_pair['state'].astype(str).str.lower() \n    new_pair[\"near_text\"] = new_pair[\"near_name\"].astype(str).str.lower() + \" \" + new_pair[\"near_categories\"].astype(str).str.lower()+\\\n                             new_pair['near_address'].astype(str).str.lower() + \" \" + new_pair['near_city'].astype(str).str.lower() + \" \" + new_pair['near_state'].astype(str).str.lower()\n\n    new_pair = token_sort_4th(new_pair,fourth_tokenizer)\n    for c in [\"latitude\",\"longitude\"]:\n        new_pair[c] = (new_pair[c] - sc_dict[c][0]) / (sc_dict[c][1])\n\n    num_features = new_pair[[\"latitude\",\"longitude\"]].values\n    test_ = FourSquareDataset(new_pair[\"text\"].values,\n                              new_pair[\"near_text\"].values,\n                              num_features,\n                              fourth_tokenizer, FOURTH_MAX_LEN)\n    test_loader = DataLoader(\n            dataset=test_, batch_size=FOURTH_BS, shuffle=False)\n    fourth_model =  FourSquare_model3()\n    fourth_model.load_state_dict(torch.load(fourth_stage_model1_path))\n    fourth_model = fourth_model.to(device)\n    fourth_model.eval()\n\n    pred_pp = make_thrid_pred_amp(fourth_model,test_loader)\n    new_pair[\"pred\"] = pred_pp.reshape(-1)\n    new_pair = blocking_4th(new_pair,fourth_stage_blocking)\n    sub_pair = pd.concat([sub_pair,new_pair]).reset_index(drop=True) \n    sub_pair = sub_pair.groupby(\"id\")[\"near_id\"].apply(join)\n    sub_pair = sub_pair.reset_index()\n    sub_pair.columns = [\"id\",\"matches\"]\n    \nelse:\n    sub_pair = sub_pair.groupby(\"id\")[\"near_id\"].apply(join)\n    sub_pair = sub_pair.reset_index()\n    sub_pair.columns = [\"id\",\"matches\"]\n\nif DEBUG:\n    train_raw = pd.read_csv(\"../input/foursquare-fold/fold_train.csv\")\n    id2poi = get_id2poi(train_raw[[\"id\",\"point_of_interest\"]])\n    poi2ids = get_poi2ids(train_raw[[\"id\",\"point_of_interest\"]])\n    score = get_score(sub_pair)\n    print(score)\nelse:\n    # sub_pp = pd.concat([sub_under_3[[\"id\",\"matches\"]],sub_3[[\"id\",\"matches\"]]]).reset_index(drop=True)\n    null_country = test[~test[\"id\"].isin(sub_pair[\"id\"])][[\"id\"]].reset_index(drop=True)\n    null_country[\"matches\"] = null_country[\"id\"].values\n    sub_pp = pd.concat([sub_pair,null_country]).reset_index(drop=True)\n    #sub_pp.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T05:39:27.780763Z","iopub.execute_input":"2022-07-03T05:39:27.781276Z","iopub.status.idle":"2022-07-03T05:40:40.369256Z","shell.execute_reply.started":"2022-07-03T05:39:27.781234Z","shell.execute_reply":"2022-07-03T05:40:40.368503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# merge train pairs\ntrain = pd.read_csv(TRAIN_PATH)\ntest = pd.read_csv(TEST_PATH)\ntrain_dict = {}\nids = train[\"id\"].values\nnames = train[\"name\"].values\nlatitudes = train[\"latitude\"].values\nlongitudes = train[\"longitude\"].values\nfor idx in tqdm(range(len(train))):\n    id = ids[idx]\n    name = names[idx]\n    latitude = latitudes[idx]\n    longitude = longitudes[idx]\n    train_dict[(name, latitude, longitude)] = id\n\ndel ids, names, latitudes, longitudes\n\nrename_dict = {}\nids = test[\"id\"].values\nnames = test[\"name\"].values\nlatitudes = test[\"latitude\"].values\nlongitudes = test[\"longitude\"].values\nfor idx in tqdm(range(len(test))):\n    id = ids[idx]\n    name = names[idx]\n    latitude = latitudes[idx]\n    longitude = longitudes[idx]\n    if (name, latitude, longitude) in train_dict:\n        rename_dict[train_dict[(name, latitude, longitude)]] = id\ndel train_dict\ndel ids, names, latitudes, longitudes\n\ntrain[\"id\"] = train[\"id\"].map(lambda x:rename_dict[x] if x in rename_dict else x+\"_t\")\n\nids = []\nnear_ids = []\nfor poi, poi_df in tqdm(train[[\"id\", \"point_of_interest\"]].groupby(\"point_of_interest\")):\n    for id1 in poi_df[\"id\"].values:\n        for id2 in poi_df[\"id\"].values:\n            if not id1.endswith(\"_t\") and not id2.endswith(\"_t\"): # どちらもtestに含まれるもののみ残す\n                ids.append(id1)\n                near_ids.append(id2)\n\n                \ndef matches2pairs(df_matches):\n    pair_ids = []\n    pair_near_ids = []\n    ids_val = df_matches[\"id\"].values\n    matches_val = df_matches[\"matches\"].values\n    for i in tqdm(range(len(df_matches))):\n        idx = ids_val[i]\n        matches = matches_val[i].split()\n        pair_ids += [idx]*len(matches)\n        pair_near_ids += matches\n    df_pairs = pd.DataFrame(data={\"id\":pair_ids, \"near_id\":pair_near_ids})\n    return df_pairs\n\ndef join(df):\n    x = [str(e) for e in list(df)]\n    return \" \".join(x)\n\ndef pairs2matches(df_pairs):\n    df_matches = df_pairs.groupby(\"id\")[\"near_id\"].apply(join)\n    df_matches = pd.DataFrame(df_matches).reset_index().sort_values(by=\"id\").reset_index(drop=True)\n    return df_matches.rename({\"near_id\":\"matches\"}, axis=1)\n\n\ntrain_all_pair = pd.DataFrame(data={\"id\":ids, \"near_id\":near_ids})\ndel ids, near_ids","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_pair = matches2pairs(sub_pp)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 両方testだけ残す\nsub_pair = sub_pair[(~(sub_pair[\"id\"].isin(train[\"id\"].values))) \n                      & (~(sub_pair[\"near_id\"].isin(train[\"id\"].values)))].reset_index(drop=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_pair = pd.concat([sub_pair, train_all_pair]).drop_duplicates().reset_index(drop=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 念の為\nsub_pair = sub_pair[(~sub_pair[\"id\"].map(lambda x:x.endswith(\"_t\")))&(~sub_pair[\"near_id\"].map(lambda x:x.endswith(\"_t\")))]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pairs2matches(sub_pair)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\",index=False)","metadata":{},"execution_count":null,"outputs":[]}]}