{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%bash\n# install sentence transformers from dataset\ncp -r ../input/sentencetransformers-sourceandsomemodels/sentence-transformers ./sentence-transformers\npip install -U --no-build-isolation --no-deps ./sentence-transformers\n\n## copy sentence transformers pretrained models and configuration files from dataset to local caches\nmkdir -p  /root/.cache/torch/\n\ncp -r ../input/sentencetransformers-sourceandsomemodels/torch/sentence_transformers /root/.cache/torch","metadata":{"execution":{"iopub.status.busy":"2022-06-08T08:17:19.616083Z","iopub.execute_input":"2022-06-08T08:17:19.616678Z","iopub.status.idle":"2022-06-08T08:18:19.116075Z","shell.execute_reply.started":"2022-06-08T08:17:19.616584Z","shell.execute_reply":"2022-06-08T08:18:19.114967Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!unzip  ../input/pykakasi/pykakasi.deps -d ./\n!pip install --no-index --find-links=./pykakasi pykakasi","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n\nVERBOSE = False\n\ndef display_data():\n    if VERBOSE:\n        data = pd.read_feather(\"data.feather\")\n        display(data.head())\n\ndef display_pairs():\n    if VERBOSE:\n        pairs = pd.read_feather(\"pairs.feather\")\n        display(pairs.head())\ndef display_features():\n    if VERBOSE:\n        pairs = pd.read_feather(\"pairs.feather\")\n        cols = [c for c in pairs.columns]\n        print(len(cols))\n        display(cols)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T08:18:19.118544Z","iopub.execute_input":"2022-06-08T08:18:19.118945Z","iopub.status.idle":"2022-06-08T08:18:19.129493Z","shell.execute_reply.started":"2022-06-08T08:18:19.118906Z","shell.execute_reply":"2022-06-08T08:18:19.128628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile setup.py\nimport pandas as pd\nimport sys\ndef main():\n    args = sys.argv[1:]\n    if args[0] == \"test\":\n        df = pd.read_csv(\"../input/foursquare-location-matching/test.csv\")\n        cols = [ c for c in df.columns]\n        if df.shape[0] == 5:\n            df = pd.read_csv(\"../input/foursquare-location-matching/train.csv\")\n            df = df[cols][:100]\n    elif args[0] == \"train\":\n        df = pd.read_feather(\"../input/flm-kfold-pairs/train_data.feather\")\n        if DEBUG:\n            df = df[:100]\n    elif args[0] == \"valid\":\n        df = pd.read_feather(\"../input/flm-kfold-pairs/valid_data.feather\")\n        if DEBUG:\n            df = df[:100]        \n    \n    print (f\"{args[0]} data shape:{df.shape}\")\n    cols = [c for c in df.columns]\n    print (cols)\n    df.to_feather (\"data.feather\")\n    \nDEBUG = False\nif __name__ == \"__main__\":\n    main()","metadata":{"execution":{"iopub.status.busy":"2022-06-08T08:18:19.130709Z","iopub.execute_input":"2022-06-08T08:18:19.132961Z","iopub.status.idle":"2022-06-08T08:18:22.011618Z","shell.execute_reply.started":"2022-06-08T08:18:19.132919Z","shell.execute_reply":"2022-06-08T08:18:22.010235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n!python setup.py test","metadata":{"execution":{"iopub.status.busy":"2022-06-08T08:18:22.016546Z","iopub.execute_input":"2022-06-08T08:18:22.017616Z","iopub.status.idle":"2022-06-08T08:18:31.348534Z","shell.execute_reply.started":"2022-06-08T08:18:22.017566Z","shell.execute_reply":"2022-06-08T08:18:31.347531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Text2Vec","metadata":{}},{"cell_type":"code","source":"%%writefile text2vec.py\nimport numpy as np\nimport pandas as pd\n\nfrom sentence_transformers import SentenceTransformer\nimport pickle\n\ndef sentence_transformer_feat2Vec (model_name,df, feat): \n\n    model = SentenceTransformer(model_name)\n    text_df = df[[feat]].drop_duplicates().reset_index(drop=True)\n    \n    vec =  model.encode(text_df[feat].values, show_progress_bar=True)\n\n    return  text_df, vec                    \n\ndef main():\n\n\n    df = pd.read_feather(\"data.feather\")\n\n    df[\"name\"] =  df[\"name\"].fillna(\"unknow\")\n    df[\"address\"] =  df[\"address\"].fillna(\"unknow\")\n    df[\"city\"] =  df[\"city\"].fillna(\"unknow\")\n    df[\"state\"] =  df[\"state\"].fillna(\"unknow\")\n    df[\"country\"] =  df[\"country\"].fillna(\"unknow\")\n    df[\"categories\"] =  df[\"categories\"].fillna(\"unknow\")\n    #df[\"city_state_country\"] = df[\"city\"]+ \", \" +  df[\"state\"] + \", \" + df[\"country\"]\n\n\n    for model_name in [\n        'sentence-transformers_paraphrase-xlm-r-multilingual-v1', \n        'sentence-transformers_paraphrase-multilingual-mpnet-base-v2',\n        'sentence-transformers_all-mpnet-base-v2',\n    ]:\n\n        for feat in [ \"categories\", \"city\", \"state\", \"name\", \"address\" ]:\n\n            print(f\"{model_name}: {feat}\")\n            text_df, vec = sentence_transformer_feat2Vec (f\"{LOCAL_CACHE}sentence_transformers/{model_name}\",   df, feat)\n            text_df.to_csv(f\"{model_name}_{feat}.csv\", index=False)\n\n            with open(f'{model_name}_{feat}.vec', 'wb') as handle:\n                pickle.dump(vec, handle)\n            print(f\"vec:{vec.shape}\")\n\n            \nLOCAL_CACHE = '../input/sentencetransformers-sourceandsomemodels/torch/'\nif __name__ == \"__main__\":\n    main()","metadata":{"execution":{"iopub.status.busy":"2022-06-08T08:18:31.35033Z","iopub.execute_input":"2022-06-08T08:18:31.350721Z","iopub.status.idle":"2022-06-08T08:18:31.357153Z","shell.execute_reply.started":"2022-06-08T08:18:31.350667Z","shell.execute_reply":"2022-06-08T08:18:31.356431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n!python text2vec.py","metadata":{"execution":{"iopub.status.busy":"2022-06-08T08:18:31.358591Z","iopub.execute_input":"2022-06-08T08:18:31.35908Z","iopub.status.idle":"2022-06-08T08:19:39.353357Z","shell.execute_reply.started":"2022-06-08T08:18:31.359042Z","shell.execute_reply":"2022-06-08T08:19:39.352338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LatLong2Vect","metadata":{}},{"cell_type":"code","source":"%%writefile latlong2vec.py\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport pickle\n\n\n#https://stackoverflow.com/questions/10473852/convert-latitude-and-longitude-to-point-in-3d-space\ndef LLHtoECEF(lat, lon):\n    rad = np.float64(6378137.0)        # Radius of the Earth (in meters)\n    f = np.float64(1.0/298.257223563)  # Flattening factor WGS84 Model\n    LLHtoECEF_FF = (1.0-f)**2\n\n    cosLat = np.cos(lat)\n    sinLat = np.sin(lat)\n    C = 1/np.sqrt(cosLat**2 + LLHtoECEF_FF * sinLat**2)\n    S = C * LLHtoECEF_FF\n\n    x = (rad * C)*cosLat * np.cos(lon)\n    y = (rad * C)*cosLat * np.sin(lon)\n    z = (rad * S)*sinLat\n    \n\n    mat = np.vstack((x,y,z)).T\n\n    mat = mat / np.linalg.norm(mat, axis=1).reshape((-1, 1))\n\n    \n    return mat\n\ndef lat_lon_feat2vec(df):\n    lat_lon_matrix = df[[\"latitude\",\"longitude\"]].values\n    vec = LLHtoECEF ( lat_lon_matrix[:,0], lat_lon_matrix[:,1] )\n    \n    return vec\n\ndef main():\n    data = pd.read_feather(\"data.feather\")\n    vec = lat_lon_feat2vec(data)\n    with open(f'lat_lon.vec', 'wb') as handle:\n        pickle.dump(vec, handle)\n\n    print(f\"latlong2vec shape:{vec.shape}\")\n    \nif __name__ == \"__main__\":\n    main()","metadata":{"execution":{"iopub.status.busy":"2022-06-08T08:19:39.355339Z","iopub.execute_input":"2022-06-08T08:19:39.355752Z","iopub.status.idle":"2022-06-08T08:19:39.362681Z","shell.execute_reply.started":"2022-06-08T08:19:39.355705Z","shell.execute_reply":"2022-06-08T08:19:39.361507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n!python latlong2vec.py","metadata":{"execution":{"iopub.status.busy":"2022-06-08T08:19:39.363975Z","iopub.execute_input":"2022-06-08T08:19:39.364314Z","iopub.status.idle":"2022-06-08T08:19:40.625459Z","shell.execute_reply.started":"2022-06-08T08:19:39.364268Z","shell.execute_reply":"2022-06-08T08:19:40.624362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generate pairs (pykakasi)","metadata":{}},{"cell_type":"code","source":"%%writefile convert.py\nimport pandas as pd\nimport pykakasi\n\n\ndef convert_japanese_alphabet(df: pd.DataFrame):\n    kakasi = pykakasi.kakasi()\n    kakasi.setMode('H', 'a')  # Convert Hiragana into alphabet\n    kakasi.setMode('K', 'a')  # Convert Katakana into alphabet\n    kakasi.setMode('J', 'a')  # Convert Kanji into alphabet\n    conversion = kakasi.getConverter()\n\n    def convert(row):\n        for column in [\"name\", \"address\", \"city\", \"state\"]:\n            try:\n                row[column] = conversion.do(row[column])\n            except:\n                pass\n        return row\n\n    df[df[\"country\"] == \"JP\"] = df[df[\"country\"] == \"JP\"].apply(convert, axis=1)\n    return df\n\n\ndef main():\n    data = pd.read_feather(\"data.feather\")\n    print(data.query(\"country == 'JP'\")[\"name\"].values[:10])\n    \n    data = data [[\"id\", \"name\", \"address\", \"city\", \"state\", \"country\"]]\n    print(f\"shape:{data.shape}\")\n\n    data = convert_japanese_alphabet(data)\n    print(data.query(\"country == 'JP'\")[\"name\"].values[:10])    \n    print(f\"shape:{data.shape}\")\n    \n    data.to_feather(\"convert_data.feather\")\n\n\nif __name__ == \"__main__\":\n    main()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n!python convert.py","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile generate_pairs.py\nimport warnings\nwarnings.filterwarnings('ignore')\nimport os\n\nimport random\nimport pandas as pd\nimport numpy as np\nimport pickle\n\nimport torch\nif torch.cuda.is_available():\n    import cuml\n    from cuml.neighbors import NearestNeighbors \n    print(f\"cuml:{cuml.__version__}\")\nelse:\n    from sklearn.neighbors import NearestNeighbors \nfrom sklearn.feature_extraction.text import TfidfVectorizer    \n    \ndef seed_everything(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)    \n    \n\n\nfrom tqdm import tqdm\n\n#https://stackoverflow.com/questions/10473852/convert-latitude-and-longitude-to-point-in-3d-space\ndef LLHtoECEF(lat, lon):\n    rad = np.float64(6378137.0)        # Radius of the Earth (in meters)\n    f = np.float64(1.0/298.257223563)  # Flattening factor WGS84 Model\n    LLHtoECEF_FF = (1.0-f)**2\n\n    cosLat = np.cos(lat)\n    sinLat = np.sin(lat)\n    C = 1/np.sqrt(cosLat**2 + LLHtoECEF_FF * sinLat**2)\n    S = C * LLHtoECEF_FF\n\n    x = (rad * C)*cosLat * np.cos(lon)\n    y = (rad * C)*cosLat * np.sin(lon)\n    z = (rad * S)*sinLat\n    \n\n    mat = np.vstack((x,y,z)).T\n\n    mat = mat / np.linalg.norm(mat, axis=1).reshape((-1, 1))\n\n    \n    return mat\n\ndef lat_lon_feat2vec(df):\n    lat_lon_matrix = df[[\"latitude\",\"longitude\"]].values\n    vec = LLHtoECEF ( lat_lon_matrix[:,0], lat_lon_matrix[:,1] )\n    \n    return vec\n\n\ndef recall_knn_latlong (df, neighbors, total_neighbors, threshold  ):\n\n\n    \n    pairs = []\n    for country, country_df in tqdm(df.groupby('country')):\n        pairs_country = []\n        size = country_df.shape[0] \n        vec = lat_lon_feat2vec(country_df)\n        \n\n \n        total_neighbors_ = min ( size, total_neighbors)\n        neighbors_ = min(size, neighbors)\n   \n        if total_neighbors_ > 1:        \n            knn = NearestNeighbors(n_neighbors = total_neighbors_, metric = 'cosine',  algorithm=\"brute\", n_jobs = -1)\n          \n            knn.fit(vec)\n            dists, nears = knn.kneighbors(vec, return_distance = True)    \n            mean_1 = dists[:, :neighbors_].mean(axis=1)\n            mean_2 = dists[:, :total_neighbors_].mean(axis=1)\n            for k in range(0,total_neighbors_):            \n                cur_df = country_df[['id']]\n                cur_df['match_id'] = country_df['id'].values[nears[:, k]]\n                cur_df[f'kdist'] = dists[:, k]\n                cur_df[f'kneighbors'] = k\n                cur_df[f'kdist_mean_1'] = mean_1\n                cur_df[f'kdist_mean_2'] = mean_2\n                cur_df = cur_df.query(\"id != match_id  and (kneighbors < @neighbors_ or kdist < @threshold) \")\n\n                if len(cur_df) > 0:\n\n                    flag = cur_df[\"id\"] > cur_df[\"match_id\"]\n                    ids = cur_df[flag][\"id\"].values\n                    match_ids = cur_df[flag][\"match_id\"].values\n\n                    cur_df.loc[flag, \"id\"] = match_ids\n                    cur_df.loc[flag, \"match_id\"] = ids\n                    pairs_country.append(cur_df)\n                    \n            pairs_country = pd.concat (pairs_country)\n                    \n            pairs_country = pairs_country.groupby ( [\"id\",\"match_id\"]).agg ( {\n                                                                f'kneighbors':[\"mean\"],\n                                                                f'kdist_mean_1': [\"mean\"],\n                                                                f'kdist_mean_2': [\"mean\"],\n                                                                f'kdist':[\"count\"]\n                                                                   } ).reset_index() \n\n            pairs_country.columns =  ['_'.join(col) for col in pairs_country.columns.values]\n            pairs_country = pairs_country.rename(columns={\"id_\":\"id\", \"match_id_\":\"match_id\"})\n            pairs.append(pairs_country)\n\n\n    pairs = pd.concat(pairs)\n    \n    print(f\"recall_knn_latlong: {pairs.shape}\")\n    \n    return pairs\n\n\ndef recall_knn_vec ( df, neighbors, total_neighbors, threshold, vec_name ):\n\n    \n    name = pd.read_csv(f\"{TEXT2VEC_PATH}/{TEXT2VEC_PREFIX}{vec_name}.csv\")\n    with open(f'{TEXT2VEC_PATH}/{TEXT2VEC_PREFIX}{vec_name}.vec', 'rb') as handle:\n        name_vec = pickle.load(handle)\n    \n    d = name[vec_name].to_dict()\n    name2ids = { d[k]:k for k in d  }        \n    pairs = []\n    for country, country_df in tqdm(df.groupby('country')):\n        pairs_country = []\n        country_df = country_df[ country_df[vec_name].notnull() ]\n        names = country_df[vec_name].values\n        size = names.shape[0] \n        vec = np.zeros ((size, name_vec.shape[1]))\n        for i in range (size):\n            vec[i] = name_vec[ name2ids [names[i]] ,:] \n\n        total_neighbors_ = min ( size, total_neighbors)\n        neighbors_ = min(size, neighbors)\n        \n        if total_neighbors_ > 1:        \n            knn = NearestNeighbors(n_neighbors = total_neighbors_, metric = 'cosine',  algorithm=\"brute\", n_jobs = -1)\n            \n            #print (country, size, total_neighbors_, neighbors_)\n            \n            knn.fit(vec)\n            dists, nears = knn.kneighbors(vec, return_distance = True)    \n            mean_1 = dists[:, :neighbors_].mean(axis=1)\n            mean_2 = dists[:, :total_neighbors_].mean(axis=1)\n            \n            for k in range(0,total_neighbors_):            \n                cur_df = country_df[['id']]\n                cur_df['match_id'] = country_df['id'].values[nears[:, k]]\n                cur_df[f'xml_{vec_name}_kdist'] = dists[:, k]\n                cur_df[f'xml_{vec_name}_kneighbors'] = k\n                cur_df[f'xml_{vec_name}_kdist_mean_1'] = mean_1\n                cur_df[f'xml_{vec_name}_kdist_mean_2'] = mean_2\n                cur_df = cur_df.query(f\"id != match_id  and (xml_{vec_name}_kneighbors < @neighbors_ or xml_{vec_name}_kdist < @threshold) \")\n                if len(cur_df) > 0:\n                    flag = cur_df[\"id\"] > cur_df[\"match_id\"]\n                    ids = cur_df[flag][\"id\"].values\n                    match_ids = cur_df[flag][\"match_id\"].values\n\n                    cur_df.loc[flag, \"id\"] = match_ids\n                    cur_df.loc[flag, \"match_id\"] = ids\n                    pairs_country.append(cur_df)\n                    \n            pairs_country = pd.concat (pairs_country)\n                    \n            pairs_country = pairs_country.groupby ( [\"id\",\"match_id\"]).agg ( {\n                                                                f'xml_{vec_name}_kneighbors':[\"mean\"],\n                                                                f'xml_{vec_name}_kdist_mean_1': [\"mean\"],\n                                                                f'xml_{vec_name}_kdist_mean_2': [\"mean\"],\n                                                                f'xml_{vec_name}_kdist':[\"count\"]\n                                                                   } ).reset_index() \n\n            pairs_country.columns =  ['_'.join(col) for col in pairs_country.columns.values]\n            pairs_country = pairs_country.rename(columns={\"id_\":\"id\", \"match_id_\":\"match_id\", f'xml_{vec_name}_kdist': f\"xml_{vec_name}_count\"})\n            pairs.append(pairs_country)\n\n    pairs = pd.concat(pairs)\n    \n    print(f\"recall_knn_vec: {vec_name}: {pairs.shape}\")\n    \n    return pairs\n\ndef recall_knn_tfid ( df, neighbors, total_neighbors, col ):\n    tfidf = TfidfVectorizer()\n    tv_fit = tfidf.fit_transform(df[col].fillna('nan'))\n    pairs = []\n    for country, country_df in tqdm(df.groupby('country')):\n        pairs_country = []\n        country_df = country_df[ country_df[col].notnull() ]\n        size = len(country_df)\n        total_neighbors_ = min ( size, total_neighbors)\n        neighbors_ = min(size, neighbors)\n        \n        \n        \n        if total_neighbors_ > 1:        \n\n            knn = NearestNeighbors(n_neighbors = total_neighbors_, metric = 'cosine', algorithm=\"brute\", n_jobs = -1)\n            \n            #print (country, size, total_neighbors_, neighbors_)\n\n            \n            idx = country_df.index\n            knn.fit(tv_fit[idx])            \n            \n            dists, nears = knn.kneighbors(tv_fit[idx], return_distance = True)    \n            mean_1 = dists[:, :neighbors_].mean(axis=1)\n            mean_2 = dists[:, :total_neighbors_].mean(axis=1)\n            \n            for k in range(0,neighbors_):            \n                cur_df = country_df[['id']]\n                cur_df['match_id'] = country_df['id'].values[nears[:, k]]\n                cur_df[f'tfidf_{col}_kdist'] = dists[:, k]\n                cur_df[f'tfid_{col}_kneighbors'] = k\n                cur_df[f'tfid_{col}_kdist_mean_1'] = mean_1\n                cur_df[f'tfid_{col}_kdist_mean_2'] = mean_2\n                cur_df = cur_df.query(\"id != match_id\")\n                if len(cur_df) > 0:\n                    flag = cur_df[\"id\"] > cur_df[\"match_id\"]\n                    ids = cur_df[flag][\"id\"].values\n                    match_ids = cur_df[flag][\"match_id\"].values\n\n                    cur_df.loc[flag, \"id\"] = match_ids\n                    cur_df.loc[flag, \"match_id\"] = ids\n\n                    pairs_country.append(cur_df)\n            pairs_country = pd.concat (pairs_country)\n            pairs_country = pairs_country.groupby ( [\"id\",\"match_id\"]).agg ( {\n                                                            f'tfid_{col}_kneighbors':[\"mean\"],\n                                                            f'tfid_{col}_kdist_mean_1': [\"mean\"],\n                                                            f'tfid_{col}_kdist_mean_2': [\"mean\"],\n                                                            f'tfidf_{col}_kdist':[\"count\"]\n                                                               } ).reset_index() \n            pairs_country.columns =  ['_'.join(col) for col in pairs_country.columns.values]\n            pairs_country = pairs_country.rename(columns={\"id_\":\"id\", \"match_id_\":\"match_id\" })\n            pairs.append(pairs_country)\n            \n    pairs = pd.concat(pairs)\n    print(f\"recall_knn_tfid: {col}: {pairs.shape}\")\n    return pairs\n\ndef recall_knn_tfid_char ( df, neighbors, total_neighbors, col ):\n    tfidf = TfidfVectorizer(ngram_range=(3, 3), analyzer=\"char_wb\", use_idf=False)\n    tv_fit = tfidf.fit_transform(df[col].fillna('nan'))\n    pairs = []\n    for country, country_df in tqdm(df.groupby('country')):\n        pairs_country = []\n        country_df = country_df[ country_df[col].notnull() ]\n        size = len(country_df)\n        total_neighbors_ = min ( size, total_neighbors)\n        neighbors_ = min(size, neighbors)\n        \n        \n        \n        if total_neighbors_ > 1:        \n\n            knn = NearestNeighbors(n_neighbors = total_neighbors_, metric = 'cosine', algorithm=\"brute\", n_jobs = -1)\n            \n            #print (country, size, total_neighbors_, neighbors_)\n\n            \n            idx = country_df.index\n            knn.fit(tv_fit[idx])            \n            \n            dists, nears = knn.kneighbors(tv_fit[idx], return_distance = True)    \n            mean_1 = dists[:, :neighbors_].mean(axis=1)\n            mean_2 = dists[:, :total_neighbors_].mean(axis=1)\n            \n            for k in range(0,neighbors_):            \n                cur_df = country_df[['id']]\n                cur_df['match_id'] = country_df['id'].values[nears[:, k]]\n                cur_df[f'tfid_char_{col}_kdist'] = dists[:, k]\n                cur_df[f'tfid_char_{col}_kneighbors'] = k\n                cur_df[f'tfid_char_{col}_kdist_mean_1'] = mean_1\n                cur_df[f'tfid_char_{col}_kdist_mean_2'] = mean_2\n                cur_df = cur_df.query(\"id != match_id\")\n                if len(cur_df) > 0:\n                    flag = cur_df[\"id\"] > cur_df[\"match_id\"]\n                    ids = cur_df[flag][\"id\"].values\n                    match_ids = cur_df[flag][\"match_id\"].values\n\n                    cur_df.loc[flag, \"id\"] = match_ids\n                    cur_df.loc[flag, \"match_id\"] = ids\n\n                    pairs_country.append(cur_df)\n            pairs_country = pd.concat (pairs_country)\n            pairs_country = pairs_country.groupby ( [\"id\",\"match_id\"]).agg ( {\n                                                            f'tfid_char_{col}_kneighbors':[\"mean\"],\n                                                            f'tfid_char_{col}_kdist_mean_1': [\"mean\"],\n                                                            f'tfid_char_{col}_kdist_mean_2': [\"mean\"],\n                                                            f'tfid_char_{col}_kdist':[\"count\"]\n                                                               } ).reset_index() \n            pairs_country.columns =  ['_'.join(col) for col in pairs_country.columns.values]\n            pairs_country = pairs_country.rename(columns={\"id_\":\"id\", \"match_id_\":\"match_id\" })\n            pairs.append(pairs_country)\n            \n    pairs = pd.concat(pairs)\n    print(f\"recall_knn_tfid_char: {col}: {pairs.shape}\")\n    return pairs\n\n\ndef recall_knn(df, verbose=False):\n    \n    pairs = recall_knn_latlong (df, LATLONG_NEIGHBORS,  TOTAL_NEIGHBORS, LATLONG_THRESHOLD)\n\n    pairs = reduce_mem_usage(pairs)\n    print(f\"shape: {pairs.shape}\")\n\n    \n    #tfidf char name\n    data =  pd.read_feather( \"convert_data.feather\")\n    recall_df = recall_knn_tfid_char(data, NEIGHBORS, TOTAL_NEIGHBORS, \"name\")\n    recall_df = reduce_mem_usage(recall_df)\n    pairs = pairs.merge(recall_df, on=[\"id\",\"match_id\"], how=\"outer\")    \n    print(f\"tfidf_char name shape: {pairs.shape}\")\n    del data\n                       \n    #vec name\n    recall_df = recall_knn_vec(df, NEIGHBORS, TOTAL_NEIGHBORS_2, NAME_THRESHOLD, \"name\")\n    recall_df = reduce_mem_usage(recall_df)\n    pairs = pairs.merge(recall_df, on=[\"id\",\"match_id\"], how=\"outer\")\n    print(f\"vec name shape: {pairs.shape}\")\n\n    #vec address\n    recall_df = recall_knn_vec(df, NEIGHBORS, TOTAL_NEIGHBORS_2, ADDRESS_THRESHOLD, \"address\")\n    recall_df = reduce_mem_usage(recall_df)\n    pairs = pairs.merge(recall_df, on=[\"id\",\"match_id\"], how=\"outer\")\n    pairs = reduce_mem_usage(pairs)\n    print(f\"vec address shape: {pairs.shape}\")\n    \n    #vec categoris\n    \"\"\"\n    recall_df = recall_knn_vec(df,NEIGHBORS, TOTAL_NEIGHBORS_2, CATEGORIES_THRESHOLD, \"categories\")\n    recall_df = reduce_mem_usage(recall_df)\n    pairs = pairs.merge(recall_df, on=[\"id\",\"match_id\"], how=\"outer\")\n\n    pairs = reduce_mem_usage(pairs)\n    print(f\"vec categories shape: {pairs.shape}\")\n    \"\"\"\n    \n\n    print(pairs.dtypes)\n\n    \n    return pairs\n\ndef generate_pairs (data):\n    \n    train_data = recall_knn(data, verbose=False)\n    \n\n    #generate target\n    if \"point_of_interest\" in data:\n        data = data.set_index('id')\n        ids = train_data['id'].tolist()\n        match_ids = train_data['match_id'].tolist()\n\n        poi = data.loc[ids]['point_of_interest'].values\n        match_poi = data.loc[match_ids]['point_of_interest'].values\n\n        train_data['label'] = np.array(poi == match_poi, dtype = np.int8)\n    \n    return train_data\n\ndef reduce_mem_usage(df, verbose=True):\n\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2\n\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose:\n        print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n        print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df\n\ndef main():\n    data = pd.read_feather(\"data.feather\")\n\n\n    pairs = generate_pairs (data)\n\n\n    print(f\"pairs shape: {pairs.shape}\")\n\n    pairs = pairs.reset_index(drop=True)\n    print(\"save\")\n    pairs.to_feather(f\"pairs.feather\")\n    \n    \nSEED = 2022\nseed_everything(SEED)\nTOTAL_NEIGHBORS = 100\nTOTAL_NEIGHBORS_2 = 50\nLATLONG_NEIGHBORS = 12\nLATLONG_THRESHOLD = 0.000015\nNAME_THRESHOLD = 0.00099\nADDRESS_THRESHOLD = 0.00492\n#CATEGORIES_THRESHOLD = 0.00001\nNEIGHBORS = 10\nTEXT2VEC_PATH = \".\"\nTEXT2VEC_PREFIX = \"sentence-transformers_paraphrase-xlm-r-multilingual-v1_\"\n\n\nif __name__ == \"__main__\":\n    main()","metadata":{"execution":{"iopub.status.busy":"2022-06-08T08:19:40.627519Z","iopub.execute_input":"2022-06-08T08:19:40.627946Z","iopub.status.idle":"2022-06-08T08:19:40.640854Z","shell.execute_reply.started":"2022-06-08T08:19:40.627903Z","shell.execute_reply":"2022-06-08T08:19:40.639824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n!python generate_pairs.py","metadata":{"execution":{"iopub.status.busy":"2022-06-08T08:19:40.644054Z","iopub.execute_input":"2022-06-08T08:19:40.644737Z","iopub.status.idle":"2022-06-08T08:20:01.633858Z","shell.execute_reply.started":"2022-06-08T08:19:40.644663Z","shell.execute_reply":"2022-06-08T08:20:01.632827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Extraction","metadata":{}},{"cell_type":"code","source":"%%writefile generate_fe.py\nimport warnings\nwarnings.filterwarnings('ignore')\nimport os\n\nimport random\nimport pandas as pd\nimport numpy as np\nimport pickle\n\nfrom tqdm import tqdm\n\nimport torch\nif torch.cuda.is_available():\n    DEVICE = \"cuda\"\nelse:\n    DEVICE = \"cpu\"\n\n\ndef seed_everything(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)    \n    \n\ndef reduce_mem_usage(df, verbose=True):\n\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2\n\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    \n    if verbose:\n        print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n        print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df\n\ndef cosine_similarity (a,b):\n    \"\"\"\n    a /= np.linalg.norm(a, axis=1).reshape((-1, 1))\n    b /= np.linalg.norm(b, axis=1).reshape((-1, 1))\n\n    cos_sim = np.sum(a*b, axis=1)\n    \"\"\"\n    \n    \n    t1 = torch.from_numpy(a).to(DEVICE)\n    t2 = torch.from_numpy(b).to(DEVICE)\n    cos = torch.nn.CosineSimilarity(dim=1, eps=1e-6)\n    output = cos(t1, t2)\n    return output.to('cpu').numpy()\n\ndef lat_lon_cos_sim (pairs, inv_d, path):\n\n    file = open(path,'rb')\n    lat_lon_vec = pickle.load(file)\n\n    ids1_x = pairs[\"id\"].map(inv_d)\n    ids2_x = pairs[\"match_id\"].map(inv_d)\n\n\n    a = lat_lon_vec[ids1_x]\n    b = lat_lon_vec[ids2_x]\n    #cos_sim = np.sum(a*b, axis=1)\n    #return cos_sim\n    return cosine_similarity(a,b)\n\ndef get_vec (feat_series, feat_name, prefix  ):\n    \n    df_feat = pd.read_csv(f'{prefix}{feat_name}.csv')\n    file = open(f'{prefix}{feat_name}.vec','rb')\n    feat_vec = pickle.load(file)\n\n    \n    d = df_feat[feat_name].to_dict()\n    inv_d = {v: k for k, v in d.items()}\n\n    feat_values = feat_series.values\n\n    vec = np.zeros((feat_values.shape[0], feat_vec.shape[1]))\n    for i in range(vec.shape[0]):\n        vec[i,:] = feat_vec[ inv_d[feat_values[i]], : ]\n    return vec\n\ndef feat_cos_sim(feat_name,  prefix , feat_series , inv_d, pairs, batch_size):\n    vec = get_vec ( feat_series,feat_name = feat_name, prefix = prefix )    \n    step = len(pairs)//batch_size\n    if batch_size*step < len(pairs):\n        step += 1\n\n    ret = []\n    start = 0\n    for i in tqdm(range(step)):\n        end = start + batch_size\n\n        ids1_x = pairs[start:end][\"id\"].map(inv_d).values\n        ids2_x = pairs[start:end][\"match_id\"].map(inv_d).values\n\n        a = vec[ids1_x] \n        b = vec[ids2_x] \n\n        \n        \"\"\"\n        a /= np.linalg.norm(a, axis=1).reshape((-1, 1))\n        b /= np.linalg.norm(b, axis=1).reshape((-1, 1))\n\n        cos_sim = np.sum(a*b, axis=1)\n        \"\"\"\n\n        cos_sim = cosine_similarity(a,b)\n\n        \n        ret.append(cos_sim)\n        start += batch_size\n    \n    return np.concatenate(ret)\n\n\ndef generate_fe (path, batch_size = 100_000):\n    data = pd.read_feather(\"data.feather\")\n    d = data[\"id\"].to_dict()\n    # train[\"id\"] --> train.index\n    inv_d = {v: k for k, v in d.items()}\n    data.index = data[\"id\"]\n\n    pairs = pd.read_feather(\"pairs.feather\")\n    if DEBUG:\n        pairs = pairs [:1000]\n    \n    if \"label\" in pairs:\n        targets = pairs[\"label\"].values\n        pairs = pairs.drop([\"label\"], axis = 1)\n    else:\n        targets = None\n    \n    \n    pairs = pairs.rename(columns={\"id_count\":\"kdist_count\"})\n\n    for feat_name in [\"address\", \"city\", \"categories\"]:\n        print(feat_name)\n        d = data[feat_name].notnull().to_dict()\n        pairs[f\"{feat_name}_id_notnull\"] = pairs[\"id\"].map(d).astype(np.int8)\n        pairs[f\"{feat_name}_match_id_notnull\"] = pairs[\"match_id\"].map(d).astype(np.int8)\n    pairs = reduce_mem_usage (pairs)\n\n    print(\"lat_lon_cs\")\n    cos_sim = lat_lon_cos_sim (pairs, inv_d, path=f\"{INPUT_VEC}/lat_lon.vec\")\n    pairs[\"lat_lon_cs\"] = cos_sim\n    pairs = reduce_mem_usage (pairs)\n\n    postfix = \"xml\"\n    prefix = \"./sentence-transformers_paraphrase-xlm-r-multilingual-v1_\"\n\n    for feat_name in [ \"name\", \"address\", \"categories\", \"city\" ]:\n        print(feat_name)\n        cos_sim = feat_cos_sim (feat_name = feat_name, prefix = prefix , \n                                feat_series = data[feat_name].fillna(\"unknow\"), \n                                inv_d = inv_d, \n                                pairs = pairs, \n                                batch_size = batch_size)\n        pairs[f\"{feat_name}_{postfix}_cs\"] = cos_sim\n        pairs = reduce_mem_usage ( pairs , verbose=True)\n\n    postfix = \"mpnet-ml\"\n    path = \"./sentence-transformers_paraphrase-multilingual-mpnet-base-v2_\"\n\n    for feat_name in [ \"name\", \"address\", \"categories\", \"city\" ]:\n        print(feat_name)\n        cos_sim = feat_cos_sim (feat_name = feat_name, prefix = prefix , \n                                feat_series = data[feat_name].fillna(\"unknow\"), \n                                inv_d = inv_d, \n                                pairs = pairs, \n                                batch_size = batch_size)\n        pairs[f\"{feat_name}_{postfix}_cs\"] = cos_sim\n        pairs = reduce_mem_usage ( pairs , verbose=True)\n\n    postfix = \"mpnet\"\n    path = \"./sentence-transformers_all-mpnet-base-v2_\"\n\n    for feat_name in [ \"name\", \"address\", \"categories\", \"city\" ]:\n        print(feat_name)\n        cos_sim = feat_cos_sim (feat_name = feat_name, prefix = prefix, \n                                feat_series = data[feat_name].fillna(\"unknow\"), \n                                inv_d = inv_d, \n                                pairs = pairs, \n                                batch_size = batch_size)\n        pairs[f\"{feat_name}_{postfix}_cs\"] = cos_sim\n        pairs = reduce_mem_usage ( pairs , verbose=True)\n\n    if not targets is None:\n        pairs[\"label\"] = targets\n        \n    return pairs \n\n\n\nSEED = 2022\nseed_everything(SEED)\n\nINPUT_PATH = \"./\"\nINPUT_VEC = \"./\"\nDEBUG = False\n\ndef main():\n    pairs = generate_fe(path = INPUT_PATH,  batch_size = 200_000)\n    pairs.to_feather(f\"pairs.feather\")\n    \nif __name__ == \"__main__\":\n    main()","metadata":{"execution":{"iopub.status.busy":"2022-06-08T08:20:01.635912Z","iopub.execute_input":"2022-06-08T08:20:01.636296Z","iopub.status.idle":"2022-06-08T08:20:01.646842Z","shell.execute_reply.started":"2022-06-08T08:20:01.636253Z","shell.execute_reply":"2022-06-08T08:20:01.64606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n!python generate_fe.py","metadata":{"execution":{"iopub.status.busy":"2022-06-08T08:20:01.648554Z","iopub.execute_input":"2022-06-08T08:20:01.649199Z","iopub.status.idle":"2022-06-08T08:20:05.303134Z","shell.execute_reply.started":"2022-06-08T08:20:01.649161Z","shell.execute_reply":"2022-06-08T08:20:05.302008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_features()","metadata":{"execution":{"iopub.status.busy":"2022-06-08T08:20:05.305488Z","iopub.execute_input":"2022-06-08T08:20:05.306167Z","iopub.status.idle":"2022-06-08T08:20:05.310695Z","shell.execute_reply.started":"2022-06-08T08:20:05.30612Z","shell.execute_reply":"2022-06-08T08:20:05.309627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Extraction EXT","metadata":{}},{"cell_type":"code","source":"%%writefile generate_fe_ext.py\nimport pandas as pd\nimport numpy as np\n\nimport os\nimport gc\nimport random\nimport Levenshtein\nimport difflib\n\nfrom tqdm import tqdm\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n\nimport joblib\n\ndef calculate_jaccard_char(str1, str2):\n    \n    # Combine both tokens to find union.\n    both_tokens = str1 + str2\n    union = set(both_tokens)\n    if len(union) == 0:\n        return 0\n    \n    # Calculate intersection.\n    intersection = set()\n    for w in set(str1):\n        if w in set(str2):\n            intersection.add(w)\n\n    jaccard_score = len(intersection)/len(union)\n    \n    return jaccard_score\n\ndef calculate_jaccard_word(str1, str2):\n    \n    # Combine both tokens to find union.\n    words1 = str1.split()\n    words2 = str2.split()\n    union = set(words1 + words2)\n    if len(union) == 0:\n        return 0\n    \n    # Calculate intersection.\n    intersection = set()\n    for word in union:\n        if word in words1 and word in words2:\n            intersection.add(word)\n\n    jaccard_score = len(intersection)/len(union)\n    \n    return jaccard_score\n    \ndef calculate_jaccard_word_smallest(str1, str2):\n    \n    if str1 == str2:\n        return 1\n    \n    # Combine both tokens to find union.\n    words1 = str1.split()\n    words2 = str2.split()\n    union = set(words1 + words2)\n    small = min(len(set(words1)), len(set(words2)))\n    if small == 0:\n        return 0\n    \n    # Calculate intersection.\n    intersection = set()\n    for word in union:\n        if word in words1 and word in words2:\n            intersection.add(word)\n\n    jaccard_score = len(intersection)/small\n    \n    return jaccard_score\n\n\ndef reduce_mem_usage(df, verbose=True):\n\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2\n\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose:\n        print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n        print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df\n\n\n\ndef generate_fe_ext (train_pairs, batch_size ):\n    def process_tfidf (col):\n        \n        train = pd.read_feather(f\"col_{col}.feather\")\n        \n        tfidf = TfidfVectorizer()\n        tv_fit = tfidf.fit_transform(train[col].fillna('nan'))\n        a = np.array(tv_fit[indexs].multiply(tv_fit[match_indexs]).sum(axis = 1)).ravel().astype(np.float16)\n        np.save(f\"{col}_tfid.npy\", a)\n        \n        del train, tfidf, a\n        gc.collect()    \n\n    def process_tfidf_char (col):\n        \n        train = pd.read_feather(f\"conv_col_{col}.feather\")\n        \n        tfidf = TfidfVectorizer(ngram_range=(3, 3), analyzer=\"char_wb\", use_idf=False)\n        tv_fit = tfidf.fit_transform(train[col].fillna('nan'))        \n\n        a = np.array(tv_fit[indexs].multiply(tv_fit[match_indexs]).sum(axis = 1)).ravel().astype(np.float16)\n        np.save(f\"conv_{col}_tfid.npy\", a)\n        \n        del train, tfidf, a\n        gc.collect()    \n        \n        \n\n    def process_ext(col):\n        \n        with open(\"log.txt\", 'a') as f:\n            f.write(f\"process_ext: start {col}\\n\")\n        \n        train = pd.read_feather(f\"col_{col}.feather\")\n\n        ret_gesh = []\n        ret_levens = []\n        ret_jaros = []\n        ret_lcs = []\n        ret_jaccard_words = []\n        ret_jaccard_chars = []        \n\n\n        start = 0\n        for i in range(step):\n            gesh = []\n            levens = []\n            jaros = []\n            lcs = []\n            jaccard_words = []\n            jaccard_chars = []\n\n            end = start + batch_size          \n            id_values = train.loc[indexs[start:end]][col].values.astype(str)\n            match_id_values = train.loc[match_indexs[start:end]][col].values.astype(str)\n\n\n            \n            for s, match_s in zip(id_values, match_id_values):\n                if s != 'nan' and match_s != 'nan':                    \n                    sm = difflib.SequenceMatcher(None, s, match_s)\n                    \n                    gesh.append(sm.ratio())\n                    levens.append(Levenshtein.distance(s, match_s))\n                    jaros.append(Levenshtein.jaro_winkler(s, match_s))                    \n                    lcs.append(sm.find_longest_match(0,len(s), 0,len(match_s)).size)\n                    \n                    if col in [\"name\", \"address\", \"categories\"]:\n                        jaccard_words.append(calculate_jaccard_word(s, match_s))\n                        jaccard_chars.append(calculate_jaccard_char(s, match_s))                    \n                else:\n                    gesh.append(np.nan)\n                    levens.append(np.nan)\n                    jaros.append(np.nan)\n                    lcs.append(np.nan)\n                    if col in [\"name\", \"address\", \"categories\"]:\n                        jaccard_words.append(np.nan)\n                        jaccard_chars.append(np.nan)        \n                    \n            start += batch_size            \n            \n            ret_gesh.append(np.array(gesh).astype(np.float16))\n            ret_levens.append(np.array(levens).astype(np.float16))\n            ret_jaros.append(np.array(jaros).astype(np.float16))\n            ret_lcs.append(np.array(lcs).astype(np.float16))\n            del gesh, levens, jaros, lcs\n\n            if col in [\"name\", \"address\", \"categories\"]:\n                ret_jaccard_words.append(np.array(jaccard_words).astype(np.float16))\n                ret_jaccard_chars.append(np.array(jaccard_chars).astype(np.float16))\n            \n            del jaccard_words, jaccard_chars\n            gc.collect()\n            \n        np.save(f\"{col}_gesh.npy\", np.concatenate(ret_gesh))\n        np.save(f\"{col}_levens.npy\", np.concatenate(ret_levens))\n        np.save(f\"{col}_jeros.npy\", np.concatenate(ret_jaros))\n        np.save(f\"{col}_lcs.npy\", np.concatenate(ret_lcs))\n        del train, ret_gesh, ret_levens, ret_jaros, ret_lcs\n\n        if col in [\"name\", \"address\", \"categories\"]:        \n            np.save(f\"{col}_jaccard_words.npy\", np.concatenate(ret_jaccard_words))\n            np.save(f\"{col}_jaccard_chars.npy\", np.concatenate(ret_jaccard_chars))        \n            del ret_jaccard_words, ret_jaccard_chars\n        \n        gc.collect()            \n        with open(\"log.txt\", \"a\") as f:\n            f.write(f\"process_ext: end {col}\\n\")\n        \n        \n    \n    train = pd.read_feather(\"data.feather\")\n    \n    #train_pairs = train_pairs.drop([\"xml_name_kdist\", \"xml_address_kdist\", \"xml_categories_kdist\"], axis=1)\n    #train_pairs = reduce_mem_usage (train_pairs, verbose = True)\n    tfidf_cols = [\"name\", \"address\", \"categories\", 'state', 'zip', 'url', 'phone' ] \n    tfidf_conv_cols = [\"name\", \"address\", \"city\", \"state\"] \n    feat_cols = [\"name\", \"address\", \"categories\", 'state', 'zip', 'url', 'phone']\n    \n    \n    d = train[\"id\"].to_dict()\n    # train[\"id\"] --> train.index\n    inv_d = {v: k for k, v in d.items()}\n\n    indexs = [inv_d[i] for i in train_pairs['id']]\n    match_indexs = [inv_d[i] for i in train_pairs['match_id']]\n    del d, inv_d \n    gc.collect()\n    \n    print(\"save cols\")\n    for col in set(tfidf_cols+feat_cols):\n        print(col)\n        train[[col]].to_feather(f\"col_{col}.feather\")\n    \n    del train\n    gc.collect()    \n    \n    train = pd.read_feather(\"convert_data.feather\")\n    print(\"save conv cols\")\n    for col in set(tfidf_conv_cols):\n        print(col)\n        train[[col]].to_feather(f\"conv_col_{col}.feather\")\n    \n    del train\n    gc.collect()        \n\n    \n    print(\"TF-IDF char feats\")\n    joblib.Parallel(n_jobs=2, timeout=9999999)(joblib.delayed(process_tfidf_char)(col) for col in tfidf_conv_cols)  \n    \n    \n    print(\"TF-IDF feats\")\n    joblib.Parallel(n_jobs=2, timeout=9999999)(joblib.delayed(process_tfidf)(col) for col in tfidf_cols)  \n\n\n    print(\"Distance feats\")\n    #train.index = train[\"id\"]\n\n    step = train_pairs.shape[0]//batch_size\n    if batch_size*step < train_pairs.shape[0]:\n        step += 1\n    \n    joblib.Parallel(n_jobs=2, timeout=9999999)(joblib.delayed(process_ext)(col) for col in feat_cols)\n\n    del indexs, match_indexs\n    gc.collect()    \n\n    print(\"loading TF-IDF conv feats\")\n    for col in tfidf_conv_cols:\n        print(f\"loading {col}\" )\n        train_pairs[f'conv_{col}_tfid'] =  np.load(f'conv_{col}_tfid.npy')\n    train_pairs = reduce_mem_usage (train_pairs, verbose = False)    \n    \n    print(\"loading TF-IDF feats\")\n    for col in tfidf_cols:\n        print(f\"loading {col}\" )\n        train_pairs[f'{col}_tfid'] =  np.load(f'{col}_tfid.npy')\n    train_pairs = reduce_mem_usage (train_pairs, verbose = False)\n    \n    print(\"loading Distance feats\")\n    for col in feat_cols:\n\n        print(f\"loading {col}\" )\n        \n        train_pairs[f\"{col}_gesh\"] = np.load(f\"{col}_gesh.npy\")\n        train_pairs[f\"{col}_levens\"] =  np.load(f\"{col}_levens.npy\")\n        train_pairs[f\"{col}_jeros\"] =  np.load(f\"{col}_jeros.npy\")\n        train_pairs[f\"{col}_lcs\"] =  np.load(f\"{col}_lcs.npy\")   \n    \n        if col in [\"name\", \"address\", \"categories\"]:       \n            train_pairs[f\"{col}_jaccard_words\"] =  np.load(f\"{col}_jaccard_words.npy\")\n            train_pairs[f\"{col}_jaccard_chars\"] =  np.load(f\"{col}_jaccard_chars.npy\")               \n        train_pairs = reduce_mem_usage (train_pairs, verbose=False)   \n            \n    train = pd.read_feather(\"data.feather\")            \n    train.index = train[\"id\"]         \n    print(\"create len feats\")        \n    for col in feat_cols:\n        print(col)        \n        train[f\"len_{col}\"]=train[col].fillna(\"\").map(lambda x:len(x))\n        d=train[f\"len_{col}\"].to_dict()\n        train_pairs[f\"{col}_len_diff\"] = np.abs( train_pairs[\"id\"].map(d) - train_pairs[\"match_id\"].map(d)  )    \n\n\n    country_size = train.groupby(\"country\").size()\n    train.index=train[\"id\"]\n    d=train[\"country\"].to_dict()\n\n    train_pairs[\"country_id_size\"] = train_pairs[\"id\"].map(lambda x: country_size[d[x]])/150_000\n    del d\n    train_pairs = reduce_mem_usage (train_pairs, verbose = True)\n\n    print(\"lat/long feats\")\n    d_train_lat = train[\"latitude\"].to_dict()\n    d_train_long = train[\"longitude\"].to_dict()\n\n    train_pairs[\"lat_id\"] = train_pairs[\"id\"].map(d_train_lat)\n    train_pairs[\"lat_match_id\"] = train_pairs[\"match_id\"].map(d_train_lat)\n    train_pairs[\"long_id\"] = train_pairs[\"id\"].map(d_train_long)\n    train_pairs[\"long_match_id\"] = train_pairs[\"match_id\"].map(d_train_long)\n    \n    del d_train_lat, d_train_long\n    train_pairs[\"euclidean\"] = np.sqrt(((train_pairs['lat_id'] - train_pairs['lat_match_id'])**2) + ((train_pairs['long_id'] - train_pairs['long_match_id'])**2))\n    train_pairs[\"diff_lat\"] = np.abs(train_pairs['lat_id'] - train_pairs['lat_match_id'])\n    train_pairs[\"diff_long\"] = np.abs(train_pairs['long_match_id'] - train_pairs['long_id'])    \n    \n    train_pairs = reduce_mem_usage (train_pairs, verbose = True)\n    gc.collect()\n    \n    print(\"size feats\")    \n   \n    for feat in [\"name\", \"address\", \"url\"]:\n        print(feat)\n        df = train.groupby([feat,\"country\"]).agg({\"id\":\"count\"}).reset_index()\n        df = df.query(\"id >1\").reset_index()\n        df = df.rename(columns={\"id\":f\"{feat}_country_size\"})\n        df = train.merge( df, on = feat, how=\"left\" )\n        df[f\"{feat}_country_size\"] =   df[f\"{feat}_country_size\"].fillna(1)/10_000\n        df.index = df[\"id\"]    \n        d = df[f\"{feat}_country_size\"].to_dict()\n        del df\n        train_pairs[f\"{feat}_country_size_id\"] = train_pairs[\"id\"].map(d)\n        train_pairs[f\"{feat}_country_size_match_id\"] = train_pairs[\"match_id\"].map(d)\n        del d    \n        train_pairs = reduce_mem_usage (train_pairs, verbose = True)\n        gc.collect()    \n\n        df = train.groupby([feat]).agg({\"id\":\"count\"}).reset_index()\n        df = df.query(\"id >1\").reset_index()\n        df = df.rename(columns={\"id\":f\"{feat}_size\"})\n        df = train.merge( df, on = feat, how=\"left\" )\n        df[f\"{feat}_size\"] =   df[f\"{feat}_size\"].fillna(1)/10_000\n        df.index = df[\"id\"]    \n        d = df[f\"{feat}_size\"].to_dict()\n        del df\n        train_pairs[f\"{feat}_size_id\"] = train_pairs[\"id\"].map(d)\n        train_pairs[f\"{feat}_size_match_id\"] = train_pairs[\"match_id\"].map(d)\n        del d    \n        train_pairs = reduce_mem_usage (train_pairs, verbose = True)\n        gc.collect()    \n    \n    \n    return train_pairs\n\ndef main():\n    DEBUG = False\n    pairs  = pd.read_feather(\"pairs.feather\")\n    if DEBUG :\n        pairs = pairs [:66333]\n\n\n    pairs = generate_fe_ext (pairs, batch_size = 50_000)\n    pairs.to_feather (\"pairs.feather\")\n    \nif __name__ == \"__main__\":\n    main()","metadata":{"execution":{"iopub.status.busy":"2022-06-08T08:20:05.312399Z","iopub.execute_input":"2022-06-08T08:20:05.313194Z","iopub.status.idle":"2022-06-08T08:20:05.32721Z","shell.execute_reply.started":"2022-06-08T08:20:05.313165Z","shell.execute_reply":"2022-06-08T08:20:05.326471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n!python generate_fe_ext.py","metadata":{"execution":{"iopub.status.busy":"2022-06-08T08:20:05.330113Z","iopub.execute_input":"2022-06-08T08:20:05.330976Z","iopub.status.idle":"2022-06-08T08:20:09.659705Z","shell.execute_reply.started":"2022-06-08T08:20:05.330895Z","shell.execute_reply":"2022-06-08T08:20:09.658169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference","metadata":{}},{"cell_type":"code","source":"%%writefile predict.py\nimport pandas as pd\nimport numpy as np\nimport lightgbm as lgbm\nfrom collections import defaultdict\nfrom itertools import combinations\nfrom tqdm import tqdm\nimport gc\n\ndef reduce_mem_usage(df, verbose=True):\n\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2\n    if verbose:\n        print('Memory usage before optimization is: {:.2f} MB'.format(start_mem))\n\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n    \n    if verbose:\n        end_mem = df.memory_usage().sum() / 1024**2\n        print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n        print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df\n\ndef predict_xgb (model_name, pairs, cols, batch_size = 64):\n    import torch\n    from tqdm import tqdm\n    if torch.cuda.is_available():\n        from cuml import ForestInference\n        model = ForestInference.load(filename=model_name,model_type='xgboost')\n        step = pairs.shape[0]//batch_size\n        if batch_size*step < pairs.shape[0]:\n            step += 1\n        ret = []\n        start = 0\n        for i in tqdm(range(step)):\n            end = start + batch_size        \n            \n            X = pairs[start:end][cols].values\n            \n            pred = model.predict(X)\n            \n            ret.append(np.asarray(pred).astype(np.float16))\n            start += batch_size\n        pred = np.concatenate(ret)       \n        \n        \n    else:\n        model = lgbm.Booster(model_file=model_name)\n        step = pairs.shape[0]//batch_size\n        if batch_size*step < pairs.shape[0]:\n            step += 1\n        ret = []\n        start = 0\n        for i in tqdm(range(step)):\n            end = start + batch_size        \n            \n            X = pairs[start:end][cols].values\n            \n            pred = model.predict(X)\n            \n            ret.append(np.asarray(pred).astype(np.float16))\n            start += batch_size\n        pred = np.concatenate(ret)               \n        \n        \n\n        \n    #print(pred.shape, type(pred))    \n        \n    return pred\n\n\ndef predict_lgb (model_name, pairs, cols, batch_size = 64):\n    import torch\n    from tqdm import tqdm\n    if torch.cuda.is_available():\n        from cuml import ForestInference\n        model = ForestInference.load(filename=model_name,model_type='lightgbm')\n        step = pairs.shape[0]//batch_size\n        if batch_size*step < pairs.shape[0]:\n            step += 1\n        ret = []\n        start = 0\n        for i in tqdm(range(step)):\n            end = start + batch_size        \n            \n            X = pairs[start:end][cols].values\n            \n            pred = model.predict(X)\n            \n            ret.append(np.asarray(pred).astype(np.float16))\n            start += batch_size\n        pred = np.concatenate(ret)       \n        \n        \n    else:\n        model = lgbm.Booster(model_file=model_name)\n        step = pairs.shape[0]//batch_size\n        if batch_size*step < pairs.shape[0]:\n            step += 1\n        ret = []\n        start = 0\n        for i in tqdm(range(step)):\n            end = start + batch_size        \n            \n            X = pairs[start:end][cols].values\n            \n            pred = model.predict(X)\n            \n            ret.append(np.asarray(pred).astype(np.float16))\n            start += batch_size\n        pred = np.concatenate(ret)               \n        \n        \n\n        \n    #print(pred.shape, type(pred))    \n        \n    return pred\n\ndef create_submission (data, pairs, threshold ):\n    probs = {}\n\n    def get_probs(a,b):\n        if a == b:\n            return 1.0\n        if a < b:\n            return probs[(a,b)]\n        else:\n            return probs[(b,a)]\n\n    \n    print (\"create_submission\")\n    pairs = pairs.append ( pairs.rename(columns = {\"id\":\"match_id\", \"match_id\":\"id\"}) )\n    pairs = pairs.groupby([\"id\",\"match_id\"]).agg({\"pred\":\"mean\"}).reset_index()\n    pairs = pairs[pairs[\"pred\"] > threshold ]\n    \n    \n    print (\"load probs\")\n    \n    for a, b, c in tqdm(zip ( pairs[\"id\"].values, pairs[\"match_id\"].values, pairs[\"pred\"].values ), total=len(pairs)):\n        if a < b:\n            probs[(a,b)] = c\n        else:\n            probs[(b,a)] = c\n    \n    ids = data[\"id\"].unique()\n    id_df = pd.DataFrame ({\n        \"id\": ids,\n        \"match_id\":ids\n    })\n\n    \n    submission = pairs[['id', 'match_id']]\n    submission = submission.append ( id_df )\n\n    print (\"create list\")\n\n    submission = submission.groupby('id')['match_id'].apply(list).reset_index()\n    submission[\"match_id\"] = submission[\"match_id\"].map(lambda x: list(set(x)))\n    \n    sol = defaultdict(list)\n\n    mat = submission[[\"id\",\"match_id\"]].values\n    for k in tqdm(range ( mat.shape[0] )):\n        id = mat[k,0]\n        matches = mat[k,1]\n\n        #first level (a,b) -> (b,a)\n        for x in matches:\n            if not x in sol[id]:\n                sol[id].append(x)\n            if not id in sol[x]:\n                sol[x].append(id)\n\n        #second level (a,b), (a,c) -> (b,c) ?\n        matches.remove(id)\n        if len(matches) > 2:\n            for x, y in combinations(matches, 2):\n                    if y not in sol[x]:\n                        sol[x].append(y)\n                    if x not in sol[y]:\n                        sol[y].append(x)    \n\n    submission[\"match_id\"] = submission[\"id\"].map(sol)\n    submission['match_id'] = submission['match_id'].map(lambda x: list(set(x)))\n    submission['matches'] = submission['match_id'].apply(lambda x: ' '.join(set(x)))\n\n    return submission [[\"id\", \"matches\"]]\n\n\n\nXGB_MODELS = [\n    \n    \"../input/flm-xgb-99h-v06a-sample-50\",\n    \"../input/flm-xgb-99h-v06b-sample-50\",\n    \"../input/flm-xgb-99h-v06c-sample-50\",\n    \"../input/flm-xgb-99h-v06d-sample-50\",    \n    \n    \"../input/flm-xgb-99h-v06a-sample-65-s01\",\n    \"../input/flm-xgb-99h-v06b-sample-65-s01\",\n    \"../input/flm-xgb-99h-v06c-sample-65-s01\",\n    \"../input/flm-xgb-99h-v06d-sample-65-s01\",\n    \n    \"../input/flm-xgb-99h-v06a-sample-65-s02\",\n    \"../input/flm-xgb-99h-v06b-sample-65-s02\",\n    \"../input/flm-xgb-99h-v06c-sample-65-s02\",\n    \"../input/flm-xgb-99h-v06d-sample-65-s02\",\n\n    \n    \"../input/flm-xgb-99h-v06a-sample-65\",\n    \"../input/flm-xgb-99h-v06b-sample-65\",\n    \"../input/flm-xgb-99h-v06c-sample-65\",\n    \"../input/flm-xgb-99h-v06d-sample-65\",\n]\n\n\nLGB_MODELS = [\n]\n\nBATCH_SIZE = 80_000\nTHRESHOLD = 0.65\n\nSIZE = len(XGB_MODELS + LGB_MODELS)\n\n\n\ndef main ():\n    print(\"read pairs\")\n    test_pairs = pd.read_feather(\"pairs.feather\")\n\n\n    \n    \n    print(\"drop id, match_id\")  \n\n    test_pairs = test_pairs.drop( ['id', 'match_id'], axis=1)\n    cols = [ c for c in  test_pairs.columns if c not in (\"label\", \"index\") ]\n\n\n    print(\"predict pairs\")\n    \n    \n    for n, model in enumerate (XGB_MODELS): \n        p = predict_xgb (f\"{model}/model.txt\", test_pairs, cols, batch_size = BATCH_SIZE)\n        if n == 0:\n            pred = p/SIZE\n        else:\n            pred += p/SIZE\n\n    for n, model in enumerate (LGB_MODELS): \n        p = predict_lgb (f\"{model}/model.txt\", test_pairs, cols, batch_size = BATCH_SIZE)\n        pred += p/SIZE\n\n    \n    del test_pairs\n    gc.collect()    \n    \n    \n    print(\"read pairs\")\n    test_pairs = pd.read_feather(\"pairs.feather\")\n    test_pairs = test_pairs [[\"id\", \"match_id\"]]\n    test_pairs[\"pred\"] = pred\n\n    print(\"read data\")\n    test = pd.read_feather(\"data.feather\")\n    \n    submission  = create_submission (test, test_pairs, THRESHOLD)\n    \n    submission.to_csv(\"submission.csv\", index=False)\n    \n\nif __name__ == \"__main__\":\n    main()","metadata":{"execution":{"iopub.status.busy":"2022-06-08T08:20:09.661735Z","iopub.execute_input":"2022-06-08T08:20:09.662179Z","iopub.status.idle":"2022-06-08T08:20:09.670694Z","shell.execute_reply.started":"2022-06-08T08:20:09.662134Z","shell.execute_reply":"2022-06-08T08:20:09.669673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python predict.py","metadata":{"execution":{"iopub.status.busy":"2022-06-08T08:20:09.672459Z","iopub.execute_input":"2022-06-08T08:20:09.672821Z","iopub.status.idle":"2022-06-08T08:20:19.326183Z","shell.execute_reply.started":"2022-06-08T08:20:09.672787Z","shell.execute_reply":"2022-06-08T08:20:19.325211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm *.npy *.vec *.py","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsubmission = pd.read_csv(\"submission.csv\")\n\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-06-08T08:20:19.327777Z","iopub.execute_input":"2022-06-08T08:20:19.328165Z","iopub.status.idle":"2022-06-08T08:20:19.355001Z","shell.execute_reply.started":"2022-06-08T08:20:19.328123Z","shell.execute_reply":"2022-06-08T08:20:19.354199Z"},"trusted":true},"execution_count":null,"outputs":[]}]}