{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from sklearnex import patch_sklearn\n\npatch_sklearn()\n\n# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.feature_extraction.text import TfidfVectorizer,CountVectorizer\nfrom sklearn.decomposition import TruncatedSVD\nfrom sklearn.neighbors import NearestNeighbors\nimport Levenshtein\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport gc\nfrom sklearn import preprocessing\nimport lightgbm as lgb\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import KFold\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.metrics import precision_score\nfrom sklearn.metrics import recall_score\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import roc_auc_score\nfrom IPython.display import display\nimport collections\nfrom tqdm import tqdm\nimport string\nimport Levenshtein\nimport difflib\nimport unidecode\nimport pickle\n\ntqdm.pandas()\n\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n%load_ext Cython\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-07T16:02:35.119225Z","iopub.execute_input":"2022-07-07T16:02:35.120959Z","iopub.status.idle":"2022-07-07T16:02:38.216052Z","shell.execute_reply.started":"2022-07-07T16:02:35.120806Z","shell.execute_reply":"2022-07-07T16:02:38.214683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        #else:\n        #    df[col] = df[col].astype('category')\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:38.217671Z","iopub.execute_input":"2022-07-07T16:02:38.218072Z","iopub.status.idle":"2022-07-07T16:02:38.23305Z","shell.execute_reply.started":"2022-07-07T16:02:38.218037Z","shell.execute_reply":"2022-07-07T16:02:38.232103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train=pd.read_csv('../input/foursquare-location-matching/test.csv')\n\nwith open(\"../input/baseline-four-square-test/list_models_baseline_foursquare_v17.pkl\", 'rb') as f:\n    list_models = pickle.load(f)\n    \n\nwith open(\"../input/baseline-four-square-test/count_vectorizer_categories_foursquare_v10.pkl\", 'rb') as f:\n    \n    count_vectorizer = pickle.load(f)\n    \n\nwith open(\"../input/baseline-four-square-test/variable_vectorizer_name_foursquare_v10.pkl\", 'rb') as f:\n    name_vectorizer = pickle.load(f)\n    \nwith open(\"../input/baseline-four-square-test/tsvd_vectorizer_name_foursquare_v10.pkl\", 'rb') as f:\n    name_tsvd_transformer = pickle.load(f)\n    \n    \nwith open(\"../input/baseline-four-square-test/variable_vectorizer_categories_foursquare_v10.pkl\", 'rb') as f:\n    categories_vectorizer = pickle.load(f)\n    \nwith open(\"../input/baseline-four-square-test/tsvd_vectorizer_categories_foursquare_v10.pkl\", 'rb') as f:\n    categories_tsvd_transformer = pickle.load(f)\n    \n    \n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:38.234727Z","iopub.execute_input":"2022-07-07T16:02:38.235418Z","iopub.status.idle":"2022-07-07T16:02:40.672937Z","shell.execute_reply.started":"2022-07-07T16:02:38.235382Z","shell.execute_reply":"2022-07-07T16:02:40.671544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:40.676754Z","iopub.execute_input":"2022-07-07T16:02:40.677554Z","iopub.status.idle":"2022-07-07T16:02:40.701212Z","shell.execute_reply.started":"2022-07-07T16:02:40.677499Z","shell.execute_reply":"2022-07-07T16:02:40.699558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_models","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:40.702717Z","iopub.execute_input":"2022-07-07T16:02:40.703074Z","iopub.status.idle":"2022-07-07T16:02:40.717951Z","shell.execute_reply.started":"2022-07-07T16:02:40.703045Z","shell.execute_reply":"2022-07-07T16:02:40.716314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#count_vectorizer = CountVectorizer(max_features=10,strip_accents=\"unicode\")\n#count_vectorizer = count_vectorizer.fit(df_train[\"categories\"].dropna())","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:40.719365Z","iopub.execute_input":"2022-07-07T16:02:40.720303Z","iopub.status.idle":"2022-07-07T16:02:40.735289Z","shell.execute_reply.started":"2022-07-07T16:02:40.720259Z","shell.execute_reply":"2022-07-07T16:02:40.734007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%cython\nimport numpy as np  # noqa\ncpdef int FastLCS(str S, str T):\n    cdef int i, j\n    cdef int cost\n    cdef int v1,v2,v3,v4\n    cdef int[:, :] dp = np.zeros((len(S) + 1, len(T) + 1), dtype=np.int32)\n    for i in range(len(S)):\n        for j in range(len(T)):\n            cost = (int)(S[i] == T[j])\n            v1 = dp[i, j] + cost\n            v2 = dp[i + 1, j]\n            v3 = dp[i, j + 1]\n            v4 = dp[i + 1, j + 1]\n            dp[i + 1, j + 1] = max((v1,v2,v3,v4))\n    return dp[len(S)][len(T)]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:40.739458Z","iopub.execute_input":"2022-07-07T16:02:40.740378Z","iopub.status.idle":"2022-07-07T16:02:40.759371Z","shell.execute_reply.started":"2022-07-07T16:02:40.740329Z","shell.execute_reply":"2022-07-07T16:02:40.758156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_country(df_data):\n    \n    list_country=  ['US',\n     'TR',\n     'ID',\n     'JP',\n     'TH',\n     'RU',\n     'BR',\n     'MY',\n     'BE',\n     'GB',\n     'PH',\n     'MX',\n     'SG',\n     'KR',\n     'DE',\n     'FR',\n     'ES']\n        \n    df_data[\"country_popular\"]=df_data[\"country\"].copy()\n        \n    df_data.loc[~df_data[\"country\"].isin(list_country),\"country_popular\"]=\"OTHER\"\n        \n        \n    return df_data\n\n\ndef create_n_words(data):\n\n    data.loc[data['name'].notnull(),'n_words_name']=data.loc[data['name'].notnull(),'name'].apply(lambda x: len(str(x).split()))\n\n    data.loc[data['name'].notnull(),'n_characters_name']=data.loc[data['name'].notnull(),'name'].apply(lambda x: len(str(x).replace(\" \",\"\")))\n    \n    data.loc[data['name_match_id'].notnull(),'n_words_name_match_id']=data.loc[data['name_match_id'].notnull(),'name_match_id'].apply(lambda x: len(str(x).split()))\n    \n    data.loc[data['name_match_id'].notnull(),'n_characters_name_match_id']=data.loc[data['name_match_id'].notnull(),'name_match_id'].apply(lambda x: len(str(x).replace(\" \",\"\")))\n    \n    data.loc[data['address'].notnull(),'n_words_address']=data.loc[data['address'].notnull(),'address'].apply(lambda x: len(str(x).split()))\n    \n    data.loc[data['address'].notnull(),'n_characters_address']=data.loc[data['address'].notnull(),'address'].apply(lambda x: len(str(x).replace(\" \",\"\")))\n    \n    data.loc[data['address_match_id'].notnull(),'n_words_address_match_id']=data.loc[data['address_match_id'].notnull(),'address_match_id'].apply(lambda x: len(str(x).split()))\n    \n    data.loc[data['address_match_id'].notnull(),'n_characters_address_match_id']=data.loc[data['address_match_id'].notnull(),'address_match_id'].apply(lambda x: len(str(x).replace(\" \",\"\")))\n    \n    data.loc[data['categories'].notnull(),'n_words_categories']=data.loc[data['categories'].notnull(),'categories'].apply(lambda x: len(str(x).split()))\n    \n    data.loc[data['categories_match_id'].notnull(),'n_words_categories_match_id']=data.loc[data['categories_match_id'].notnull(),'categories_match_id'].apply(lambda x: len(str(x).split()))\n    \n    return data\n\n    \ndef create_equal_indicator_features(data,equal_variables):\n    \n    for name_variable in equal_variables:\n        \n        name_match_variable=f'{name_variable}_match_id'\n        name_equal_indicator=f'equal_{name_variable}'\n        \n        data[name_equal_indicator]=(data[name_variable]==data[name_match_variable]).astype('int8')\n        \n        condition=(data[name_variable]==np.nan)|(data[name_match_variable]==np.nan)\n        \n        data.loc[condition,name_equal_indicator]=np.nan\n        \n        print(name_variable)\n        \n    return data\n\n\n\ndef jaccard_distance(id_str,id_match_str):\n\n    try :\n        score = len((id_str & id_match_str)) / len((id_str | id_match_str))\n        \n    except :\n        \n        score=np.nan\n    \n    return score\n\ndef first_indicator(id_str,id_match_str):\n\n    try :\n        score = int(id_str.split()[0] in id_match_str)\n        \n    except :\n        \n        score=np.nan\n    \n    return score\n\ndef last_indicator(id_str,id_match_str):\n\n    try :\n        score = int(id_str.split()[-1] in id_match_str)\n        \n    except :\n        \n        score=np.nan\n    \n    return score\n\ndef first_match_indicator(id_str,id_match_str):\n\n    try :\n        score = len((id_str & id_match_str)) / len(id_str)\n        \n    except :\n        \n        score=np.nan\n    \n    return score\n\ndef second_match_indicator(id_str,id_match_str):\n\n    try :\n        score = len((id_str & id_match_str)) / len(id_match_str)\n        \n    except :\n        \n        score=np.nan\n    \n    return score\n\n\n\ndef levenshtein_distance(id_str,id_match_str):\n    \n    try :\n    \n        score=Levenshtein.distance(id_str, id_match_str)\n        \n    except :\n        \n        score=np.nan\n    \n    return score\n\ndef jaro_winkler(id_str,id_match_str):\n    \n    try :\n    \n        score=Levenshtein.jaro_winkler(id_str, id_match_str)\n        \n    except :\n        \n        score=np.nan\n    \n    return score\n\ndef sequence_matcher(id_str,id_match_str):\n    \n    try :\n    \n        score=difflib.SequenceMatcher(None, id_str,id_match_str).ratio()\n        \n    except :\n        \n        score=np.nan\n    \n    return score\n\ndef process_string_variable(id_str):\n\n    if id_str is float :\n\n        id_str=str(int(id_str))\n        \n    else :\n        \n        id_str=str(id_str)\n        \n    return id_str\n\n\ndef metrics_similiarity(data,name_variable,name_match_variable=None,distances_list=[\"jaccard\"],suffix='',make_unidecode=0,remove_vowels=0):\n    ## check if there should be separation between words for some punctuations\n    \n    scores_jaccard = []\n    scores_levenshtein=[]\n    scores_jaro_winkler=[]\n    scores_sequence_matcher=[]\n    scores_lcs=[]\n    scores_first_match_indicator=[]\n    scores_second_match_indicator=[]\n    scores_first_indicator=[]\n    scores_last_indicator=[]\n    str_vowels=\"aeiou\"\n    \n    \n    if name_match_variable is None :\n    \n        name_match_variable=f'{name_variable}_match_id'\n    \n    for id_str, id_match_str in tqdm(zip(data[name_variable].fillna(\"nullvalue\").to_numpy(), data[name_match_variable].fillna(\"nullvalue_match\").to_numpy())):\n    \n        if make_unidecode==1 :\n            id_str=unidecode.unidecode(id_str)\n            id_match_str=unidecode.unidecode(id_match_str)\n            \n        id_str=id_str.lower().translate(str.maketrans(string.punctuation,' '*len(string.punctuation)))\n        if remove_vowels==1:\n            id_str=id_str.lower().translate(str.maketrans('', '',str_vowels))\n        id_str=\" \".join(id_str.split()) # this is new\n            \n        id_match_str=id_match_str.lower().translate(str.maketrans(string.punctuation,' '*len(string.punctuation)))\n        if remove_vowels==1:\n            id_match_str=id_match_str.lower().translate(str.maketrans('', '',str_vowels))\n        id_match_str=\" \".join(id_match_str.split())\n\n        \n        id_str_set=set(id_str.split())\n        id_match_str_set=set(id_match_str.split())\n        \n                \n        if \"jaccard\" in distances_list :\n                \n            score_jaccard=jaccard_distance(id_str_set,id_match_str_set)\n                \n            scores_jaccard.append(score_jaccard)\n            \n        if \"first_match_indicator\" in distances_list :\n                \n            score_first_match_indicator=first_match_indicator(id_str_set,id_match_str_set)\n                \n            scores_first_match_indicator.append(score_first_match_indicator)\n            \n        if \"second_match_indicator\" in distances_list :\n                \n            score_second_match_indicator=second_match_indicator(id_str_set,id_match_str_set)\n                \n            scores_second_match_indicator.append(score_second_match_indicator)\n            \n        if \"first_indicator\" in distances_list :\n                \n            score_first_indicator=first_indicator(id_str,id_match_str)\n                \n            scores_first_indicator.append(score_first_indicator)\n            \n        if \"last_indicator\" in distances_list :\n                \n            score_last_indicator=last_indicator(id_str,id_match_str)\n                \n            scores_last_indicator.append(score_last_indicator)\n            \n            \n        if \"levenshtein\" in distances_list:\n                \n            score_levenshtein=levenshtein_distance(id_str,id_match_str)\n                \n            scores_levenshtein.append(score_levenshtein)\n                \n        if \"jaro_winkler\" in distances_list:\n                \n            score_jaro_winkler=jaro_winkler(id_str,id_match_str)\n                \n            scores_jaro_winkler.append(score_jaro_winkler)\n            \n        if \"sequence_matcher\" in distances_list:\n            \n            score_sequence_matcher=sequence_matcher(id_str,id_match_str)\n                \n            scores_sequence_matcher.append(score_sequence_matcher)\n            \n        if \"lcs\" in distances_list:\n            \n            score_lcs=FastLCS(id_str,id_match_str)\n                \n            scores_lcs.append(score_lcs)\n                \n     \n    condition=data[name_variable].isnull() | data[name_match_variable].isnull()\n            \n            \n    if \"jaccard\" in distances_list :\n                            \n        scores_jaccard=np.array(scores_jaccard)\n        \n        data[f'jaccard_distance_match_{name_variable}_{suffix}']=scores_jaccard\n        \n        data.loc[condition,f'jaccard_distance_match_{name_variable}_{suffix}']=np.nan\n        \n    if \"first_match_indicator\" in distances_list :\n        \n        scores_first_match_indicator=np.array(scores_first_match_indicator)\n        \n        data[f'first_match_indicator_match_{name_variable}_{suffix}']=scores_first_match_indicator\n        \n        data.loc[condition,f'first_match_indicator_match_{name_variable}_{suffix}']=np.nan\n        \n    if \"second_match_indicator\" in distances_list :\n        \n        scores_second_match_indicator=np.array(scores_second_match_indicator)\n        \n        data[f'second_match_indicator_match_{name_variable}_{suffix}']=scores_second_match_indicator\n        \n        data.loc[condition,f'second_match_indicator_match_{name_variable}_{suffix}']=np.nan\n        \n        \n    if \"first_indicator\" in distances_list :\n                            \n        scores_first_indicator=np.array(scores_first_indicator)\n        \n        data[f'first_indicator_match_{name_variable}_{suffix}']=scores_first_indicator\n        \n        data.loc[condition,f'first_indicator_match_{name_variable}_{suffix}']=np.nan\n        \n    if \"last_indicator\" in distances_list :\n                            \n        scores_last_indicator=np.array(scores_last_indicator)\n        \n        data[f'last_indicator_match_{name_variable}_{suffix}']=scores_last_indicator\n        \n        data.loc[condition,f'last_indicator_match_{name_variable}_{suffix}']=np.nan\n            \n            \n    if \"levenshtein\" in distances_list:\n                \n        scores_levenshtein=np.array(scores_levenshtein)\n        \n        data[f'levenshtein_distance_match_{name_variable}_{suffix}']=scores_levenshtein\n        \n        data.loc[condition,f'levenshtein_distance_match_{name_variable}_{suffix}']=np.nan\n                \n    if \"jaro_winkler\" in distances_list:\n        \n        scores_jaro_winkler=np.array(scores_jaro_winkler)\n        \n        data[f'jaro_winkler_distance_match_{name_variable}_{suffix}']=scores_jaro_winkler\n        \n        data.loc[condition,f'jaro_winkler_distance_match_{name_variable}_{suffix}']=np.nan\n        \n    if \"sequence_matcher\" in distances_list:\n        \n        scores_sequence_matcher=np.array(scores_sequence_matcher)\n        \n        data[f'sequence_matcher_distance_match_{name_variable}_{suffix}']=scores_sequence_matcher\n        \n        data.loc[condition,f'sequence_matcher_distance_match_{name_variable}_{suffix}']=np.nan\n        \n    if \"lcs\" in distances_list:\n        \n        scores_lcs=np.array(scores_lcs)\n        \n        data[f'lcs_distance_match_{name_variable}_{suffix}']=scores_lcs\n        \n        data.loc[condition,f'lcs_distance_match_{name_variable}_{suffix}']=np.nan\n    \n    \n\n    return data\n\n\ndef get_numbers_from_string(data,name_variable):\n    \n    name_number_variable=f'numbers_{name_variable}'\n    \n    data[name_number_variable]=data[name_variable].fillna('nullvalue').apply(lambda x: (\" \".join([''.join(filter(str.isdigit, string)) for string in x.split()])).strip())\n    \n    data.loc[data[name_number_variable]=='',name_number_variable]=np.nan\n    \n    return data\n\ndef get_numbers_match(data,name_variable):\n\n    data=get_numbers_from_string(data,name_variable)\n\n    data=get_numbers_from_string(data,f'{name_variable}_match_id')\n\n    data=metrics_similiarity(data,f'numbers_{name_variable}')\n    \n    return data\n\n\ndef get_categories_indicator(data,variable_name,transformer,n_features=10):\n    \n    data_result=transformer.transform(data[variable_name].fillna(\"null\"))\n    \n    data_result=data_result.toarray()\n    \n    data_result=pd.DataFrame(data_result)\n    \n    dict_names = {transformer.vocabulary_[k] : k for k in transformer.vocabulary_}\n\n    list_names=[f'{dict_names[x]}_{variable_name}' for x in range(0,n_features) ]\n    \n    data_result.columns=list_names\n    \n    data = pd.concat([data, data_result], axis=1)\n\n    \n    return data\n\ndef get_tsvd_lda_vectors(data,variable_name,vector_transformer,tsvd_transformer):\n    \n    variable_vectors=vector_transformer.transform(data[variable_name].fillna(\"\"))\n    \n    data_result=tsvd_transformer.transform(variable_vectors)\n    \n    n_features=np.shape(data_result)[1]\n    \n    data_result=pd.DataFrame(data_result)\n        \n    list_names=[f'{variable_name}_component_{x}' for x in range(0,n_features) ]\n    \n    data_result.columns=list_names\n    \n    data = pd.concat([data, data_result], axis=1)\n\n    return data\n\n\n\n\ndef nearest_neighbors(data,n_nearest):\n    \n    knn = NearestNeighbors(n_neighbors = n_nearest,metric='haversine')\n    knn.fit(data, data.index)\n    distances,nearest_ids = knn.kneighbors(data,return_distance = True)\n    \n    return distances,nearest_ids\n\n\n# based on this https://www.kaggle.com/competitions/coleridgeinitiative-show-us-the-data/discussion/228814#1253355\ndef jaccard_similarity(row,variable):\n    \n    try : \n        l1 = row[variable].split(\" \")\n        l2 = row[f'{variable}_match_id'].split(\" \")    \n        intersection = len([s for s in l1 if s in l2])\n        union = (len(l1) + len(l2)) - intersection\n        return float(intersection) / union\n    \n    except :\n        return np.nan\n    \n    \ndef jaccard_similarity_like(data,name_variable):\n    \n    name_match_variable=f'{name_variable}_match_id'\n    \n    count_vectorizer = CountVectorizer(binary=True,stop_words=['nullvalue'],token_pattern=r\"(?u)\\b\\w+\\b\",strip_accents=\"unicode\")\n    data_result = count_vectorizer.fit_transform(data[name_variable].fillna('nullvalue'))\n    dict_id_index = dict(zip(data[name_variable].values, data.index))\n    \n    index_match_id=[dict_id_index[element] for element in data[name_match_variable]]\n    \n    distance_vector=data_result.multiply(data_result[index_match_id]).sum(axis=1).A.ravel()\n    \n    return distance_vector\n\n\ndef filter_candidates(data,n_neighbor):\n    \n    #data=data.loc[(data['jaccard_like_distance_name']>0)|(data['jaccard_like_distance_address']>0)].reset_index(drop=True)\n    \n    #data=data.loc[(data['jaccard_like_distance_name']>0)].reset_index(drop=True)\n        \n    #data=data.loc[(data['jaccard_like_distance_name']>0)|(data['jaro_winkler_distance_match_name_']>0.7)].reset_index(drop=True)\n    \n    #data=data.loc[(data['jaccard_distance_match_name_']>0)|(data['jaro_winkler_distance_match_name_']>0.7)].reset_index(drop=True)\n    \n    #data=data.loc[(data['jaccard_distance_match_name_']>0)|(data['jaro_winkler_distance_match_name_']>0.4)|(data['jaccard_distance_match_name_unidecode']>0)].reset_index(drop=True)\n        \n    #data=data.loc[(data['jaccard_distance_match_name_']>0)|(data['jaro_winkler_distance_match_name_']>0.7)|(data['jaccard_distance_match_name_unidecode']>0)|(data['jaro_winkler_distance_match_name_unidecode']>0.7)].reset_index(drop=True)\n    \n    #data=data.loc[(data['jaccard_distance_match_name_']>0)|(data['jaro_winkler_distance_match_name_']>0.7)|(data['jaccard_distance_match_name_unidecode']>0)|(data['jaro_winkler_distance_match_name_unidecode']>0.7)|(data[\"jaccard_distance_match_numbers_address_\"]==1)|(data['jaccard_distance_match_address_']==1)].reset_index(drop=True)\n    \n    #data=data.loc[(data['jaccard_distance_match_name_']>0)|(data['jaro_winkler_distance_match_name_']>0.65)|(data['jaccard_distance_match_name_unidecode']>0)|(data['jaro_winkler_distance_match_name_unidecode']>0.65)].reset_index(drop=True)\n    \n    \n    if n_neighbor > 1 :\n    \n        data=data.loc[(data['jaccard_distance_match_name_']>0)|(data['jaro_winkler_distance_match_name_']>0.7)|(data['jaccard_distance_match_name_unidecode']>0)|(data['jaro_winkler_distance_match_name_unidecode']>0.7)].reset_index(drop=True)\n    \n    return data\n\n\ndef get_match_variables(data,data_variables):\n\n    data=data.merge(data_variables,on='id',how='left',suffixes=(None, '_id'))\n\n    data=data.merge(data_variables,left_on='id_match',right_on='id',how='left',suffixes=(None, '_match_id'))\n    \n    return data\n    \n    \n\ndef create_candidates(data_position,measure_variables,n_nearest,data_features):\n    \n    data_position[\"longitude\"]=np.radians(data_position[\"longitude\"].values)\n    \n    data_position[\"latitude\"]=np.radians(data_position[\"latitude\"].values)\n    \n    \n    distances,nearest_ids=nearest_neighbors(data_position[measure_variables],n_nearest=n_nearest)\n    \n    df_neighbors=[]\n    \n    \n    for n_neighbor in range(1,n_nearest):\n        \n        df_temp=data_position[['id']].copy()\n        \n        df_temp['distance']=distances[:,n_neighbor]\n        \n        index_id_match=nearest_ids[:,n_neighbor]\n        \n        df_temp['id_match']=df_temp[['id']].loc[index_id_match].values\n        \n        df_temp['n_neighbor']=n_neighbor\n        \n        df_temp=get_match_variables(df_temp,data_features[['id','name']])\n        \n        df_temp=metrics_similiarity(df_temp,'name',distances_list=[\"jaccard\",\"jaro_winkler\"])\n        \n        df_temp=metrics_similiarity(df_temp,'name',distances_list=[\"jaccard\",\"jaro_winkler\"],make_unidecode=1,suffix=\"unidecode\")\n        \n        print(len(df_temp))\n\n        df_temp=filter_candidates(df_temp,n_neighbor)\n\n        print(len(df_temp))\n        print(df_temp[\"distance\"].describe())\n\n        df_neighbors.append(df_temp)\n        \n        print(n_neighbor)\n        \n        print('')\n        \n        \n    df_neighbors=pd.concat(df_neighbors,ignore_index=True).reset_index(drop=True)\n    \n    df_neighbors=df_neighbors.drop([\"id_match_id\"],axis='columns')\n    \n    print(len(df_neighbors))\n    \n    df_temp=df_neighbors.copy()\n    \n    df_temp[[\"id\",\"id_match\"]]=df_temp[[\"id_match\",\"id\"]].copy()\n    \n    df_temp[\"n_neighbor\"]=99\n    \n    df_neighbors=pd.concat([df_neighbors,df_temp],ignore_index=True)\n    \n    del df_temp ; gc.collect()\n    \n    df_neighbors=df_neighbors.drop_duplicates(subset=[\"id\",\"id_match\"],keep=\"first\",ignore_index=True)\n    \n    print(len(df_neighbors))\n\n    return df_neighbors\n\n\ndef get_perfect_jaccard_score(data):\n    \n    dict_predicted_matches=data.loc[data['target']==1].groupby('id')['id_match'].apply(list).to_dict()    \n    #df_real_matchs=data.loc[data['target']==1].groupby('id')['id_match'].apply(list).reset_index()\n    \n    scores = []\n    for id_str, matches in tqdm(zip(df_real_matches['id'].to_numpy(), df_real_matches['id_match'].to_numpy())):\n        \n        if id_str in dict_predicted_matches:\n            targets = dict_predicted_matches[id_str]\n            targets.append(id_str)\n            targets=set(targets)\n        else :\n            targets=set([id_str])\n        preds = set(matches)\n        score = len((targets & preds)) / len((targets | preds))\n        scores.append(score)\n    scores = np.array(scores)\n    \n    metric=scores.mean()\n    \n    print(metric)\n    \n    return metric\n\n\ndef get_tsvd_vectors(data,variable,n_components):\n    \n    variable_vectorizer = CountVectorizer(strip_accents=\"unicode\",binary=True)\n    variable_vectorizer = variable_vectorizer.fit(data[variable].fillna(\"\"))\n    variable_vectors=variable_vectorizer.transform(data[variable].fillna(\"\"))\n\n    tsvd_tranformer=TruncatedSVD(n_components=n_components)\n    tsvd_tranformer=tsvd_tranformer.fit(variable_vectors)\n    tsvd_vectors=tsvd_tranformer.transform(variable_vectors)\n    \n    return tsvd_vectors\n\n\ndef nearest_neighbors_arrays(data,n_nearest):\n    \n    knn = NearestNeighbors(n_neighbors = n_nearest)\n    knn.fit(data,[1]*len(data))\n    distances,nearest_ids = knn.kneighbors(data,return_distance = True)\n    \n    return distances,nearest_ids\n\n\n\ndef create_candidates_strings_vectors(data_id,data_position,n_nearest,data_features,variable=\"name\"):\n    \n    distances,nearest_ids=nearest_neighbors_arrays(data_position,n_nearest=n_nearest)\n    \n    df_neighbors=[]\n    \n    \n    for n_neighbor in range(0,n_nearest):\n        \n        df_temp=data_id[['id']].copy()\n        \n        df_temp['distance_vectors']=distances[:,n_neighbor]\n        \n        index_id_match=nearest_ids[:,n_neighbor]\n        \n        df_temp['id_match']=df_temp[['id']].loc[index_id_match].values\n        \n        df_temp['n_neighbor']=n_neighbor\n        \n        print(len(df_temp))\n        \n        df_temp=get_match_variables(df_temp,data_features)\n        \n        df_temp=metrics_similiarity(df_temp,variable,distances_list=[\"jaccard\",\"jaro_winkler\"])\n        \n        df_temp=metrics_similiarity(df_temp,variable,distances_list=[\"jaccard\",\"jaro_winkler\"],make_unidecode=1,suffix=\"unidecode\")\n                \n        if variable == \"name\" :\n        \n            df_temp=filter_candidates_names(df_temp)\n            \n        elif variable==\"address\" :\n            \n            df_temp=filter_candidates_address(df_temp)\n            \n            \n        \n        print(len(df_temp))\n\n        df_neighbors.append(df_temp)\n        \n        print(n_neighbor)\n        \n        print('')\n        \n        \n    df_neighbors=pd.concat(df_neighbors,ignore_index=True).reset_index(drop=True)\n    \n    df_neighbors=df_neighbors.drop([\"id_match_id\"],axis='columns')\n    \n    print((df_neighbors[\"id\"]==df_neighbors[\"id_match\"]).mean())\n    \n    df_neighbors=df_neighbors.loc[df_neighbors[\"id\"]!=df_neighbors[\"id_match\"]].reset_index(drop=True)\n    \n    df_neighbors[\"n_neighbor\"]=999\n    \n    df_neighbors[\"distance\"]=999\n    \n    print(len(df_neighbors))\n\n    return df_neighbors\n\n  \ndef filter_candidates_names(data):\n    \n    data=data.loc[(data['jaccard_distance_match_name_']>0.875)|(data['jaro_winkler_distance_match_name_']>0.9)|(data['jaccard_distance_match_name_unidecode']>0.875)|(data['jaro_winkler_distance_match_name_unidecode']>0.9)].reset_index(drop=True)\n    \n    return data\n\n\ndef filter_candidates_address(data):\n    \n    data=data.loc[(data['jaccard_distance_match_address_']>0.875)|(data['jaro_winkler_distance_match_address_']>0.9)|(data['jaccard_distance_match_address_unidecode']>0.875)|(data['jaro_winkler_distance_match_address_unidecode']>0.9)].reset_index(drop=True)\n    \n    return data","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:40.761312Z","iopub.execute_input":"2022-07-07T16:02:40.76169Z","iopub.status.idle":"2022-07-07T16:02:41.017088Z","shell.execute_reply.started":"2022-07-07T16:02:40.761649Z","shell.execute_reply":"2022-07-07T16:02:41.015697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nn=7 if len(df_train)>100 else 3\n\n\ndf_train_strings=df_train.dropna(subset=[\"address\"])[[\"id\",\"name\",\"address\",\"latitude\",\"longitude\"]].reset_index(drop=True)\n\nname_vectors=get_tsvd_vectors(df_train_strings,\"name\",n)\n\naddress_vectors=get_tsvd_vectors(df_train_strings,\"address\",n)\n\n\nn=15 if len(df_train)>100 else 3\n\nneighbors_strings=create_candidates_strings_vectors(df_train_strings[[\"id\"]],name_vectors,n,df_train_strings)\n\ndel name_vectors ; gc.collect()\n\n\nn=10 if len(df_train)>100 else 3\n\naddress_neighbors_strings=create_candidates_strings_vectors(df_train_strings[[\"id\"]],address_vectors,n,df_train_strings,variable='address')\n\ndel address_vectors\ndel df_train_strings; gc.collect()\n\n\nprint(len(neighbors_strings))\n\nneighbors_strings=neighbors_strings.merge(address_neighbors_strings[[\"id\",\"id_match\"]],on=[\"id\",\"id_match\"])\n\nprint(len(neighbors_strings))\n\ndel address_neighbors_strings ; gc.collect()\n\nneighbors_strings=neighbors_strings[['id','distance','id_match','n_neighbor','name','name_match_id','jaccard_distance_match_name_','jaro_winkler_distance_match_name_','jaccard_distance_match_name_unidecode','jaro_winkler_distance_match_name_unidecode']]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:41.019818Z","iopub.execute_input":"2022-07-07T16:02:41.020471Z","iopub.status.idle":"2022-07-07T16:02:41.792521Z","shell.execute_reply.started":"2022-07-07T16:02:41.020378Z","shell.execute_reply":"2022-07-07T16:02:41.791001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_nearest=61 if len(df_train)>100 else 2\n\n\ndf_neighbors=create_candidates(df_train[['id','latitude','longitude']],measure_variables=[\"latitude\",\"longitude\"],n_nearest=n_nearest,data_features=df_train[[\"id\",\"name\"]])\n\nprint(len(df_neighbors))\n\ndf_neighbors=pd.concat([df_neighbors,neighbors_strings],ignore_index=True)\n\ndf_neighbors=df_neighbors.drop_duplicates([\"id\",\"id_match\"]).reset_index(drop=True)\n\ndel neighbors_strings ; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:41.794771Z","iopub.execute_input":"2022-07-07T16:02:41.795297Z","iopub.status.idle":"2022-07-07T16:02:42.179221Z","shell.execute_reply.started":"2022-07-07T16:02:41.795225Z","shell.execute_reply":"2022-07-07T16:02:42.17834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.loc[df_train[\"zip\"].notnull(),\"zip\"]=df_train.loc[df_train[\"zip\"].notnull(),\"zip\"].apply(process_string_variable)\n\ndf_train.loc[df_train[\"phone\"].notnull(),\"phone\"]=df_train.loc[df_train[\"phone\"].notnull(),\"phone\"].apply(process_string_variable)\n\n\ndf_submission=df_train[[\"id\"]].copy()\n\n\ndf_neighbors=get_match_variables(df_neighbors,df_train[['id',\"address\",'city', 'state', 'zip','country', 'url', 'phone','categories','latitude','longitude']])\n\ndel df_train ; gc.collect()\n\ndf_neighbors=df_neighbors.drop([\"id_match_id\"],axis='columns')\n\nprint(df_neighbors['distance'].describe())\n\ndf_neighbors=create_n_words(df_neighbors)\n\ndf_neighbors=create_equal_indicator_features(df_neighbors,equal_variables=[ 'zip',\n       'country', 'url', 'phone','categories'])\n\ndf_neighbors=metrics_similiarity(df_neighbors,'name',distances_list=[\"lcs\",\"levenshtein\",\"sequence_matcher\",\"first_match_indicator\",\"second_match_indicator\",\"first_indicator\",\"last_indicator\"])\n\n\n#df_neighbors=metrics_similiarity(df_neighbors,'name',distances_list=[\"jaccard\",\"jaro_winkler\"],make_unidecode=1,suffix=\"unidecode\")\n#df_neighbors=metrics_similiarity(df_neighbors,'name',distances_list=[\"jaccard\"],remove_vowels=1,suffix=\"vowels\")\n\ndf_neighbors=metrics_similiarity(df_neighbors,'name',distances_list=[\"jaccard\"],remove_vowels=1,suffix=\"vowels_unicode\",make_unidecode=1)\n\n\n#df_neighbors=metrics_similiarity(df_neighbors,'address',distances_list=[\"jaccard\"],remove_vowels=1,suffix=\"vowels\")\n\n#df_neighbors=metrics_similiarity(df_neighbors,'address',distances_list=[\"jaccard\"],remove_vowels=1,suffix=\"vowels_unicode\",make_unidecode=1)\n\n\n\ndf_neighbors=metrics_similiarity(df_neighbors,'name',distances_list=[\"lcs\"],make_unidecode=1,suffix=\"unidecode\")\n\ndf_neighbors=metrics_similiarity(df_neighbors,'address',distances_list=[\"jaccard\",\"levenshtein\",\"jaro_winkler\",\"sequence_matcher\"])\n\ndf_neighbors=metrics_similiarity(df_neighbors,'address',distances_list=[\"jaccard\",\"jaro_winkler\"],make_unidecode=1,suffix=\"unidecode\")\n\ndf_neighbors=metrics_similiarity(df_neighbors,'categories',distances_list=[\"jaccard\",\"levenshtein\",\"jaro_winkler\",\"sequence_matcher\"])\n\ndf_neighbors=metrics_similiarity(df_neighbors,'city',distances_list=[\"jaccard\"])\n\ndf_neighbors=metrics_similiarity(df_neighbors,'city',distances_list=[\"jaro_winkler\"],make_unidecode=1,suffix=\"unidecode\")\n\ndf_neighbors=metrics_similiarity(df_neighbors,'state',distances_list=[\"jaccard\"])\n\ndf_neighbors=metrics_similiarity(df_neighbors,'state',distances_list=[\"jaro_winkler\"],make_unidecode=1,suffix=\"unidecode\")\n\ndf_neighbors=metrics_similiarity(df_neighbors,'zip',distances_list=[\"jaccard\"])\n\ndf_neighbors=metrics_similiarity(df_neighbors,'zip',distances_list=[\"jaro_winkler\"],make_unidecode=1,suffix=\"unidecode\")\n\n\ndf_neighbors=metrics_similiarity(df_neighbors,'phone',distances_list=[\"jaro_winkler\",\"sequence_matcher\"])\n\ndf_neighbors=metrics_similiarity(df_neighbors,'name','url',distances_list=[\"jaro_winkler\"],suffix='url')\n\ndf_neighbors=metrics_similiarity(df_neighbors,'name','url_match_id',distances_list=[\"jaro_winkler\"],suffix='url_match_id')\n\ndf_neighbors=metrics_similiarity(df_neighbors,'name_match_id','url',distances_list=[\"jaro_winkler\"],suffix='url')\n\ndf_neighbors=metrics_similiarity(df_neighbors,'name_match_id','url_match_id',distances_list=[\"jaro_winkler\"],suffix='url_match_id')\n\ndf_neighbors=metrics_similiarity(df_neighbors,'name','address_match_id',distances_list=[\"jaccard\"],suffix='name_address')\n\ndf_neighbors=metrics_similiarity(df_neighbors,'name_match_id','address',distances_list=[\"jaccard\"],suffix='name_match_id_address')\n\n\n\n\n#df_neighbors=get_numbers_from_string(df_neighbors,'name')\n\n#df_neighbors=get_numbers_from_string(df_neighbors,'name_match_id')\n\n#df_neighbors=metrics_similiarity(df_neighbors,'numbers_name')\n\ndf_neighbors=get_numbers_match(df_neighbors,'name')\n\n\n#df_neighbors=get_numbers_from_string(df_neighbors,'address')\n\n#df_neighbors=get_numbers_from_string(df_neighbors,'address_match_id')\n\n#df_neighbors=metrics_similiarity(df_neighbors,'numbers_address')\n\n#df_neighbors=get_numbers_match(df_neighbors,'address')\n\n\n\n#df_neighbors=process_country(df_neighbors)\n\ndf_neighbors=reduce_mem_usage(df_neighbors)\n\n\ndf_neighbors=get_categories_indicator(df_neighbors,\"categories\",count_vectorizer,10)\n\ndf_neighbors=get_categories_indicator(df_neighbors,\"categories_match_id\",count_vectorizer,10)\n\ndel count_vectorizer ; gc.collect()\n\ndf_neighbors=get_tsvd_lda_vectors(df_neighbors,\"name\",name_vectorizer,name_tsvd_transformer)\n\ndf_neighbors=get_tsvd_lda_vectors(df_neighbors,\"name_match_id\",name_vectorizer,name_tsvd_transformer)\n\ndel name_vectorizer\ndel name_tsvd_transformer ; gc.collect()\n\ndf_neighbors=get_tsvd_lda_vectors(df_neighbors,\"categories\",categories_vectorizer,categories_tsvd_transformer)\n\ndf_neighbors=get_tsvd_lda_vectors(df_neighbors,\"categories_match_id\",categories_vectorizer,categories_tsvd_transformer)\n\ndel categories_vectorizer\ndel categories_tsvd_transformer ; gc.collect()\n\n\n\n#df_neighbors=get_tsvd_lda_vectors(df_neighbors,\"address\",address_vectorizer,address_tsvd_transformer)\n\n#df_neighbors=get_tsvd_lda_vectors(df_neighbors,\"address_match_id\",address_vectorizer,address_tsvd_transformer)\n\n#del address_vectorizer\n#del address_tsvd_transformer ; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:42.180695Z","iopub.execute_input":"2022-07-07T16:02:42.181064Z","iopub.status.idle":"2022-07-07T16:02:43.134307Z","shell.execute_reply.started":"2022-07-07T16:02:42.18103Z","shell.execute_reply":"2022-07-07T16:02:43.133329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"variables_delete=['name', 'address', 'city', 'state', 'zip', 'country', 'url',\n       'phone', 'categories', \n       'name_match_id', \"latitude_match_id\",\"longitude_match_id\",\n       'address_match_id', 'city_match_id', 'state_match_id', 'zip_match_id',\n       'country_match_id', 'url_match_id', 'phone_match_id',\n       'categories_match_id', \n         'numbers_name','numbers_name_match_id']\n    \ndf_neighbors=df_neighbors.drop(variables_delete,axis=\"columns\")","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:43.135686Z","iopub.execute_input":"2022-07-07T16:02:43.135996Z","iopub.status.idle":"2022-07-07T16:02:43.147654Z","shell.execute_reply.started":"2022-07-07T16:02:43.13597Z","shell.execute_reply":"2022-07-07T16:02:43.146531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_neighbors=reduce_mem_usage(df_neighbors)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:43.153506Z","iopub.execute_input":"2022-07-07T16:02:43.15434Z","iopub.status.idle":"2022-07-07T16:02:43.22195Z","shell.execute_reply.started":"2022-07-07T16:02:43.154291Z","shell.execute_reply":"2022-07-07T16:02:43.220697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features=['distance',\n 'n_neighbor',\n 'jaccard_distance_match_name_',\n 'jaro_winkler_distance_match_name_',\n 'jaccard_distance_match_name_unidecode',\n 'jaro_winkler_distance_match_name_unidecode',\n 'latitude',\n 'longitude',\n 'n_words_name',\n 'n_characters_name',\n 'n_words_name_match_id',\n 'n_characters_name_match_id',\n 'n_words_address',\n 'n_characters_address',\n 'n_words_address_match_id',\n 'n_characters_address_match_id',\n 'n_words_categories',\n 'n_words_categories_match_id',\n 'equal_zip',\n 'equal_country',\n 'equal_url',\n 'equal_phone',\n 'equal_categories',\n 'first_match_indicator_match_name_',\n 'second_match_indicator_match_name_',\n 'first_indicator_match_name_',\n 'last_indicator_match_name_',\n 'levenshtein_distance_match_name_',\n 'sequence_matcher_distance_match_name_',\n 'lcs_distance_match_name_',\n 'jaccard_distance_match_name_vowels_unicode',\n 'lcs_distance_match_name_unidecode',\n 'jaccard_distance_match_address_',\n 'levenshtein_distance_match_address_',\n 'jaro_winkler_distance_match_address_',\n 'sequence_matcher_distance_match_address_',\n 'jaccard_distance_match_address_unidecode',\n 'jaro_winkler_distance_match_address_unidecode',\n 'jaccard_distance_match_categories_',\n 'levenshtein_distance_match_categories_',\n 'jaro_winkler_distance_match_categories_',\n 'sequence_matcher_distance_match_categories_',\n 'jaccard_distance_match_city_',\n 'jaro_winkler_distance_match_city_unidecode',\n 'jaccard_distance_match_state_',\n 'jaro_winkler_distance_match_state_unidecode',\n 'jaccard_distance_match_zip_',\n 'jaro_winkler_distance_match_zip_unidecode',\n 'jaro_winkler_distance_match_phone_',\n 'sequence_matcher_distance_match_phone_',\n 'jaro_winkler_distance_match_name_url',\n 'jaro_winkler_distance_match_name_url_match_id',\n 'jaro_winkler_distance_match_name_match_id_url',\n 'jaro_winkler_distance_match_name_match_id_url_match_id',\n 'jaccard_distance_match_name_name_address',\n 'jaccard_distance_match_name_match_id_name_match_id_address',\n 'jaccard_distance_match_numbers_name_',\n 'bars_categories',\n 'buildings_categories',\n 'cafes_categories',\n 'college_categories',\n 'food_categories',\n 'offices_categories',\n 'restaurants_categories',\n 'shops_categories',\n 'stations_categories',\n 'stores_categories',\n 'bars_categories_match_id',\n 'buildings_categories_match_id',\n 'cafes_categories_match_id',\n 'college_categories_match_id',\n 'food_categories_match_id',\n 'offices_categories_match_id',\n 'restaurants_categories_match_id',\n 'shops_categories_match_id',\n 'stations_categories_match_id',\n 'stores_categories_match_id',\n 'name_component_0',\n 'name_component_1',\n 'name_component_2',\n 'name_component_3',\n 'name_component_4',\n 'name_component_5',\n 'name_component_6',\n 'name_match_id_component_0',\n 'name_match_id_component_1',\n 'name_match_id_component_2',\n 'name_match_id_component_3',\n 'name_match_id_component_4',\n 'name_match_id_component_5',\n 'name_match_id_component_6',\n 'categories_component_0',\n 'categories_component_1',\n 'categories_component_2',\n 'categories_component_3',\n 'categories_component_4',\n 'categories_component_5',\n 'categories_component_6',\n 'categories_match_id_component_0',\n 'categories_match_id_component_1',\n 'categories_match_id_component_2',\n 'categories_match_id_component_3',\n 'categories_match_id_component_4',\n 'categories_match_id_component_5',\n 'categories_match_id_component_6']\n\n\n\ncategorical=[]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:43.223372Z","iopub.execute_input":"2022-07-07T16:02:43.22371Z","iopub.status.idle":"2022-07-07T16:02:43.235358Z","shell.execute_reply.started":"2022-07-07T16:02:43.223682Z","shell.execute_reply":"2022-07-07T16:02:43.233648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_neighbors[features]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:43.237216Z","iopub.execute_input":"2022-07-07T16:02:43.238531Z","iopub.status.idle":"2022-07-07T16:02:43.282651Z","shell.execute_reply.started":"2022-07-07T16:02:43.238464Z","shell.execute_reply":"2022-07-07T16:02:43.281508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_neighbors[\"estimator\"]=0\n\nfor modelo in list_models:\n\n    df_neighbors[\"estimator\"]=df_neighbors[\"estimator\"]+modelo.predict(df_neighbors[features].values, num_iteration=modelo.best_iteration)/5","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:43.284681Z","iopub.execute_input":"2022-07-07T16:02:43.285142Z","iopub.status.idle":"2022-07-07T16:02:43.318612Z","shell.execute_reply.started":"2022-07-07T16:02:43.285104Z","shell.execute_reply":"2022-07-07T16:02:43.317552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_neighbors.groupby(\"id\")[\"id_match\"].apply(list)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:43.320364Z","iopub.execute_input":"2022-07-07T16:02:43.320726Z","iopub.status.idle":"2022-07-07T16:02:43.331758Z","shell.execute_reply.started":"2022-07-07T16:02:43.320693Z","shell.execute_reply":"2022-07-07T16:02:43.330513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef adjust_results(df_results):\n\n    df_results_adjusted=df_results.merge(df_results,left_on=[\"id\",\"id_match\"],right_on=[\"id_match\",\"id\"],suffixes=(\"_prob_1\",\"_prob_2\"))\n\n    del df_results ; gc.collect()\n\n    df_results_adjusted=df_results_adjusted.drop(['id_prob_2','id_match_prob_2'],axis='columns')\n\n    df_results_adjusted=df_results_adjusted.rename(columns={\"id_prob_1\":\"id\",\"id_match_prob_1\":\"id_match\"})\n\n    df_results_adjusted[\"estimator\"]=(df_results_adjusted[\"estimator_prob_1\"]+df_results_adjusted[\"estimator_prob_2\"])/2\n    \n    return df_results_adjusted\n\n\ndef get_transitive_matches(df_matches):\n    \n    df_extra_matches=df_matches.merge(df_matches,left_on=\"id_match\",right_on=\"id\",how=\"left\")\n        \n    df_extra_matches=df_extra_matches.drop(['id_match_x','id_y'],axis=\"columns\")\n    \n    df_extra_matches=df_extra_matches.rename(columns={\"id_x\":\"id\",\"id_match_y\":\"id_match\"})\n    \n    df_extra_matches=pd.concat([df_matches,df_extra_matches],ignore_index=True)\n    \n    df_extra_matches=df_extra_matches.drop_duplicates([\"id\",\"id_match\"],ignore_index=True)\n    \n    return df_extra_matches\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:43.333496Z","iopub.execute_input":"2022-07-07T16:02:43.334308Z","iopub.status.idle":"2022-07-07T16:02:43.344732Z","shell.execute_reply.started":"2022-07-07T16:02:43.334259Z","shell.execute_reply":"2022-07-07T16:02:43.343797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_results=df_neighbors[['id','id_match','estimator']]\n\ndf_results_adjusted=adjust_results(df_results)\n\n\ncondition=df_results_adjusted[\"estimator\"]>0.5\n\ndf_matches=df_results_adjusted.loc[condition,[\"id\",\"id_match\"]].reset_index(drop=True)\n\n#condition=df_results[\"estimator\"]>0.5\n\n#df_matches=df_results.loc[condition,[\"id\",\"id_match\"]].reset_index(drop=True)\n\n\ndf_extra_matches=get_transitive_matches(df_matches)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:43.346093Z","iopub.execute_input":"2022-07-07T16:02:43.347258Z","iopub.status.idle":"2022-07-07T16:02:43.518416Z","shell.execute_reply.started":"2022-07-07T16:02:43.347189Z","shell.execute_reply":"2022-07-07T16:02:43.517087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_matches","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:43.519733Z","iopub.execute_input":"2022-07-07T16:02:43.52008Z","iopub.status.idle":"2022-07-07T16:02:43.536376Z","shell.execute_reply.started":"2022-07-07T16:02:43.520039Z","shell.execute_reply":"2022-07-07T16:02:43.535485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_submission(data,condition):\n    \n    #dict_predicted_matches=data.loc[condition].groupby('id')['id_match'].apply(list).to_dict() \n    \n    dict_predicted_matches=data.groupby('id')['id_match'].apply(list).to_dict()\n    \n    \n    predictions = []\n    for id_str in tqdm(df_submission['id'].to_numpy()):\n        \n        if id_str in dict_predicted_matches:\n            \n            list_predictions=list(dict_predicted_matches[id_str])\n            \n            if id_str not in list_predictions :\n\n                list_predictions.append(id_str)\n            \n            list_predictions=' '.join(list_predictions)\n            \n            predictions.append(list_predictions)\n            \n        else :\n            \n            predictions.append(id_str)\n            \n    df_submission[\"matches\"]=predictions\n\n    \n    return df_submission","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:43.537942Z","iopub.execute_input":"2022-07-07T16:02:43.538434Z","iopub.status.idle":"2022-07-07T16:02:43.548742Z","shell.execute_reply.started":"2022-07-07T16:02:43.5384Z","shell.execute_reply":"2022-07-07T16:02:43.547821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#condition=df_results_adjusted[\"estimator\"]>0.5\n\n#df_matches=df_results_adjusted.loc[condition,[\"id\",\"id_match\"]].reset_index(drop=True)\n\n\ndf_submission=get_submission(df_matches,condition)\n\ndf_submission_2=get_submission(df_extra_matches,condition)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:43.549765Z","iopub.execute_input":"2022-07-07T16:02:43.550365Z","iopub.status.idle":"2022-07-07T16:02:43.582703Z","shell.execute_reply.started":"2022-07-07T16:02:43.550329Z","shell.execute_reply":"2022-07-07T16:02:43.581726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:43.583996Z","iopub.execute_input":"2022-07-07T16:02:43.584917Z","iopub.status.idle":"2022-07-07T16:02:43.595931Z","shell.execute_reply.started":"2022-07-07T16:02:43.584865Z","shell.execute_reply":"2022-07-07T16:02:43.594641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df_submission_2","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:43.597523Z","iopub.execute_input":"2022-07-07T16:02:43.598791Z","iopub.status.idle":"2022-07-07T16:02:43.607997Z","shell.execute_reply.started":"2022-07-07T16:02:43.598705Z","shell.execute_reply":"2022-07-07T16:02:43.606657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def post_process(df):\n    id2match = dict(zip(df['id'].values, df['matches'].str.split()))\n\n    for base, match in tqdm(df[['id', 'matches']].values):\n        match = match.split()\n        if len(match) == 1:        \n            continue\n\n        for m in match:\n            if base not in id2match[m]:\n                id2match[m].append(base)\n    df['matches'] = df['id'].map(id2match).map(' '.join)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:43.609722Z","iopub.execute_input":"2022-07-07T16:02:43.610307Z","iopub.status.idle":"2022-07-07T16:02:43.623067Z","shell.execute_reply.started":"2022-07-07T16:02:43.610261Z","shell.execute_reply":"2022-07-07T16:02:43.621915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df_submission=post_process(df_submission)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:43.624939Z","iopub.execute_input":"2022-07-07T16:02:43.62572Z","iopub.status.idle":"2022-07-07T16:02:43.634887Z","shell.execute_reply.started":"2022-07-07T16:02:43.625671Z","shell.execute_reply":"2022-07-07T16:02:43.634054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf_submission_2.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:43.636369Z","iopub.execute_input":"2022-07-07T16:02:43.637098Z","iopub.status.idle":"2022-07-07T16:02:43.652773Z","shell.execute_reply.started":"2022-07-07T16:02:43.637052Z","shell.execute_reply":"2022-07-07T16:02:43.651707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission","metadata":{"execution":{"iopub.status.busy":"2022-07-07T16:02:43.654402Z","iopub.execute_input":"2022-07-07T16:02:43.656019Z","iopub.status.idle":"2022-07-07T16:02:43.671124Z","shell.execute_reply.started":"2022-07-07T16:02:43.655763Z","shell.execute_reply":"2022-07-07T16:02:43.669549Z"},"trusted":true},"execution_count":null,"outputs":[]}]}