{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from sklearnex import patch_sklearn\n\npatch_sklearn()\n\n# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.feature_extraction.text import TfidfVectorizer,CountVectorizer\nfrom sklearn.decomposition import TruncatedSVD,LatentDirichletAllocation\nfrom sklearn.neighbors import NearestNeighbors\nimport Levenshtein\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport gc\nfrom sklearn import preprocessing\nimport lightgbm as lgb\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import KFold\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.metrics import precision_score\nfrom sklearn.metrics import recall_score\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import roc_auc_score\nfrom IPython.display import display\nimport collections\nfrom tqdm import tqdm\nimport string\nimport Levenshtein\nimport difflib\nimport unidecode\nimport pickle\n\ntqdm.pandas()\n\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n%load_ext Cython\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-07T05:52:21.899142Z","iopub.execute_input":"2022-07-07T05:52:21.899721Z","iopub.status.idle":"2022-07-07T05:52:24.194981Z","shell.execute_reply.started":"2022-07-07T05:52:21.899632Z","shell.execute_reply":"2022-07-07T05:52:24.193737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**References:**\n\nhttps://www.kaggle.com/code/ryotayoshinobu/foursquare-lightgbm-baseline\n\nhttps://www.kaggle.com/code/guoyonfan/binary-lgb-baseline-0-834\n","metadata":{}},{"cell_type":"code","source":"df_train=pd.read_csv('../input/foursquare-location-matching/train.csv')\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:52:24.214889Z","iopub.execute_input":"2022-07-07T05:52:24.215378Z","iopub.status.idle":"2022-07-07T05:52:30.816147Z","shell.execute_reply.started":"2022-07-07T05:52:24.215344Z","shell.execute_reply":"2022-07-07T05:52:30.814649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:52:30.94263Z","iopub.execute_input":"2022-07-07T05:52:30.943662Z","iopub.status.idle":"2022-07-07T05:52:30.950912Z","shell.execute_reply.started":"2022-07-07T05:52:30.943624Z","shell.execute_reply":"2022-07-07T05:52:30.949916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        #else:\n        #    df[col] = df[col].astype('category')\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_vectorizer = CountVectorizer(max_features=10,strip_accents=\"unicode\")\ncount_vectorizer = count_vectorizer.fit(df_train[\"categories\"].dropna())\n\n\ndef get_tsvd_transformers(data,variable,n_components):\n    \n    variable_vectorizer = TfidfVectorizer(strip_accents=\"unicode\")\n    variable_vectorizer = variable_vectorizer.fit(data[variable].fillna(\"\"))\n    variable_vectors=variable_vectorizer.transform(data[variable].fillna(\"\"))\n\n    tsvd_transformer=TruncatedSVD(n_components=n_components)\n    tsvd_transformer=tsvd_transformer.fit(variable_vectors)\n    #tsvd_vectors=tsvd_tranformer.transform(variable_vectors)\n    \n    return variable_vectorizer,tsvd_transformer\n\ndef get_lda_transformers(data,variable,n_components):\n    \n    variable_vectorizer = TfidfVectorizer(strip_accents=\"unicode\")\n    variable_vectorizer = variable_vectorizer.fit(data[variable].fillna(\"\"))\n    variable_vectors=variable_vectorizer.transform(data[variable].fillna(\"\"))\n\n    lda_transformer=LatentDirichletAllocation(n_components=n_components)\n    lda_transformer=lda_transformer.fit(variable_vectors)\n    #tsvd_vectors=tsvd_tranformer.transform(variable_vectors)\n    \n    return variable_vectorizer,lda_transformer\n\nname_vectorizer,name_tsvd_transformer=get_tsvd_transformers(df_train,\"name\",7)\n#name_vectorizer,name_lda_transformer=get_lda_transformers(df_train,\"name\",7)\n\ncategories_vectorizer,categories_tsvd_transformer=get_tsvd_transformers(df_train,\"categories\",7)\n#categories_vectorizer,categories_lda_transformer=get_lda_transformers(df_train,\"categories\",7)\n\n#address_vectorizer,address_tsvd_transformer=get_tsvd_transformers(df_train,\"address\",7)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:52:31.18447Z","iopub.execute_input":"2022-07-07T05:52:31.18503Z","iopub.status.idle":"2022-07-07T05:52:31.734805Z","shell.execute_reply.started":"2022-07-07T05:52:31.184997Z","shell.execute_reply":"2022-07-07T05:52:31.733418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('count_vectorizer_categories_foursquare_v10.pkl', 'wb') as f:\n    pickle.dump(count_vectorizer, f)\n    \nwith open('variable_vectorizer_name_foursquare_v10.pkl', 'wb') as f:\n    pickle.dump(name_vectorizer, f)\n    \nwith open('tsvd_vectorizer_name_foursquare_v10.pkl', 'wb') as f:\n    pickle.dump(name_tsvd_transformer, f)\n    \nwith open('variable_vectorizer_categories_foursquare_v10.pkl', 'wb') as f:\n    pickle.dump(categories_vectorizer, f)\n    \nwith open('tsvd_vectorizer_categories_foursquare_v10.pkl', 'wb') as f:\n    pickle.dump(categories_tsvd_transformer, f)\n    \n\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:52:31.736435Z","iopub.execute_input":"2022-07-07T05:52:31.737207Z","iopub.status.idle":"2022-07-07T05:52:31.777878Z","shell.execute_reply.started":"2022-07-07T05:52:31.737157Z","shell.execute_reply":"2022-07-07T05:52:31.775894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%cython\nimport numpy as np  # noqa\ncpdef int FastLCS(str S, str T):\n    cdef int i, j\n    cdef int cost\n    cdef int v1,v2,v3,v4\n    cdef int[:, :] dp = np.zeros((len(S) + 1, len(T) + 1), dtype=np.int32)\n    for i in range(len(S)):\n        for j in range(len(T)):\n            cost = (int)(S[i] == T[j])\n            v1 = dp[i, j] + cost\n            v2 = dp[i + 1, j]\n            v3 = dp[i, j + 1]\n            v4 = dp[i + 1, j + 1]\n            dp[i + 1, j + 1] = max((v1,v2,v3,v4))\n    return dp[len(S)][len(T)]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:52:31.99629Z","iopub.execute_input":"2022-07-07T05:52:31.997309Z","iopub.status.idle":"2022-07-07T05:52:32.010611Z","shell.execute_reply.started":"2022-07-07T05:52:31.997263Z","shell.execute_reply":"2022-07-07T05:52:32.009317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_country(df_data):\n    \n    list_country=  ['US',\n     'TR',\n     'ID',\n     'JP',\n     'TH',\n     'RU',\n     'BR',\n     'MY',\n     'BE',\n     'GB',\n     'PH',\n     'MX',\n     'SG',\n     'KR',\n     'DE',\n     'FR',\n     'ES']\n        \n    df_data[\"country_popular\"]=df_data[\"country\"].copy()\n        \n    df_data.loc[~df_data[\"country\"].isin(list_country),\"country_popular\"]=\"OTHER\"\n        \n        \n    return df_data\n\n\ndef create_n_words(data):\n\n    data.loc[data['name'].notnull(),'n_words_name']=data.loc[data['name'].notnull(),'name'].apply(lambda x: len(str(x).split()))\n\n    data.loc[data['name'].notnull(),'n_characters_name']=data.loc[data['name'].notnull(),'name'].apply(lambda x: len(str(x).replace(\" \",\"\")))\n    \n    data.loc[data['name_match_id'].notnull(),'n_words_name_match_id']=data.loc[data['name_match_id'].notnull(),'name_match_id'].apply(lambda x: len(str(x).split()))\n    \n    data.loc[data['name_match_id'].notnull(),'n_characters_name_match_id']=data.loc[data['name_match_id'].notnull(),'name_match_id'].apply(lambda x: len(str(x).replace(\" \",\"\")))\n    \n    data.loc[data['address'].notnull(),'n_words_address']=data.loc[data['address'].notnull(),'address'].apply(lambda x: len(str(x).split()))\n    \n    data.loc[data['address'].notnull(),'n_characters_address']=data.loc[data['address'].notnull(),'address'].apply(lambda x: len(str(x).replace(\" \",\"\")))\n    \n    data.loc[data['address_match_id'].notnull(),'n_words_address_match_id']=data.loc[data['address_match_id'].notnull(),'address_match_id'].apply(lambda x: len(str(x).split()))\n    \n    data.loc[data['address_match_id'].notnull(),'n_characters_address_match_id']=data.loc[data['address_match_id'].notnull(),'address_match_id'].apply(lambda x: len(str(x).replace(\" \",\"\")))\n    \n    data.loc[data['categories'].notnull(),'n_words_categories']=data.loc[data['categories'].notnull(),'categories'].apply(lambda x: len(str(x).split()))\n    \n    data.loc[data['categories_match_id'].notnull(),'n_words_categories_match_id']=data.loc[data['categories_match_id'].notnull(),'categories_match_id'].apply(lambda x: len(str(x).split()))\n    \n    return data\n\n    \ndef create_equal_indicator_features(data,equal_variables):\n    \n    for name_variable in equal_variables:\n        \n        name_match_variable=f'{name_variable}_match_id'\n        name_equal_indicator=f'equal_{name_variable}'\n        \n        data[name_equal_indicator]=(data[name_variable]==data[name_match_variable]).astype('int8')\n        \n        condition=(data[name_variable]==np.nan)|(data[name_match_variable]==np.nan)\n        \n        data.loc[condition,name_equal_indicator]=np.nan\n        \n        print(name_variable)\n        \n    return data\n\n\n\ndef jaccard_distance(id_str,id_match_str):\n\n    try :\n        score = len((id_str & id_match_str)) / len((id_str | id_match_str))\n        \n    except :\n        \n        score=np.nan\n    \n    return score\n\ndef first_indicator(id_str,id_match_str):\n\n    try :\n        score = int(id_str.split()[0] in id_match_str)\n        \n    except :\n        \n        score=np.nan\n    \n    return score\n\ndef last_indicator(id_str,id_match_str):\n\n    try :\n        score = int(id_str.split()[-1] in id_match_str)\n        \n    except :\n        \n        score=np.nan\n    \n    return score\n\ndef first_match_indicator(id_str,id_match_str):\n\n    try :\n        score = len((id_str & id_match_str)) / len(id_str)\n        \n    except :\n        \n        score=np.nan\n    \n    return score\n\ndef second_match_indicator(id_str,id_match_str):\n\n    try :\n        score = len((id_str & id_match_str)) / len(id_match_str)\n        \n    except :\n        \n        score=np.nan\n    \n    return score\n\n\n\ndef levenshtein_distance(id_str,id_match_str):\n    \n    try :\n    \n        score=Levenshtein.distance(id_str, id_match_str)\n        \n    except :\n        \n        score=np.nan\n    \n    return score\n\ndef jaro_winkler(id_str,id_match_str):\n    \n    try :\n    \n        score=Levenshtein.jaro_winkler(id_str, id_match_str)\n        \n    except :\n        \n        score=np.nan\n    \n    return score\n\ndef sequence_matcher(id_str,id_match_str):\n    \n    try :\n    \n        score=difflib.SequenceMatcher(None, id_str,id_match_str).ratio()\n        \n    except :\n        \n        score=np.nan\n    \n    return score\n\n\ndef metrics_similiarity(data,name_variable,name_match_variable=None,distances_list=[\"jaccard\"],suffix='',make_unidecode=0,remove_vowels=0):\n    ## check if there should be separation between words for some punctuations\n    \n    scores_jaccard = []\n    scores_levenshtein=[]\n    scores_jaro_winkler=[]\n    scores_sequence_matcher=[]\n    scores_lcs=[]\n    scores_first_match_indicator=[]\n    scores_second_match_indicator=[]\n    scores_first_indicator=[]\n    scores_last_indicator=[]\n    str_vowels=\"aeiou\"\n    \n    \n    if name_match_variable is None :\n    \n        name_match_variable=f'{name_variable}_match_id'\n    \n    for id_str, id_match_str in tqdm(zip(data[name_variable].fillna(\"nullvalue\").to_numpy(), data[name_match_variable].fillna(\"nullvalue_match\").to_numpy())):\n    \n        if make_unidecode==1 :\n            id_str=unidecode.unidecode(id_str)\n            id_match_str=unidecode.unidecode(id_match_str)\n            \n        id_str=id_str.lower().translate(str.maketrans(string.punctuation,' '*len(string.punctuation)))\n        if remove_vowels==1:\n            id_str=id_str.lower().translate(str.maketrans('', '',str_vowels))\n        id_str=\" \".join(id_str.split()) # this is new\n            \n        id_match_str=id_match_str.lower().translate(str.maketrans(string.punctuation,' '*len(string.punctuation)))\n        if remove_vowels==1:\n            id_match_str=id_match_str.lower().translate(str.maketrans('', '',str_vowels))\n        id_match_str=\" \".join(id_match_str.split())\n\n        \n        id_str_set=set(id_str.split())\n        id_match_str_set=set(id_match_str.split())\n        \n                \n        if \"jaccard\" in distances_list :\n                \n            score_jaccard=jaccard_distance(id_str_set,id_match_str_set)\n                \n            scores_jaccard.append(score_jaccard)\n            \n        if \"first_match_indicator\" in distances_list :\n                \n            score_first_match_indicator=first_match_indicator(id_str_set,id_match_str_set)\n                \n            scores_first_match_indicator.append(score_first_match_indicator)\n            \n        if \"second_match_indicator\" in distances_list :\n                \n            score_second_match_indicator=second_match_indicator(id_str_set,id_match_str_set)\n                \n            scores_second_match_indicator.append(score_second_match_indicator)\n            \n        if \"first_indicator\" in distances_list :\n                \n            score_first_indicator=first_indicator(id_str,id_match_str)\n                \n            scores_first_indicator.append(score_first_indicator)\n            \n        if \"last_indicator\" in distances_list :\n                \n            score_last_indicator=last_indicator(id_str,id_match_str)\n                \n            scores_last_indicator.append(score_last_indicator)\n            \n            \n        if \"levenshtein\" in distances_list:\n                \n            score_levenshtein=levenshtein_distance(id_str,id_match_str)\n                \n            scores_levenshtein.append(score_levenshtein)\n                \n        if \"jaro_winkler\" in distances_list:\n                \n            score_jaro_winkler=jaro_winkler(id_str,id_match_str)\n                \n            scores_jaro_winkler.append(score_jaro_winkler)\n            \n        if \"sequence_matcher\" in distances_list:\n            \n            score_sequence_matcher=sequence_matcher(id_str,id_match_str)\n                \n            scores_sequence_matcher.append(score_sequence_matcher)\n            \n        if \"lcs\" in distances_list:\n            \n            score_lcs=FastLCS(id_str,id_match_str)\n                \n            scores_lcs.append(score_lcs)\n                \n     \n    condition=data[name_variable].isnull() | data[name_match_variable].isnull()\n            \n            \n    if \"jaccard\" in distances_list :\n                            \n        scores_jaccard=np.array(scores_jaccard)\n        \n        data[f'jaccard_distance_match_{name_variable}_{suffix}']=scores_jaccard\n        \n        data.loc[condition,f'jaccard_distance_match_{name_variable}_{suffix}']=np.nan\n        \n    if \"first_match_indicator\" in distances_list :\n        \n        scores_first_match_indicator=np.array(scores_first_match_indicator)\n        \n        data[f'first_match_indicator_match_{name_variable}_{suffix}']=scores_first_match_indicator\n        \n        data.loc[condition,f'first_match_indicator_match_{name_variable}_{suffix}']=np.nan\n        \n    if \"second_match_indicator\" in distances_list :\n        \n        scores_second_match_indicator=np.array(scores_second_match_indicator)\n        \n        data[f'second_match_indicator_match_{name_variable}_{suffix}']=scores_second_match_indicator\n        \n        data.loc[condition,f'second_match_indicator_match_{name_variable}_{suffix}']=np.nan\n        \n        \n    if \"first_indicator\" in distances_list :\n                            \n        scores_first_indicator=np.array(scores_first_indicator)\n        \n        data[f'first_indicator_match_{name_variable}_{suffix}']=scores_first_indicator\n        \n        data.loc[condition,f'first_indicator_match_{name_variable}_{suffix}']=np.nan\n        \n    if \"last_indicator\" in distances_list :\n                            \n        scores_last_indicator=np.array(scores_last_indicator)\n        \n        data[f'last_indicator_match_{name_variable}_{suffix}']=scores_last_indicator\n        \n        data.loc[condition,f'last_indicator_match_{name_variable}_{suffix}']=np.nan\n            \n            \n    if \"levenshtein\" in distances_list:\n                \n        scores_levenshtein=np.array(scores_levenshtein)\n        \n        data[f'levenshtein_distance_match_{name_variable}_{suffix}']=scores_levenshtein\n        \n        data.loc[condition,f'levenshtein_distance_match_{name_variable}_{suffix}']=np.nan\n                \n    if \"jaro_winkler\" in distances_list:\n        \n        scores_jaro_winkler=np.array(scores_jaro_winkler)\n        \n        data[f'jaro_winkler_distance_match_{name_variable}_{suffix}']=scores_jaro_winkler\n        \n        data.loc[condition,f'jaro_winkler_distance_match_{name_variable}_{suffix}']=np.nan\n        \n    if \"sequence_matcher\" in distances_list:\n        \n        scores_sequence_matcher=np.array(scores_sequence_matcher)\n        \n        data[f'sequence_matcher_distance_match_{name_variable}_{suffix}']=scores_sequence_matcher\n        \n        data.loc[condition,f'sequence_matcher_distance_match_{name_variable}_{suffix}']=np.nan\n        \n    if \"lcs\" in distances_list:\n        \n        scores_lcs=np.array(scores_lcs)\n        \n        data[f'lcs_distance_match_{name_variable}_{suffix}']=scores_lcs\n        \n        data.loc[condition,f'lcs_distance_match_{name_variable}_{suffix}']=np.nan\n    \n    \n\n    return data\n\n\ndef get_numbers_from_string(data,name_variable):\n    \n    name_number_variable=f'numbers_{name_variable}'\n    \n    data[name_number_variable]=data[name_variable].fillna('nullvalue').apply(lambda x: (\" \".join([''.join(filter(str.isdigit, string)) for string in x.split()])).strip())\n    \n    data.loc[data[name_number_variable]=='',name_number_variable]=np.nan\n    \n    return data\n\ndef get_numbers_match(data,name_variable):\n\n    data=get_numbers_from_string(data,name_variable)\n\n    data=get_numbers_from_string(data,f'{name_variable}_match_id')\n\n    data=metrics_similiarity(data,f'numbers_{name_variable}')\n    \n    return data\n\n\ndef get_categories_indicator(data,variable_name,transformer,n_features=10):\n    \n    data_result=transformer.transform(data[variable_name].fillna(\"null\"))\n    \n    data_result=data_result.toarray()\n    \n    data_result=pd.DataFrame(data_result)\n    \n    dict_names = {transformer.vocabulary_[k] : k for k in transformer.vocabulary_}\n\n    list_names=[f'{dict_names[x]}_{variable_name}' for x in range(0,n_features) ]\n    \n    data_result.columns=list_names\n    \n    data = pd.concat([data, data_result], axis=1)\n\n    \n    return data\n\ndef get_tsvd_lda_vectors(data,variable_name,vector_transformer,tsvd_transformer):\n    \n    variable_vectors=vector_transformer.transform(data[variable_name].fillna(\"\"))\n    \n    data_result=tsvd_transformer.transform(variable_vectors)\n    \n    n_features=np.shape(data_result)[1]\n    \n    data_result=pd.DataFrame(data_result)\n        \n    list_names=[f'{variable_name}_component_{x}' for x in range(0,n_features) ]\n    \n    data_result.columns=list_names\n    \n    data = pd.concat([data, data_result], axis=1)\n\n    return data","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:52:32.013186Z","iopub.execute_input":"2022-07-07T05:52:32.013686Z","iopub.status.idle":"2022-07-07T05:52:32.085307Z","shell.execute_reply.started":"2022-07-07T05:52:32.013639Z","shell.execute_reply":"2022-07-07T05:52:32.084159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def nearest_neighbors(data,n_nearest):\n    \n    knn = NearestNeighbors(n_neighbors = n_nearest,metric='haversine')\n    knn.fit(data, data.index)\n    distances,nearest_ids = knn.kneighbors(data,return_distance = True)\n    \n    return distances,nearest_ids\n\n\n# based on this https://www.kaggle.com/competitions/coleridgeinitiative-show-us-the-data/discussion/228814#1253355\ndef jaccard_similarity(row,variable):\n    \n    try : \n        l1 = row[variable].split(\" \")\n        l2 = row[f'{variable}_match_id'].split(\" \")    \n        intersection = len([s for s in l1 if s in l2])\n        union = (len(l1) + len(l2)) - intersection\n        return float(intersection) / union\n    \n    except :\n        return np.nan\n    \n    \ndef jaccard_similarity_like(data,name_variable):\n    \n    name_match_variable=f'{name_variable}_match_id'\n    \n    count_vectorizer = CountVectorizer(binary=True,stop_words=['nullvalue'],token_pattern=r\"(?u)\\b\\w+\\b\",strip_accents=\"unicode\")\n    data_result = count_vectorizer.fit_transform(data[name_variable].fillna('nullvalue'))\n    dict_id_index = dict(zip(data[name_variable].values, data.index))\n    \n    index_match_id=[dict_id_index[element] for element in data[name_match_variable]]\n    \n    distance_vector=data_result.multiply(data_result[index_match_id]).sum(axis=1).A.ravel()\n    \n    return distance_vector\n\n\ndef filter_candidates(data,n_neighbor):\n    \n    if n_neighbor > 1 :\n    \n        data=data.loc[(data['jaccard_distance_match_name_']>0)|(data['jaro_winkler_distance_match_name_']>0.7)|(data['jaccard_distance_match_name_unidecode']>0)|(data['jaro_winkler_distance_match_name_unidecode']>0.7)].reset_index(drop=True)\n    \n    return data\n\n\ndef get_match_variables(data,data_variables):\n\n    data=data.merge(data_variables,on='id',how='left',suffixes=(None, '_id'))\n\n    data=data.merge(data_variables,left_on='id_match',right_on='id',how='left',suffixes=(None, '_match_id'))\n    \n    return data\n    \n    \n\ndef create_candidates(data_position,measure_variables,n_nearest,data_features):\n    \n    data_position[\"longitude\"]=np.radians(data_position[\"longitude\"].values)\n    \n    data_position[\"latitude\"]=np.radians(data_position[\"latitude\"].values)\n    \n    distances,nearest_ids=nearest_neighbors(data_position[measure_variables],n_nearest=n_nearest)\n    \n    df_neighbors=[]\n    \n    \n    for n_neighbor in range(1,n_nearest):\n        \n        df_temp=data_position[['id']].copy()\n        \n        df_temp['distance']=distances[:,n_neighbor]\n        \n        index_id_match=nearest_ids[:,n_neighbor]\n        \n        df_temp['id_match']=df_temp[['id']].loc[index_id_match].values\n        \n        df_temp['n_neighbor']=n_neighbor\n        \n        df_temp=get_match_variables(df_temp,data_features[['id','name']])\n        \n        df_temp=metrics_similiarity(df_temp,'name',distances_list=[\"jaccard\",\"jaro_winkler\"])\n        \n        df_temp=metrics_similiarity(df_temp,'name',distances_list=[\"jaccard\",\"jaro_winkler\"],make_unidecode=1,suffix=\"unidecode\")\n        \n        print(len(df_temp))\n\n        df_temp=filter_candidates(df_temp,n_neighbor)\n\n        print(len(df_temp))\n        print(df_temp[\"distance\"].describe())\n\n        df_neighbors.append(df_temp)\n        \n        print(n_neighbor)\n        \n        print('')\n        \n        \n    df_neighbors=pd.concat(df_neighbors,ignore_index=True).reset_index(drop=True)\n    \n    df_neighbors=df_neighbors.drop([\"id_match_id\"],axis='columns')\n    \n    print(len(df_neighbors))\n    \n    df_temp=df_neighbors.copy()\n    \n    df_temp[[\"id\",\"id_match\"]]=df_temp[[\"id_match\",\"id\"]].copy()\n    \n    df_temp[\"n_neighbor\"]=99\n    \n    df_neighbors=pd.concat([df_neighbors,df_temp],ignore_index=True)\n    \n    del df_temp ; gc.collect()\n    \n    df_neighbors=df_neighbors.drop_duplicates(subset=[\"id\",\"id_match\"],keep=\"first\",ignore_index=True)\n    \n    print(len(df_neighbors))\n\n    return df_neighbors\n\n\ndef get_perfect_jaccard_score(data):\n    \n    dict_predicted_matches=data.loc[data['target']==1].groupby('id')['id_match'].apply(list).to_dict()    \n    \n    scores = []\n    for id_str, matches in tqdm(zip(df_real_matches['id'].to_numpy(), df_real_matches['id_match'].to_numpy())):\n        \n        if id_str in dict_predicted_matches:\n            targets = dict_predicted_matches[id_str]\n            targets.append(id_str)\n            targets=set(targets)\n        else :\n            targets=set([id_str])\n        preds = set(matches)\n        score = len((targets & preds)) / len((targets | preds))\n        scores.append(score)\n    scores = np.array(scores)\n    \n    metric=scores.mean()\n    \n    print(metric)\n    \n    return metric\n\n\ndef get_tsvd_vectors(data,variable,n_components):\n    \n    variable_vectorizer = CountVectorizer(strip_accents=\"unicode\",binary=True)\n    variable_vectorizer = variable_vectorizer.fit(data[variable].fillna(\"\"))\n    variable_vectors=variable_vectorizer.transform(data[variable].fillna(\"\"))\n\n    tsvd_tranformer=TruncatedSVD(n_components=n_components)\n    tsvd_tranformer=tsvd_tranformer.fit(variable_vectors)\n    tsvd_vectors=tsvd_tranformer.transform(variable_vectors)\n    \n    return tsvd_vectors\n\n\ndef nearest_neighbors_arrays(data,n_nearest):\n    \n    knn = NearestNeighbors(n_neighbors = n_nearest)\n    knn.fit(data,[1]*len(data))\n    distances,nearest_ids = knn.kneighbors(data,return_distance = True)\n    \n    return distances,nearest_ids\n\n\n\ndef create_candidates_strings_vectors(data_id,data_position,n_nearest,data_features,variable=\"name\"):\n    \n    distances,nearest_ids=nearest_neighbors_arrays(data_position,n_nearest=n_nearest)\n    \n    df_neighbors=[]\n    \n    \n    for n_neighbor in range(0,n_nearest):\n        \n        df_temp=data_id[['id']].copy()\n        \n        df_temp['distance_vectors']=distances[:,n_neighbor]\n        \n        index_id_match=nearest_ids[:,n_neighbor]\n        \n        df_temp['id_match']=df_temp[['id']].loc[index_id_match].values\n        \n        df_temp['n_neighbor']=n_neighbor\n        \n        print(len(df_temp))\n        \n        df_temp=get_match_variables(df_temp,data_features)\n        \n        df_temp=metrics_similiarity(df_temp,variable,distances_list=[\"jaccard\",\"jaro_winkler\"])\n        \n        df_temp=metrics_similiarity(df_temp,variable,distances_list=[\"jaccard\",\"jaro_winkler\"],make_unidecode=1,suffix=\"unidecode\")\n                \n        if variable == \"name\" :\n        \n            df_temp=filter_candidates_names(df_temp)\n            \n        elif variable==\"address\" :\n            \n            df_temp=filter_candidates_address(df_temp)\n            \n            \n        \n        print(len(df_temp))\n\n        df_neighbors.append(df_temp)\n        \n        print(n_neighbor)\n        \n        print('')\n        \n        \n    df_neighbors=pd.concat(df_neighbors,ignore_index=True).reset_index(drop=True)\n    \n    df_neighbors=df_neighbors.drop([\"id_match_id\"],axis='columns')\n    \n    print((df_neighbors[\"id\"]==df_neighbors[\"id_match\"]).mean())\n    \n    df_neighbors=df_neighbors.loc[df_neighbors[\"id\"]!=df_neighbors[\"id_match\"]].reset_index(drop=True)\n    \n    df_neighbors[\"n_neighbor\"]=999\n    \n    df_neighbors[\"distance\"]=999\n    \n    print(len(df_neighbors))\n\n    return df_neighbors\n\n  \ndef filter_candidates_names(data):\n    \n    data=data.loc[(data['jaccard_distance_match_name_']>0.95)|(data['jaro_winkler_distance_match_name_']>0.975)|(data['jaccard_distance_match_name_unidecode']>0.95)|(data['jaro_winkler_distance_match_name_unidecode']>0.975)].reset_index(drop=True)\n    \n    return data\n\n\ndef filter_candidates_address(data):\n    \n    data=data.loc[(data['jaccard_distance_match_address_']>0.95)|(data['jaro_winkler_distance_match_address_']>0.975)|(data['jaccard_distance_match_address_unidecode']>0.95)|(data['jaro_winkler_distance_match_address_unidecode']>0.975)].reset_index(drop=True)\n    \n    return data","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:52:32.087487Z","iopub.execute_input":"2022-07-07T05:52:32.088241Z","iopub.status.idle":"2022-07-07T05:52:32.137777Z","shell.execute_reply.started":"2022-07-07T05:52:32.088172Z","shell.execute_reply":"2022-07-07T05:52:32.136296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_strings=df_train.dropna(subset=[\"address\"])[[\"id\",\"name\",\"address\",\"latitude\",\"longitude\",\"point_of_interest\"]].reset_index(drop=True)\n\nname_vectors=get_tsvd_vectors(df_train_strings,\"name\",7)\n\naddress_vectors=get_tsvd_vectors(df_train_strings,\"address\",7)\n\n\nneighbors_strings=create_candidates_strings_vectors(df_train_strings[[\"id\"]],name_vectors,15,df_train_strings)\n\ndel name_vectors ; gc.collect()\n\n\naddress_neighbors_strings=create_candidates_strings_vectors(df_train_strings[[\"id\"]],address_vectors,10,df_train_strings,variable='address')\n\ndel address_vectors\ndel df_train_strings; gc.collect()\n\n\nprint(len(neighbors_strings))\n\nneighbors_strings=neighbors_strings.merge(address_neighbors_strings[[\"id\",\"id_match\"]],on=[\"id\",\"id_match\"])\n\nprint(len(neighbors_strings))\n\ndel address_neighbors_strings ; gc.collect()\n\nneighbors_strings=neighbors_strings[['id','distance','id_match','n_neighbor','name','name_match_id','jaccard_distance_match_name_','jaro_winkler_distance_match_name_','jaccard_distance_match_name_unidecode','jaro_winkler_distance_match_name_unidecode']]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:52:32.147679Z","iopub.execute_input":"2022-07-07T05:52:32.148406Z","iopub.status.idle":"2022-07-07T05:52:37.709909Z","shell.execute_reply.started":"2022-07-07T05:52:32.14835Z","shell.execute_reply":"2022-07-07T05:52:37.708727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_neighbors=create_candidates(df_train[['id','latitude','longitude']],measure_variables=[\"latitude\",\"longitude\"],n_nearest=61,data_features=df_train[[\"id\",\"name\"]])\n\nprint(len(df_neighbors))\n\ndf_neighbors=pd.concat([df_neighbors,neighbors_strings],ignore_index=True)\n\ndf_neighbors=df_neighbors.drop_duplicates([\"id\",\"id_match\"]).reset_index(drop=True)\n\ndel neighbors_strings ; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:52:37.711536Z","iopub.execute_input":"2022-07-07T05:52:37.711901Z","iopub.status.idle":"2022-07-07T05:52:53.993293Z","shell.execute_reply.started":"2022-07-07T05:52:37.711869Z","shell.execute_reply":"2022-07-07T05:52:53.992047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df_neighbors)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:52:53.995382Z","iopub.execute_input":"2022-07-07T05:52:53.99584Z","iopub.status.idle":"2022-07-07T05:52:54.003495Z","shell.execute_reply.started":"2022-07-07T05:52:53.995794Z","shell.execute_reply":"2022-07-07T05:52:54.002294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_neighbors.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:52:54.005659Z","iopub.execute_input":"2022-07-07T05:52:54.006251Z","iopub.status.idle":"2022-07-07T05:52:54.018136Z","shell.execute_reply.started":"2022-07-07T05:52:54.006186Z","shell.execute_reply":"2022-07-07T05:52:54.016819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_neighbors.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:52:54.03442Z","iopub.execute_input":"2022-07-07T05:52:54.034759Z","iopub.status.idle":"2022-07-07T05:52:54.063142Z","shell.execute_reply.started":"2022-07-07T05:52:54.03473Z","shell.execute_reply":"2022-07-07T05:52:54.062289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_real_matches=df_train[['id','point_of_interest']].copy()\n\ndf_points_matches=df_train.groupby('point_of_interest')['id'].apply(list).reset_index()\n\ndf_points_matches=df_points_matches.rename(columns={'id':'id_match'})\n\ndf_real_matches=df_real_matches.merge(df_points_matches,on='point_of_interest',how='left')\n\ndf_real_matches=df_real_matches.drop('point_of_interest',axis='columns')\n\ndel df_points_matches ; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:52:54.071375Z","iopub.execute_input":"2022-07-07T05:52:54.072333Z","iopub.status.idle":"2022-07-07T05:52:54.357805Z","shell.execute_reply.started":"2022-07-07T05:52:54.072258Z","shell.execute_reply":"2022-07-07T05:52:54.355822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf_neighbors=get_match_variables(df_neighbors,df_train[['id',\"point_of_interest\",\"address\",'city', 'state', 'zip','country', 'url', 'phone','categories','latitude','longitude']])\n\ndel df_train ; gc.collect()\n\ndf_neighbors=df_neighbors.drop([\"id_match_id\"],axis='columns')\n\ndf_neighbors['target']=(df_neighbors['point_of_interest']==df_neighbors['point_of_interest_match_id']).astype('int8')\n\nprint(df_neighbors['target'].mean())\n\nprint(df_neighbors['distance'].describe())\n\nprint(\"perfect jaccard score : \")\n\nprint(get_perfect_jaccard_score(data=df_neighbors[['id','id_match','target']]))\n\ndf_neighbors=create_n_words(df_neighbors)\n\ndf_neighbors=create_equal_indicator_features(df_neighbors,equal_variables=[ 'zip',\n       'country', 'url', 'phone','categories'])\n\ndf_neighbors=metrics_similiarity(df_neighbors,'name',distances_list=[\"lcs\",\"levenshtein\",\"sequence_matcher\",\"first_match_indicator\",\"second_match_indicator\",\"first_indicator\",\"last_indicator\"])\n\ndf_neighbors=metrics_similiarity(df_neighbors,'name',distances_list=[\"jaccard\"],remove_vowels=1,suffix=\"vowels_unicode\",make_unidecode=1)\n\n\ndf_neighbors=metrics_similiarity(df_neighbors,'name',distances_list=[\"lcs\"],make_unidecode=1,suffix=\"unidecode\")\n\ndf_neighbors=metrics_similiarity(df_neighbors,'address',distances_list=[\"jaccard\",\"levenshtein\",\"jaro_winkler\",\"sequence_matcher\"])\n\ndf_neighbors=metrics_similiarity(df_neighbors,'address',distances_list=[\"jaccard\",\"jaro_winkler\"],make_unidecode=1,suffix=\"unidecode\")\n\ndf_neighbors=metrics_similiarity(df_neighbors,'categories',distances_list=[\"jaccard\",\"levenshtein\",\"jaro_winkler\",\"sequence_matcher\"])\n\ndf_neighbors=metrics_similiarity(df_neighbors,'city',distances_list=[\"jaccard\"])\n\ndf_neighbors=metrics_similiarity(df_neighbors,'city',distances_list=[\"jaro_winkler\"],make_unidecode=1,suffix=\"unidecode\")\n\ndf_neighbors=metrics_similiarity(df_neighbors,'state',distances_list=[\"jaccard\"])\n\ndf_neighbors=metrics_similiarity(df_neighbors,'state',distances_list=[\"jaro_winkler\"],make_unidecode=1,suffix=\"unidecode\")\n\ndf_neighbors=metrics_similiarity(df_neighbors,'zip',distances_list=[\"jaccard\"])\n\ndf_neighbors=metrics_similiarity(df_neighbors,'zip',distances_list=[\"jaro_winkler\"],make_unidecode=1,suffix=\"unidecode\")\n\n\ndf_neighbors=metrics_similiarity(df_neighbors,'phone',distances_list=[\"jaro_winkler\",\"sequence_matcher\"])\n\ndf_neighbors=metrics_similiarity(df_neighbors,'name','url',distances_list=[\"jaro_winkler\"],suffix='url')\n\ndf_neighbors=metrics_similiarity(df_neighbors,'name','url_match_id',distances_list=[\"jaro_winkler\"],suffix='url_match_id')\n\ndf_neighbors=metrics_similiarity(df_neighbors,'name_match_id','url',distances_list=[\"jaro_winkler\"],suffix='url')\n\ndf_neighbors=metrics_similiarity(df_neighbors,'name_match_id','url_match_id',distances_list=[\"jaro_winkler\"],suffix='url_match_id')\n\ndf_neighbors=metrics_similiarity(df_neighbors,'name','address_match_id',distances_list=[\"jaccard\"],suffix='name_address')\n\ndf_neighbors=metrics_similiarity(df_neighbors,'name_match_id','address',distances_list=[\"jaccard\"],suffix='name_match_id_address')\n\n\ndf_neighbors=get_numbers_match(df_neighbors,'name')\n\n\ndf_neighbors=reduce_mem_usage(df_neighbors)\n\n\ndf_neighbors=get_categories_indicator(df_neighbors,\"categories\",count_vectorizer,10)\n\ndf_neighbors=get_categories_indicator(df_neighbors,\"categories_match_id\",count_vectorizer,10)\n\ndel count_vectorizer ; gc.collect()\n\ndf_neighbors=get_tsvd_lda_vectors(df_neighbors,\"name\",name_vectorizer,name_tsvd_transformer)\n\ndf_neighbors=get_tsvd_lda_vectors(df_neighbors,\"name_match_id\",name_vectorizer,name_tsvd_transformer)\n\ndel name_vectorizer\ndel name_tsvd_transformer ; gc.collect()\n\ndf_neighbors=get_tsvd_lda_vectors(df_neighbors,\"categories\",categories_vectorizer,categories_tsvd_transformer)\n\ndf_neighbors=get_tsvd_lda_vectors(df_neighbors,\"categories_match_id\",categories_vectorizer,categories_tsvd_transformer)\n\ndel categories_vectorizer\ndel categories_tsvd_transformer ; gc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:52:54.360992Z","iopub.execute_input":"2022-07-07T05:52:54.361714Z","iopub.status.idle":"2022-07-07T05:53:06.682574Z","shell.execute_reply.started":"2022-07-07T05:52:54.361658Z","shell.execute_reply":"2022-07-07T05:53:06.681381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def entrena_lgb(data,test,features,categorical,target):\n\n    kfold=GroupKFold(n_splits=5)\n\n\n    i=1\n\n    r=[]\n    \n    pred_test=np.zeros(len(test))\n\n    importancias=pd.DataFrame()\n\n    importancias['variable']=features\n    \n    list_models=[]\n    \n    \n    cat_ind=[features.index(x) for x in categorical if x in features]\n    \n    dict_cat={}\n    \n    categorical_numerical = data[categorical].dropna().select_dtypes(include=np.number).columns.tolist()\n    \n    categorical_transform=[x for x in categorical if x not in categorical_numerical]\n    \n    for l in categorical_transform:\n        le = preprocessing.LabelEncoder()\n        le.fit(list(data[l].dropna())+list(test[l].dropna()))\n\n        dict_cat[l]=le\n\n        data.loc[~data[l].isnull(),l]=le.transform(data.loc[~data[l].isnull(),l])\n        test.loc[~test[l].isnull(),l]=le.transform(test.loc[~test[l].isnull(),l])\n        \n        \n\n    for train_index,test_index in kfold.split(data,data[target],data['point_of_interest']):\n\n        lgb_data_train = lgb.Dataset(data.loc[train_index,features].values,data.loc[train_index,target].values)\n        lgb_data_eval = lgb.Dataset(data.loc[test_index,features].values,data.loc[test_index,target].values, reference=lgb_data_train)\n\n        params = {\n            'task': 'train',\n            'boosting_type': 'gbdt',\n            'objective': 'binary',\n            'metric': { 'auc'},\n            \"max_depth\":-1,\n            \"num_leaves\":256,\n            'learning_rate': 0.1,\n        \"min_child_samples\": 100,\n            'feature_fraction': 0.9,\n         \"bagging_freq\":1,\n            'bagging_fraction': 0.9,\n            \"lambda_l1\":10,\n            \"lambda_l2\":10,\n           # \"scale_pos_weight\":30,\n           # 'min_data_per_group':500,\n\n            'verbose': 1    \n        }\n\n\n\n\n        modelo = lgb.train(params,lgb_data_train,num_boost_round=13100,valid_sets=lgb_data_eval,early_stopping_rounds=50,verbose_eval=25,categorical_feature=cat_ind)\n\n        importancias['gain_'+str(i)]=modelo.feature_importance(importance_type=\"gain\")\n\n\n        data.loc[test_index,'estimator']=modelo.predict(data.loc[test_index,features].values, num_iteration=modelo.best_iteration)\n        \n        pred_test=pred_test+modelo.predict(test[features].values, num_iteration=modelo.best_iteration)\n        \n        list_models.append(modelo)\n\n        print (\"Fold_\"+str(i))\n        a= (roc_auc_score(data.loc[test_index,target],data.loc[test_index,'estimator']))\n        r.append(a)\n        print (a)\n        print (\"\")\n\n        i=i+1\n        \n    for l in categorical_transform:\n\n            data.loc[~data[l].isnull(),l]=dict_cat[l].inverse_transform(data.loc[~data[l].isnull(),l].astype(int))\n            \n            test.loc[~test[l].isnull(),l]=dict_cat[l].inverse_transform(test.loc[~test[l].isnull(),l].astype(int))\n            \n    var=[x for x in importancias.columns if 'gain_' in x]\n    importancias[\"gain_avg\"]=importancias[var].mean(axis=1)\n    importancias=importancias.sort_values(\"gain_avg\",ascending=False).reset_index(drop=True)\n    \n    pred_test=(pred_test/5)\n    \n    \n    oof=(roc_auc_score(data[target],data['estimator']))\n    \n    print (oof)\n    print (\"mean: \"+str(np.mean(np.array(r))))\n    print (\"std: \"+str(np.std(np.array(r))))\n    \n    dict_resultados={}\n    \n    dict_resultados['importancias']=importancias\n    \n    dict_resultados['predicciones']=pred_test\n    \n    dict_resultados['modelos']=list_models\n    \n    \n    \n    return dict_resultados","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:06.6843Z","iopub.execute_input":"2022-07-07T05:53:06.684612Z","iopub.status.idle":"2022-07-07T05:53:06.710583Z","shell.execute_reply.started":"2022-07-07T05:53:06.684584Z","shell.execute_reply":"2022-07-07T05:53:06.709672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"variables_delete=['name', 'address', 'city', 'state', 'zip', 'country', 'url',\n       'phone', 'categories', \n       'name_match_id', \"latitude_match_id\",\"longitude_match_id\",\n       'address_match_id', 'city_match_id', 'state_match_id', 'zip_match_id',\n       'country_match_id', 'url_match_id', 'phone_match_id',\n       'categories_match_id', 'point_of_interest_match_id',\n         'numbers_name','numbers_name_match_id']\n    \ndf_neighbors=df_neighbors.drop(variables_delete,axis=\"columns\")","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:06.711549Z","iopub.execute_input":"2022-07-07T05:53:06.711895Z","iopub.status.idle":"2022-07-07T05:53:06.748005Z","shell.execute_reply.started":"2022-07-07T05:53:06.711865Z","shell.execute_reply":"2022-07-07T05:53:06.74673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_neighbors=reduce_mem_usage(df_neighbors)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:06.74957Z","iopub.execute_input":"2022-07-07T05:53:06.749937Z","iopub.status.idle":"2022-07-07T05:53:06.904569Z","shell.execute_reply.started":"2022-07-07T05:53:06.749906Z","shell.execute_reply":"2022-07-07T05:53:06.903316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"no_usar=['id',  'id_match', 'name', 'address', 'city', 'state', 'zip', 'country', 'url',\n       'phone', 'categories', 'point_of_interest', 'id_match_id',\n       'name_match_id', \"latitude_match_id\",\"longitude_match_id\",\n       'address_match_id', 'city_match_id', 'state_match_id', 'zip_match_id',\n       'country_match_id', 'url_match_id', 'phone_match_id',\n       'categories_match_id', 'point_of_interest_match_id',\n         'numbers_name','numbers_name_match_id','numbers_address','numbers_address_match_id',\"estimator\",\"country_popular\",\n         'target']\n\nfeatures=[x for x in df_neighbors.columns if x not in no_usar]\n\ncategorical=[]\n\ntarget='target'","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:06.906358Z","iopub.execute_input":"2022-07-07T05:53:06.906718Z","iopub.status.idle":"2022-07-07T05:53:06.914775Z","shell.execute_reply.started":"2022-07-07T05:53:06.906685Z","shell.execute_reply":"2022-07-07T05:53:06.913669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:06.916589Z","iopub.execute_input":"2022-07-07T05:53:06.917Z","iopub.status.idle":"2022-07-07T05:53:06.936137Z","shell.execute_reply.started":"2022-07-07T05:53:06.916967Z","shell.execute_reply":"2022-07-07T05:53:06.934809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_neighbors['target'].mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:06.937514Z","iopub.execute_input":"2022-07-07T05:53:06.937846Z","iopub.status.idle":"2022-07-07T05:53:06.948631Z","shell.execute_reply.started":"2022-07-07T05:53:06.937815Z","shell.execute_reply":"2022-07-07T05:53:06.947785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(features)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:06.949647Z","iopub.execute_input":"2022-07-07T05:53:06.95001Z","iopub.status.idle":"2022-07-07T05:53:06.964064Z","shell.execute_reply.started":"2022-07-07T05:53:06.949978Z","shell.execute_reply":"2022-07-07T05:53:06.963019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test=df_neighbors.head(200)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:06.965547Z","iopub.execute_input":"2022-07-07T05:53:06.9659Z","iopub.status.idle":"2022-07-07T05:53:06.973407Z","shell.execute_reply.started":"2022-07-07T05:53:06.965869Z","shell.execute_reply":"2022-07-07T05:53:06.972569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dict_results=entrena_lgb(data=df_neighbors,test=df_test,features=features,categorical=categorical,target=target)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:06.974604Z","iopub.execute_input":"2022-07-07T05:53:06.975026Z","iopub.status.idle":"2022-07-07T05:53:09.650121Z","shell.execute_reply.started":"2022-07-07T05:53:06.974987Z","shell.execute_reply":"2022-07-07T05:53:09.648794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dict_results['importancias']\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:09.652076Z","iopub.execute_input":"2022-07-07T05:53:09.652433Z","iopub.status.idle":"2022-07-07T05:53:09.671412Z","shell.execute_reply.started":"2022-07-07T05:53:09.6524Z","shell.execute_reply":"2022-07-07T05:53:09.670006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dict_results['importancias'][\"variable\"].tolist()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:09.672738Z","iopub.execute_input":"2022-07-07T05:53:09.673333Z","iopub.status.idle":"2022-07-07T05:53:09.691313Z","shell.execute_reply.started":"2022-07-07T05:53:09.673275Z","shell.execute_reply":"2022-07-07T05:53:09.689972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(df_neighbors['estimator']>0.1).mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:09.692745Z","iopub.execute_input":"2022-07-07T05:53:09.693922Z","iopub.status.idle":"2022-07-07T05:53:09.707647Z","shell.execute_reply.started":"2022-07-07T05:53:09.693841Z","shell.execute_reply":"2022-07-07T05:53:09.706265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def adjust_results(df_results):\n\n    df_results_adjusted=df_results.merge(df_results,left_on=[\"id\",\"id_match\"],right_on=[\"id_match\",\"id\"],suffixes=(\"_prob_1\",\"_prob_2\"))\n\n    del df_results ; gc.collect()\n\n    df_results_adjusted=df_results_adjusted.drop(['id_prob_2','id_match_prob_2'],axis='columns')\n\n    df_results_adjusted=df_results_adjusted.rename(columns={\"id_prob_1\":\"id\",\"id_match_prob_1\":\"id_match\"})\n\n    df_results_adjusted[\"estimator\"]=(df_results_adjusted[\"estimator_prob_1\"]+df_results_adjusted[\"estimator_prob_2\"])/2\n    \n    return df_results_adjusted\n\n\ndef get_transitive_matches(df_matches):\n    \n    df_extra_matches=df_matches.merge(df_matches,left_on=\"id_match\",right_on=\"id\",how=\"left\")\n        \n    df_extra_matches=df_extra_matches.drop(['id_match_x','id_y'],axis=\"columns\")\n    \n    df_extra_matches=df_extra_matches.rename(columns={\"id_x\":\"id\",\"id_match_y\":\"id_match\"})\n    \n    df_extra_matches=pd.concat([df_matches,df_extra_matches],ignore_index=True)\n    \n    df_extra_matches=df_extra_matches.drop_duplicates([\"id\",\"id_match\"],ignore_index=True)\n    \n    return df_extra_matches\n\n\n\ndef get_jaccard_score(data):\n    \n    #dict_predicted_matches=data.loc[condition].groupby('id')['id_match'].apply(list).to_dict()\n    dict_predicted_matches=data.groupby('id')['id_match'].apply(list).to_dict()\n    #df_real_matchs=data.loc[data['target']==1].groupby('id')['id_match'].apply(list).reset_index()\n    \n    scores = []\n    for id_str, matches in tqdm(zip(df_real_matches['id'].to_numpy(), df_real_matches['id_match'].to_numpy())):\n        \n        if id_str in dict_predicted_matches:\n            targets = dict_predicted_matches[id_str]\n            if id_str not in targets :\n                targets.append(id_str)\n            targets=set(targets)\n        else :\n            targets=set([id_str])\n        preds = set(matches)\n        score = len((targets & preds)) / len((targets | preds))\n        scores.append(score)\n    scores = np.array(scores)\n    \n    metric=scores.mean()\n    \n    print(metric)\n    \n    return metric","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:09.709555Z","iopub.execute_input":"2022-07-07T05:53:09.71033Z","iopub.status.idle":"2022-07-07T05:53:09.726961Z","shell.execute_reply.started":"2022-07-07T05:53:09.710279Z","shell.execute_reply":"2022-07-07T05:53:09.72584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_results=df_neighbors[['id','id_match','estimator']].copy()\n\ndf_results_adjusted=adjust_results(df_results)\n\n\ncondition=df_results_adjusted[\"estimator\"]>0.5\n\ndf_matches=df_results_adjusted.loc[condition,[\"id\",\"id_match\"]].reset_index(drop=True)\n\n\ndf_extra_matches=get_transitive_matches(df_matches)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:09.728541Z","iopub.execute_input":"2022-07-07T05:53:09.729196Z","iopub.status.idle":"2022-07-07T05:53:09.916673Z","shell.execute_reply.started":"2022-07-07T05:53:09.729148Z","shell.execute_reply":"2022-07-07T05:53:09.91525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    \nget_jaccard_score(data=df_matches)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:09.918397Z","iopub.execute_input":"2022-07-07T05:53:09.919137Z","iopub.status.idle":"2022-07-07T05:53:09.945411Z","shell.execute_reply.started":"2022-07-07T05:53:09.919096Z","shell.execute_reply":"2022-07-07T05:53:09.94421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_jaccard_score(data=df_extra_matches)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:09.94672Z","iopub.execute_input":"2022-07-07T05:53:09.947061Z","iopub.status.idle":"2022-07-07T05:53:09.973887Z","shell.execute_reply.started":"2022-07-07T05:53:09.947029Z","shell.execute_reply":"2022-07-07T05:53:09.972982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_models=dict_results[\"modelos\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:09.975564Z","iopub.execute_input":"2022-07-07T05:53:09.976362Z","iopub.status.idle":"2022-07-07T05:53:09.982537Z","shell.execute_reply.started":"2022-07-07T05:53:09.976313Z","shell.execute_reply":"2022-07-07T05:53:09.981587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('list_models_baseline_foursquare_v17.pkl', 'wb') as f:\n    pickle.dump(list_models, f)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:09.984071Z","iopub.execute_input":"2022-07-07T05:53:09.984719Z","iopub.status.idle":"2022-07-07T05:53:10.000684Z","shell.execute_reply.started":"2022-07-07T05:53:09.984675Z","shell.execute_reply":"2022-07-07T05:53:09.999558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_results.to_pickle(\"df_probas_train_foursquare.pkl\")","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:10.002337Z","iopub.execute_input":"2022-07-07T05:53:10.00755Z","iopub.status.idle":"2022-07-07T05:53:10.019401Z","shell.execute_reply.started":"2022-07-07T05:53:10.007497Z","shell.execute_reply":"2022-07-07T05:53:10.018296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_results","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:53:10.020996Z","iopub.execute_input":"2022-07-07T05:53:10.022179Z","iopub.status.idle":"2022-07-07T05:53:10.03989Z","shell.execute_reply.started":"2022-07-07T05:53:10.02213Z","shell.execute_reply":"2022-07-07T05:53:10.038718Z"},"trusted":true},"execution_count":null,"outputs":[]}]}