{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1><center> Foursquare Location Matching </center></h1>\n<h3><center> Pairs Data Generation </center></h3>\n<h3><center> Tao Shan </center></h3>\n\nThis notebook generates data pairs by nearest locations and finds all true data pairs. The problem with generating pairs was RAM is not large enough in Kaggle when running algorithms. So you can choose the sample size base on the Computer Configuration.\n\n### Other Relevant notebooks and links\n\nCompetition: [Foursquare - Location Matching](https://www.kaggle.com/competitions/foursquare-location-matching)\n\nTrain data generation notebook: [Foursquare - train data generation](https://www.kaggle.com/taos2000/foursquare-train-data-generation)\n\nTrain data preprocessing notebook: [Foursquare - train data preprocess](https://www.kaggle.com/taos2000/foursquare-train-data-preprocess)\n\nModel Selection: [Foursquare - model selection](https://www.kaggle.com/taos2000/foursquare-model-selection)\n\nModel Training: [Foursquare - model training](https://www.kaggle.com/taos2000/foursquare-model-training)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-06T16:30:44.530832Z","iopub.execute_input":"2022-07-06T16:30:44.531481Z","iopub.status.idle":"2022-07-06T16:30:46.304334Z","shell.execute_reply.started":"2022-07-06T16:30:44.531345Z","shell.execute_reply":"2022-07-06T16:30:46.302882Z"}}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.neighbors import NearestNeighbors\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n\nfrom geopy.geocoders import Nominatim\n\nimport re\nimport string\nfrom nltk.tokenize import word_tokenize\nfrom nltk.corpus import stopwords, wordnet as wn\nfrom nltk.stem import PorterStemmer, WordNetLemmatizer\n\nimport Levenshtein as lev\nimport math\nfrom collections import Counter\n\nfrom pickle import dump, load\nimport time\nfrom sklearn.neighbors import BallTree\n\n\nimport itertools\nfrom tqdm.auto import tqdm\ntqdm.pandas()\nimport gc\n\nfrom fuzzywuzzy import fuzz\nfrom xgboost import XGBClassifier\nfrom sklearn.preprocessing import MinMaxScaler\nstart_time = time.time()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T19:24:57.433052Z","iopub.execute_input":"2022-07-25T19:24:57.433826Z","iopub.status.idle":"2022-07-25T19:24:59.526664Z","shell.execute_reply.started":"2022-07-25T19:24:57.433722Z","shell.execute_reply":"2022-07-25T19:24:59.525349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/foursquare-location-matching/train.csv\")\ntrain_merged = pd.merge(train_data, train_data, on='point_of_interest', suffixes=('_1', '_2'), how='inner')\ntrain_pairs_true = train_merged[train_merged['id_1'] != train_merged['id_2']]\ntrain_pairs_true = train_pairs_true.drop(['point_of_interest'], axis=1)\ntrain_pairs_true['match'] = True\ntrain_pairs_true.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-25T19:24:59.528848Z","iopub.execute_input":"2022-07-25T19:24:59.529525Z","iopub.status.idle":"2022-07-25T19:25:20.148446Z","shell.execute_reply.started":"2022-07-25T19:24:59.529458Z","shell.execute_reply":"2022-07-25T19:25:20.147100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_copy = train_data.copy()\ntrain_data_copy.index = range(1,len(train_data_copy)+1)\ntrain_data_copy = train_data_copy.add_suffix('_2')\nnon_pairs = pd.concat([train_data.add_suffix('_1'),train_data_copy], axis=1).dropna(subset=['point_of_interest_1', 'point_of_interest_2'])\nnon_pairs = non_pairs[non_pairs['point_of_interest_1'] != non_pairs['point_of_interest_2']]\nnon_pairs = non_pairs.drop(['point_of_interest_1', 'point_of_interest_2'], axis=1)\nnon_pairs['match'] = False\n","metadata":{"execution":{"iopub.status.busy":"2022-07-25T19:25:20.151767Z","iopub.execute_input":"2022-07-25T19:25:20.152113Z","iopub.status.idle":"2022-07-25T19:25:25.619012Z","shell.execute_reply.started":"2022-07-25T19:25:20.152055Z","shell.execute_reply":"2022-07-25T19:25:25.617844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_match_loc(test, neighbour = 3):\n    # minimum neighbour: 3 (include itself)\n    if len(test) < neighbour:\n        neighbour = len(test)\n    tree = BallTree(np.deg2rad(test[['latitude', 'longitude']].values), metric='haversine')\n    dist, ind = tree.query(np.deg2rad(test[['latitude', 'longitude']].values), k=neighbour)\n    dist = dist[:,1:].squeeze()\n    ind = ind[:,1:].squeeze()\n    test_col = test.columns.tolist()\n    combine_col = [str + '_1' for str in tqdm(test_col)] + [str + '_2' for str in tqdm(test_col)]\n    df_combine = pd.DataFrame(np.concatenate([\n                np.repeat(np.array(test), neighbour-1, axis = 0),\n                test.iloc[list(itertools.chain.from_iterable(ind.tolist())),:]\n               ], axis=1))    \n    df_combine.columns = combine_col\n    return df_combine  ","metadata":{"execution":{"iopub.status.busy":"2022-07-25T19:25:25.621878Z","iopub.execute_input":"2022-07-25T19:25:25.622217Z","iopub.status.idle":"2022-07-25T19:25:25.632235Z","shell.execute_reply.started":"2022-07-25T19:25:25.622186Z","shell.execute_reply":"2022-07-25T19:25:25.631126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pairs_close = create_match_loc(train_data, neighbour = 3)\ntrain_pairs_close_True = train_pairs_close[train_pairs_close['point_of_interest_1'] == train_pairs_close['point_of_interest_2']]\ntrain_pairs_close_False = train_pairs_close[train_pairs_close['point_of_interest_1'] != train_pairs_close['point_of_interest_2']]\n\ntrain_pairs_close_True = train_pairs_close_True.drop(['point_of_interest_1','point_of_interest_2'], axis=1)\ntrain_pairs_close_False = train_pairs_close_False.drop(['point_of_interest_1','point_of_interest_2'], axis=1)\n\ntrain_pairs_close_True['match'] = True\ntrain_pairs_close_False['match'] = False","metadata":{"execution":{"iopub.status.busy":"2022-07-25T19:25:25.633581Z","iopub.execute_input":"2022-07-25T19:25:25.633876Z","iopub.status.idle":"2022-07-25T19:36:41.297946Z","shell.execute_reply.started":"2022-07-25T19:25:25.633846Z","shell.execute_reply":"2022-07-25T19:36:41.296059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pairs = pd.concat([train_pairs_close_False,non_pairs,train_pairs_true],axis = 0)\ntrain_pairs.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-25T19:36:41.300121Z","iopub.execute_input":"2022-07-25T19:36:41.300469Z","iopub.status.idle":"2022-07-25T19:36:46.821112Z","shell.execute_reply.started":"2022-07-25T19:36:41.300437Z","shell.execute_reply":"2022-07-25T19:36:46.819853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pairs['match'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T19:36:46.823579Z","iopub.execute_input":"2022-07-25T19:36:46.824112Z","iopub.status.idle":"2022-07-25T19:36:46.859742Z","shell.execute_reply.started":"2022-07-25T19:36:46.824033Z","shell.execute_reply":"2022-07-25T19:36:46.858423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pairs = train_pairs.sample(frac=1).reset_index(drop=True) # shuffle\npairs_sample = pd.read_csv('../input/foursquare-location-matching/pairs.csv').iloc[0:2,:]\n# change original data type\ndtype_dict = pairs_sample.dtypes.apply(lambda x: x.name).to_dict()\ndel pairs_sample\ngc.collect()\ntrain_pairs = train_pairs.astype(dtype_dict)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T19:36:46.861839Z","iopub.execute_input":"2022-07-25T19:36:46.862341Z","iopub.status.idle":"2022-07-25T19:37:38.585016Z","shell.execute_reply.started":"2022-07-25T19:36:46.862292Z","shell.execute_reply":"2022-07-25T19:37:38.583521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pairs.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T19:37:38.586617Z","iopub.execute_input":"2022-07-25T19:37:38.586941Z","iopub.status.idle":"2022-07-25T19:37:38.626877Z","shell.execute_reply.started":"2022-07-25T19:37:38.586910Z","shell.execute_reply":"2022-07-25T19:37:38.625433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pairs.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-25T19:37:38.630148Z","iopub.execute_input":"2022-07-25T19:37:38.630499Z","iopub.status.idle":"2022-07-25T19:37:38.637722Z","shell.execute_reply.started":"2022-07-25T19:37:38.630468Z","shell.execute_reply":"2022-07-25T19:37:38.636430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"--- %s seconds ---\" % (time.time() - start_time))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T19:37:38.639246Z","iopub.execute_input":"2022-07-25T19:37:38.639641Z","iopub.status.idle":"2022-07-25T19:37:38.648978Z","shell.execute_reply.started":"2022-07-25T19:37:38.639600Z","shell.execute_reply":"2022-07-25T19:37:38.647911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pairs.to_pickle('./train_pairs_raw.pkl')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T19:37:38.650505Z","iopub.execute_input":"2022-07-25T19:37:38.650878Z","iopub.status.idle":"2022-07-25T19:38:01.532653Z","shell.execute_reply.started":"2022-07-25T19:37:38.650846Z","shell.execute_reply":"2022-07-25T19:38:01.531294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Reference: https://www.kaggle.com/code/ficklemaverick/generating-new-pairs-of-match-and-mismatch","metadata":{}}]}