{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Clean Data","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\npd.options.mode.chained_assignment=None\nimport tensorflow as tf\nimport tensorflow_hub as hub\nimport time\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport string\nimport re\nimport random\nimport gc\nimport fasttext\nprint(\"Tensorflow version:\",tf.__version__)\nprint(\"Num GPUs Available: \", len(tf.config.list_physical_devices('GPU')))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:55:02.783774Z","iopub.execute_input":"2022-07-07T12:55:02.784154Z","iopub.status.idle":"2022-07-07T12:55:09.611441Z","shell.execute_reply.started":"2022-07-07T12:55:02.784038Z","shell.execute_reply":"2022-07-07T12:55:09.610142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**There are a few approaches to gain a clean dataset:**\n1. Fill nan/null values and remove punctuation\n2. Split vacabulary and number (ex. Club123 ==> Club 123)\n3. Split upper and lower words (ex. NextTech ==> Next Tech)\n4. Lower alphabets\n5. Split Chinese, Japanese and Korean characters (ex. 神奈川 ==> 神 奈 川)\n6. Spread noise (whose contains unreasonable words)","metadata":{}},{"cell_type":"code","source":"train=pd.read_csv(\"../input/foursquare-location-matching/train.csv\")\n\ndef split_eastern_asian_language(s):\n    #splitting Chinese, Japanese and Korean characters\n    pattern=\"[\\u3040-\\u30ff\\u3400-\\u4dbf\\u4e00-\\u9fff\\uf900-\\ufaff\\uff66-\\uff9f]\"\n    lst=[]\n    for i in re.finditer(pattern,s):\n        lst.append((i.span(0)[0]))\n    if len(lst)>0:\n        new_s=\"\"\n        for i in range(len(s)):\n            if i in lst:\n                new_s=new_s+s[i]+\" \"\n            else:new_s=new_s+s[i]\n    else:new_s=s\n    #remove unnecessary spaces\n    new_s=new_s.split(\" \")\n    if '' in new_s: new_s.remove('')\n    new_s=\" \".join(new_s)\n\n    return new_s\n\ndef clean_data(train):\n    col=[\"name\",\"city\",\"state\",\"country\",\"categories\",\"address\"]\n    split_number_vocab=r\"(?i)(?<=\\d)(?=[a-z])|(?<=[a-z])(?=\\d)\"\n    split_upper_lower=r\"([a-z](?=[A-Z])|[A-Z](?=[A-Z][a-z]))\"\n    \n    for i in col:\n        #filling nan/null values and removing punctuation\n        train[i]=train[i].str.replace(\"[{}]\".format(string.punctuation),'',regex=True).fillna(\"nan\")\n        train[i][train[i]==\"\"]=0\n        train[i][train[i]==\"ERROR\"]=0\n        train[i]=train[i].astype(\"string\")\n        #splitting vacabulary and number\n        train[i]=train[i].map(lambda x: re.sub(split_number_vocab,\" \", x))\n        #splitting upper and lower words\n        train[i]=train[i].map(lambda x: re.sub(split_upper_lower,\" \", x))\n        train[i]=train[i].str.lower()\n        \n    for i in col[:-1]:\n        \n        #we tried to map numbers to words(123 ==> one two three);\n        #however, the training result is unsatisfactory. But it's always welcome to do more experiments if any interested.\n        \n        #train[i]=train[i].str.replace(\"0\",\"zero \")\n        #train[i]=train[i].str.replace(\"1\",\"one \")\n        #train[i]=train[i].str.replace(\"2\",\"two \")\n        #train[i]=train[i].str.replace(\"3\",\"three \")\n        #train[i]=train[i].str.replace(\"4\",\"four \")\n        #train[i]=train[i].str.replace(\"5\",\"five \")\n        #train[i]=train[i].str.replace(\"6\",\"six \")\n        #train[i]=train[i].str.replace(\"7\",\"seven \")\n        #train[i]=train[i].str.replace(\"8\",\"eight \")\n        #train[i]=train[i].str.replace(\"9\",\"nigh \")\n        train[i]=train[i].map(split_eastern_asian_language)\n    \n    #noise because addresses comprised of these words can be hardly finded in real map.\n    noise=dict()\n    x=train[train[\"address\"].str.contains(\"高仿|微信|精仿\")][\"id\"].to_numpy()\n    if len(x)>0:\n        for i in x:\n            noise[i]=[i]\n            \n    #these are delected locations\n    Deleted=dict()\n    x=train[train[\"name\"].str.contains(\"deleted venue\")][\"id\"].to_numpy()\n    if len(x)>0:\n        lst=\" \".join(x)\n        for i in x:\n            Deleted[i]=lst\n    \n    train=train[~((train[\"address\"].str.contains(\"高仿|微信|精仿\")) | (train[\"name\"].str.contains(\"deleted venue\")))]\n    #pruning unnecessary columns for saving memories\n    train=train[[\"id\",\"latitude\",\"longitude\",\"name\",\"city\",\"state\",\"country\",\"categories\",\"point_of_interest\"]]\n    train=train.sort_values(by=[\"longitude\"])\n    train=train.reset_index(drop=True)\n    \n    return train, noise, Deleted\n\ntrain, noise, Deleted=clean_data(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:55:09.613250Z","iopub.execute_input":"2022-07-07T12:55:09.614038Z","iopub.status.idle":"2022-07-07T12:56:35.971393Z","shell.execute_reply.started":"2022-07-07T12:55:09.614000Z","shell.execute_reply":"2022-07-07T12:56:35.970610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**After these, the dataset should look like this:**","metadata":{}},{"cell_type":"code","source":"gc.collect()\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:56:35.976035Z","iopub.execute_input":"2022-07-07T12:56:35.978306Z","iopub.status.idle":"2022-07-07T12:56:36.403235Z","shell.execute_reply.started":"2022-07-07T12:56:35.978263Z","shell.execute_reply":"2022-07-07T12:56:36.402455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**And for eastern Asian languages, it will look like this:**","metadata":{}},{"cell_type":"code","source":"train[(train[\"country\"]==\"tw\")|(train[\"country\"]==\"jp\")]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:56:36.405264Z","iopub.execute_input":"2022-07-07T12:56:36.408011Z","iopub.status.idle":"2022-07-07T12:56:37.107638Z","shell.execute_reply.started":"2022-07-07T12:56:36.407972Z","shell.execute_reply":"2022-07-07T12:56:37.106921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**There are over 739k different PoIs but only 42.5% of PoIs consist of 2 or more data points, and in those PoIs, only less than 2% have 6 or higher number of points.**","metadata":{}},{"cell_type":"code","source":"PoI_lst=train.groupby(\"point_of_interest\").size().sort_values(ascending=False)\nmultiple_PoI_lst=PoI_lst[PoI_lst>1]\nprint(\"the number of PoIs:\",PoI_lst.shape[0])\nprint(\"the percentage of repeated PoIs:\",\"{:.4f}%\".format((multiple_PoI_lst.shape[0]/PoI_lst.shape[0])*100))\nprint(\"the percentage of repeated PoIs(less than 5 times):\",\"{:.4f}%\".format((multiple_PoI_lst[multiple_PoI_lst<=5].shape[0]/PoI_lst.shape[0])*100))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:56:37.111710Z","iopub.execute_input":"2022-07-07T12:56:37.111990Z","iopub.status.idle":"2022-07-07T12:56:40.690222Z","shell.execute_reply.started":"2022-07-07T12:56:37.111953Z","shell.execute_reply":"2022-07-07T12:56:40.689475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#randomly picking PoIs for training\nsingle_PoI_index=list(PoI_lst[PoI_lst==1].index)\nmultiple_PoI_index=list(multiple_PoI_lst.index)\nmultiple_PoI_lst_lessthan5=list(multiple_PoI_lst[multiple_PoI_lst<=5].index)\n\nrandom.shuffle(single_PoI_index)\nrandom.shuffle(multiple_PoI_index)\nrandom.shuffle(multiple_PoI_lst_lessthan5)\n\ntrain_index=single_PoI_index[:2000]+multiple_PoI_index[:200000]\ntest_index=single_PoI_index[-40:]+multiple_PoI_index[-4000:]\nrandom.shuffle(train_index)\nrandom.shuffle(test_index)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:56:40.693927Z","iopub.execute_input":"2022-07-07T12:56:40.696246Z","iopub.status.idle":"2022-07-07T12:56:43.337672Z","shell.execute_reply.started":"2022-07-07T12:56:40.696206Z","shell.execute_reply":"2022-07-07T12:56:43.336903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Then, we measure the possible range of rows for any PoI. Because now the training set is ordered by longitude, thus it can also be looked upon as the degree of deviation.**<br>\n**It turns out that for nearly 95% PoIs, the set of their data points are concentrated within the range of 1200 rows. For example, if data point 0 share the same PoI with others, then we can expect these points should fall onto 1~1200.**<br>","metadata":{}},{"cell_type":"code","source":"def range_of_PoI(lst):\n    PoI_Lst=train[\"point_of_interest\"].to_numpy().astype(\"str\")\n    L=[]\n    #we just sample it rather than the whole dataset for saving time\n    #we tried numba but it has some issue with string and cannot really help\n    #if you have any better idea to speed this function up, please tell us\n    for i in lst[:12000]: \n        y=np.where(PoI_Lst==i)[0]\n        m=min(y)\n        M=max(y)\n        L.append(M-m)\n    return L\nt0=time.time()\n\nRangeofPoI=range_of_PoI(multiple_PoI_index)\nRangeofPoI_lessthan5=range_of_PoI(multiple_PoI_lst_lessthan5)\n\nprint(\"total time spent:\",\"{:.4f}\".format(time.time()-t0))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T12:56:43.338690Z","iopub.execute_input":"2022-07-07T12:56:43.338951Z","iopub.status.idle":"2022-07-07T13:02:22.965001Z","shell.execute_reply.started":"2022-07-07T12:56:43.338916Z","shell.execute_reply":"2022-07-07T13:02:22.964244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import stats\nPoI_range=[]\nPoI_range_lessthan5=[]\nk=1200\nper_k0=stats.percentileofscore(RangeofPoI, k)/100\nper_k1=stats.percentileofscore(RangeofPoI_lessthan5, k)/100\nfor i in range(1,66):\n    PoI_range.append(np.quantile(RangeofPoI,i/66,axis=0))\nfor i in range(1,66):\n    PoI_range_lessthan5.append(np.quantile(RangeofPoI_lessthan5,i/66,axis=0))\nplt.plot(np.arange(1,66)/66,PoI_range)\nplt.plot(np.arange(1,66)/66,PoI_range_lessthan5,c=\"g\")\nplt.plot(per_k1,k,'go')\nplt.text(per_k1-0.26,k,\"(%s, %s)\"%(\"{:.4f}\".format(per_k1),k))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:02:22.966507Z","iopub.execute_input":"2022-07-07T13:02:22.967319Z","iopub.status.idle":"2022-07-07T13:02:23.541378Z","shell.execute_reply.started":"2022-07-07T13:02:22.967280Z","shell.execute_reply":"2022-07-07T13:02:23.540712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Embed and Patch Words","metadata":{}},{"cell_type":"markdown","source":"**We first tried self-trained FastText method but the result is quite rubbishy (maybe it's because of too small dataset). GloVe from Stanford (https://nlp.stanford.edu/projects/glove/), by contrast, did a good job in the aspects of either accuracy or efficiency.**","metadata":{}},{"cell_type":"code","source":"if False:\n    data=(train[\"name\"]+\" <s> \"+train[\"categories\"]+\" <s> \"+train[\"country\"]+\" <s> \"+train[\"state\"]+\" <s> \"+train[\"city\"]).values\n    np.savetxt(\"data.txt\",data,fmt=\"%s\")\n    fasttext_model = fasttext.train_unsupervised('data.txt', model='skipgram')\n    fasttext_model.save_model(\"foursquare_model.bin\")\n    fasttext_model=fasttext.load_model(\"../input/fasttext-for-foursquare/foursquare_model.bin\")","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:02:23.542696Z","iopub.execute_input":"2022-07-07T13:02:23.542901Z","iopub.status.idle":"2022-07-07T13:02:23.548718Z","shell.execute_reply.started":"2022-07-07T13:02:23.542874Z","shell.execute_reply":"2022-07-07T13:02:23.547970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"glove_twitter=dict()\nwith open(\"../input/glove-twitter-27b/glove.twitter.27B.50d.txt\") as f:\n    for i in f:\n        line=i.split(\" \")\n        line[-1]=re.sub(r\"\\n\",\"\",line[-1])\n        glove_twitter[line[0]]=[float(i) for i in line[1:]]\n\nprint(\"number of words in Glove:\",len(glove_twitter.keys()))\n#however, Glove is not perfect. There are many special nouns (name of shops, roads and so on) not in the dictionary.\n#and also, it doesn't works very well in other languages.","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:02:23.552197Z","iopub.execute_input":"2022-07-07T13:02:23.552655Z","iopub.status.idle":"2022-07-07T13:02:51.219894Z","shell.execute_reply.started":"2022-07-07T13:02:23.552619Z","shell.execute_reply":"2022-07-07T13:02:51.219126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We only acquire 5 columns for embedding: name, categories, country, state and city.**","metadata":{}},{"cell_type":"code","source":"long_name=[]\nfor i in train[\"name\"].to_numpy():\n    x=i.split(\" \")\n    x=len([i for i in x if 1!=\"\"])\n    long_name.append(x)\nlong_categories=[]\nfor i in train[\"categories\"].to_numpy():\n    x=i.split(\" \")\n    x=len([i for i in x if 1!=\"\"])\n    long_categories.append(x)\nlong_country=[]\nfor i in train[\"country\"].to_numpy():\n    x=i.split(\" \")\n    x=len([i for i in x if 1!=\"\"])\n    long_country.append(x)\nlong_state=[]\nfor i in train[\"state\"].to_numpy():\n    x=i.split(\" \")\n    x=len([i for i in x if 1!=\"\"])\n    long_state.append(x)\nlong_city=[]\nfor i in train[\"city\"].to_numpy():\n    x=i.split(\" \")\n    x=len([i for i in x if 1!=\"\"])\n    long_city.append(x)\n\nfor i in [(\"name\",long_name),(\"categories\",long_categories),(\"country\",long_country),(\"state\",long_state),(\"city\",long_city)]: \n    print(\"the longest word of %s:\"%i[0],np.max(i[1]))\n    print(\"the long of 99.5th word of %s:\"%i[0],np.quantile(i[1],0.995),\"\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:02:51.221170Z","iopub.execute_input":"2022-07-07T13:02:51.221501Z","iopub.status.idle":"2022-07-07T13:02:59.007946Z","shell.execute_reply.started":"2022-07-07T13:02:51.221464Z","shell.execute_reply":"2022-07-07T13:02:59.006193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Write a helper function to embed dataset for our neural network.**","metadata":{}},{"cell_type":"code","source":"def embed_data(data,word_embedding,embedded_len=50,poi=True):\n    output=[]\n    x=data\n    \n    def embed_num(num,k):\n        try: \n            num=int(num)\n        except Exception as e:\n            return  [0 for i in range(k)]\n        \n        #because Glove doesn't understand numbers, we need to build our own number embedding list.\n        long=len(str(num))\n        term=[0.]*k\n        for i in range(long+1):\n            x=(num//(10**(long-i)))\n            term[long-i]=x/10\n            num=num-x*(10**(long-i))\n        return term\n    \n    Id=x[\"id\"]\n    if poi==True: PoI=x[\"point_of_interest\"]\n    lat_long=[x[\"latitude\"],x[\"longitude\"]]\n    \n    \n    text1=x[\"name\"]+r\" <s> \"+x[\"categories\"]\n    text1=text1.split(\" \")\n    text1=(text1+[\"0\"]*30)[:30] # padding [name, categories] list\n    output=[]\n    for j in text1:\n        try: output.append(word_embedding[j])\n        except Exception as e: \n            output.append(embed_num(j,embedded_len))\n            \n    text2=x[\"country\"]+r\" <s> \"+x[\"state\"]+r\" <s> \"+x[\"city\"]\n    text2=text2.split(\" \")\n    text2=(text2+[\"0\"]*15)[:15] # padding [country, state, city] list\n    country_state=[]\n    for j in text2:\n        try: country_state.append(word_embedding[j])\n        except Exception as e: \n            country_state.append(embed_num(j,embedded_len))\n            \n    if poi==True: return output, lat_long, country_state, Id, PoI\n    else: return output, lat_long, country_state, Id","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:02:59.009439Z","iopub.execute_input":"2022-07-07T13:02:59.009720Z","iopub.status.idle":"2022-07-07T13:02:59.021348Z","shell.execute_reply.started":"2022-07-07T13:02:59.009682Z","shell.execute_reply":"2022-07-07T13:02:59.020572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Then, we create a training set. it's impossibel to squeeze all dataset into the network (RAM will be choked). Therefore, for each PoI as an unit, we randomly select a small portion of data and train the model repeatedly.**","metadata":{}},{"cell_type":"code","source":"def create_train_data(k0,k,word_embedding,train,train_index):\n\n    token_anchor=[]\n    lat_long_anchor=[]\n    country_state_anchor=[]\n    \n    token_feature=[]\n    lat_long_feature=[]\n    country_state_feature=[]\n    \n    labels=[]\n    \n    ID_anchor=[]\n    ID_feature=[]\n    \n    for i in range(k0,k):\n        total_picked_sampe=660\n        random_num=15\n        anchor_index=random.choices(train[train[\"point_of_interest\"]==train_index[i]].index,k=random_num)\n        \n        #determining the window for picking data points\n        m=max(0,min(anchor_index)-2500) \n        M=min(train.shape[0]-1,max(anchor_index)+2500)\n        \n        feature_index=random.sample([i for i in range(m,M) if i not in anchor_index],total_picked_sampe-random_num)\n\n        indices1=random.choices(anchor_index,k=total_picked_sampe)\n        indices2=random.sample(anchor_index+feature_index,total_picked_sampe)\n\n        for j in indices2:\n            if j in anchor_index:labels.append([1])\n            else: labels.append([0])\n    \n        for j in indices1:\n            x=train.iloc[j]\n            output, lat_long, country_state, Id, PoI=embed_data(x,word_embedding)\n            token_anchor.append(output)\n            lat_long_anchor.append(lat_long)\n            country_state_anchor.append(country_state)\n            #ID_anchor.append(Id)\n\n        for j in indices2:\n            x=train.iloc[j]\n            output, lat_long, country_state, Id, PoI=embed_data(x,word_embedding)\n            token_feature.append(output)\n            lat_long_feature.append(lat_long)\n            country_state_feature.append(country_state)\n            #ID_feature.append(Id)\n    \n    return token_anchor, lat_long_anchor, country_state_anchor, token_feature, lat_long_feature, country_state_feature, labels","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:02:59.022745Z","iopub.execute_input":"2022-07-07T13:02:59.023344Z","iopub.status.idle":"2022-07-07T13:02:59.037463Z","shell.execute_reply.started":"2022-07-07T13:02:59.023306Z","shell.execute_reply":"2022-07-07T13:02:59.036613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Siamese Network","metadata":{}},{"cell_type":"markdown","source":"**We did this before. If you are interested in knowing the method, please read the following materials:**\n* C4W4L03 Siamese Network: https://youtu.be/6jfw8MuKwpI\n* Few-Shot Learning (2/3): Siamese Networks: https://youtu.be/4S-XDefSjTM\n* My Previous Work: https://www.kaggle.com/code/jimkaihuang/location-matching-siamese-network","metadata":{}},{"cell_type":"code","source":"def baseline_model():\n    \n    inputs_word=tf.keras.Input(shape=(None,50),name=\"word\")\n    word=tf.keras.layers.Bidirectional(tf.keras.layers.GRU(100,return_sequences=True))(inputs_word)\n    word=tf.keras.layers.Bidirectional(tf.keras.layers.GRU(100,return_sequences=True))(word)\n    word=tf.keras.layers.Bidirectional(tf.keras.layers.GRU(100,return_sequences=False))(word)\n    word=tf.keras.layers.Dense(100)(word)\n    word=tf.keras.layers.Dense(100,activation=\"relu\")(word)\n    word=tf.keras.layers.Dense(100,activation=\"relu\")(word)\n    word=tf.keras.layers.Dense(100)(word)\n    word=tf.keras.layers.Dense(70,activation=\"tanh\",name=\"word_output\")(word)\n    \n    inputs_lat_long=tf.keras.Input(shape=(2),name=\"lat_long\")\n    lat_long=inputs_lat_long\n    \n    inputs_country_state=tf.keras.Input(shape=(None,50),name=\"country_state\")\n    country_state=tf.keras.layers.Bidirectional(tf.keras.layers.GRU(16,return_sequences=True))(inputs_country_state)\n    country_state=tf.keras.layers.Bidirectional(tf.keras.layers.GRU(16,return_sequences=True))(inputs_country_state)\n    country_state=tf.keras.layers.Bidirectional(tf.keras.layers.GRU(16,return_sequences=False))(country_state)\n    country_state=tf.keras.layers.Dense(16)(country_state)\n    country_state=tf.keras.layers.Dense(16,activation=\"relu\")(country_state)\n    country_state=tf.keras.layers.Dense(16,activation=\"relu\")(country_state)\n    country_state=tf.keras.layers.Dense(16)(country_state)\n    country_state=tf.keras.layers.Dense(8,activation=\"tanh\",name=\"country_state_output\")(country_state)\n    \n    model=tf.keras.Model(inputs=[inputs_word,inputs_lat_long,inputs_country_state], \n                         outputs=[word,lat_long,country_state])\n    return model\n\nbaseline=baseline_model()\nbaseline.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:02:59.040371Z","iopub.execute_input":"2022-07-07T13:02:59.041127Z","iopub.status.idle":"2022-07-07T13:03:04.167608Z","shell.execute_reply.started":"2022-07-07T13:02:59.041085Z","shell.execute_reply":"2022-07-07T13:03:04.166908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We simplify the architecture. Now the model with roughly 523k trainable parameters contains three parts:**\n1. word (name + catagories) ==> ( ,70) output shape\n2. lat_long (latitude and longitude) ==> ( ,2) output shape\n3. country_state (country + state + city) ==> ( ,8) output shape","metadata":{}},{"cell_type":"code","source":"tf.keras.utils.plot_model(baseline,show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:03:04.169926Z","iopub.execute_input":"2022-07-07T13:03:04.170189Z","iopub.status.idle":"2022-07-07T13:03:05.491552Z","shell.execute_reply.started":"2022-07-07T13:03:04.170155Z","shell.execute_reply":"2022-07-07T13:03:05.490485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The Siamese part is even simpler. It now just calculate the sum of square error for each part and add them up with weights.**","metadata":{}},{"cell_type":"code","source":"def diff_of_square(vec):\n    x, y=vec\n    diff=(x-y)**2\n    output=tf.math.reduce_sum(diff,axis=1,keepdims=True)\n    return output\n\ndef siamese_model(baseline): \n\n    anchor_word=tf.keras.Input(shape=(None,50),name=\"token_anchor\")\n    anchor_lat_long=tf.keras.Input(shape=(2),name=\"lat_long_anchor\")\n    anchor_country_state=tf.keras.Input(shape=(None,50),name=\"country_state_anchor\")\n    \n    feature_word=tf.keras.Input(shape=(None,50),name=\"token_feature\")\n    feature_lat_long=tf.keras.Input(shape=(2),name=\"lat_long_feature\")\n    feature_country_state=tf.keras.Input(shape=(None,50),name=\"country_state_feature\")\n\n    anchor=baseline([anchor_word,anchor_lat_long,anchor_country_state])\n    feature=baseline([feature_word,feature_lat_long,feature_country_state])\n\n    cross01=tf.keras.layers.Lambda(diff_of_square,name=\"cross_all\")([anchor[0],feature[0]])\n    lat_long=tf.keras.layers.Lambda(diff_of_square,name=\"lat_long\")([anchor[1],feature[1]])\n    cross02=tf.keras.layers.Lambda(diff_of_square,name=\"cross_geography\")([anchor[2],feature[2]])\n    \n    \n    x=tf.keras.layers.Concatenate()([cross01,lat_long,cross02])\n    output=tf.keras.layers.Dense(1,activation=\"sigmoid\",name=\"score\")(x)\n\n    model=tf.keras.Model(\n            inputs=[anchor_word,anchor_lat_long,anchor_country_state,feature_word,feature_lat_long,feature_country_state],\n            outputs=output)\n    return model\n\nsiamese=siamese_model(baseline)\nsiamese.load_weights(\"../input/siamese/siamese\")\nsiamese.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:03:05.493161Z","iopub.execute_input":"2022-07-07T13:03:05.493958Z","iopub.status.idle":"2022-07-07T13:03:09.460497Z","shell.execute_reply.started":"2022-07-07T13:03:05.493920Z","shell.execute_reply":"2022-07-07T13:03:09.459770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(siamese,show_shapes=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:03:09.462004Z","iopub.execute_input":"2022-07-07T13:03:09.462288Z","iopub.status.idle":"2022-07-07T13:03:09.739516Z","shell.execute_reply.started":"2022-07-07T13:03:09.462251Z","shell.execute_reply":"2022-07-07T13:03:09.738721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We manually set up weights in the last layer for speeding up the training process; however, this step is totally optional.**","metadata":{}},{"cell_type":"code","source":"if False:\n    weights=np.array([[-0.05],[-0.1],[-0.05]],dtype=\"float32\")\n    bias=np.array([0.1],dtype=\"float32\")\n    siamese.layers[-1].set_weights([weights,bias])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:03:09.741529Z","iopub.execute_input":"2022-07-07T13:03:09.741811Z","iopub.status.idle":"2022-07-07T13:03:09.747452Z","shell.execute_reply.started":"2022-07-07T13:03:09.741776Z","shell.execute_reply":"2022-07-07T13:03:09.746208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"siamese.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4),\n    loss=tf.keras.losses.BinaryCrossentropy(),\n    metrics=[tf.keras.metrics.BinaryAccuracy(name='accuracy'),\n             tf.keras.metrics.Precision(name='precision'),\n             tf.keras.metrics.Recall(name='recall')])\n\ncallbacks=tf.keras.callbacks.EarlyStopping(monitor='accuracy',patience=4) #automatically stop if there isn't any improvement","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:03:09.749186Z","iopub.execute_input":"2022-07-07T13:03:09.749580Z","iopub.status.idle":"2022-07-07T13:03:09.783613Z","shell.execute_reply.started":"2022-07-07T13:03:09.749547Z","shell.execute_reply":"2022-07-07T13:03:09.782916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we'd already trained this model. It's just here for demonstration.\naccuracy=[]\nprecision=[]\nrecall=[]\nfor l in [0]:\n    t0=time.time()\n    token_anchor, lat_long_anchor, country_state_anchor, token_feature, lat_long_feature, country_state_feature, labels=create_train_data(l*76,(l+1)*76,glove_twitter,train,train_index)\n    print(\"data finished:\",\"{:.4f}\".format(time.time()-t0))\n\n    t0=time.time()\n    inputs=[tf.constant(token_anchor),tf.constant(lat_long_anchor),tf.constant(country_state_anchor),\n            tf.constant(token_feature),tf.constant(lat_long_feature),tf.constant(country_state_feature)]\n\n    his=siamese.fit(x=inputs,y=tf.constant(labels),epochs=20,verbose=1,batch_size=4000,callbacks=[callbacks])\n    #siamese.save_weights(\"siamese\")\n    print(\"%s-%sth training finished:\"%(l,l),\"{:.4f}\".format(time.time()-t0))\n\n    Acc=np.array(his.history[\"accuracy\"]).mean()\n    Pre=np.array(his.history[\"precision\"]).mean()\n    Rec=np.array(his.history[\"recall\"]).mean()\n    accuracy.append(Acc)\n    precision.append(Pre)\n    recall.append(Rec)\n    print(\"average accuracy:\",\"{:.4f}%\".format(Acc*100))\n    print(\"average precision:\",\"{:.4f}%\".format(Pre*100))\n    print(\"average recall:\",\"{:.4f}%\".format(Rec*100),\"\\n\")\n    del token_anchor, lat_long_anchor, country_state_anchor, token_feature, lat_long_feature, country_state_feature, labels, inputs\ndel siamese\nsiamese=siamese_model(baseline)\nsiamese.load_weights(\"../input/siamese/siamese\")\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:03:09.785174Z","iopub.execute_input":"2022-07-07T13:03:09.785446Z","iopub.status.idle":"2022-07-07T13:07:06.567201Z","shell.execute_reply.started":"2022-07-07T13:03:09.785408Z","shell.execute_reply":"2022-07-07T13:07:06.566345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3-1. What Exactly Siamese Saw","metadata":{}},{"cell_type":"markdown","source":"**The last layer of our model decides the weights of word, lat_long and country_state.<br>\nWe first observe the Bias which sets an initial status of each comparison, so if the training process matches two identical data points or two points infinitesimally close, then it gets the highest score *4.659*.<br>**","metadata":{}},{"cell_type":"code","source":"baseline_model=tf.keras.Model(inputs=siamese.layers[-6].input,outputs=siamese.layers[-6].output)\ndistant=(siamese.layers[-1].weights[0].numpy())\nBias=siamese.layers[-1].weights[1].numpy()\nprint(\"Bias:\",Bias[0])\nprint(\"weights:\",distant[:,0])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:07:41.023463Z","iopub.execute_input":"2022-07-07T13:07:41.023721Z","iopub.status.idle":"2022-07-07T13:07:41.036908Z","shell.execute_reply.started":"2022-07-07T13:07:41.023692Z","shell.execute_reply":"2022-07-07T13:07:41.035778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_data_feature(k0,k,data,word_embedding,model,PoI):\n    x0=[]\n    x1=[]\n    x2=[]\n    point_of_interest=[]\n    ID=[]\n    t0=time.time()\n    \n    if PoI==True:\n        for i in range(k0,k):\n            word, lat_long, country_state, Id, PoI=embed_data(data.iloc[i],word_embedding,embedded_len=50,poi=True)\n            x0.append(word)\n            x1.append(lat_long)\n            x2.append(country_state)\n            point_of_interest.append(PoI)\n            ID.append(Id)\n        output=model([tf.constant(x0),tf.constant(x1),tf.constant(x2)])\n        point_of_interest=np.array(point_of_interest)\n        ID=np.array(ID)\n        print(\"task finished:\",\"{:.4f}\".format(time.time()-t0))\n        return output, point_of_interest, ID\n    else:\n        for i in range(k0,k):\n            word, lat_long, country_state, Id=embed_data(data.iloc[i],word_embedding,embedded_len=50,poi=False)\n            x0.append(word)\n            x1.append(lat_long)\n            x2.append(country_state)\n            ID.append(Id)\n        output=model([tf.constant(x0),tf.constant(x1),tf.constant(x2)])\n        ID=np.array(ID)\n        print(\"task finished:\",\"{:.4f}\".format(time.time()-t0))\n        return output, ID\n\n#with GPU, this process is extremely fast.\noutput, ID=get_data_feature(0,2000,train,glove_twitter,baseline_model,PoI=False)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:07:42.397089Z","iopub.execute_input":"2022-07-07T13:07:42.397581Z","iopub.status.idle":"2022-07-07T13:07:44.720549Z","shell.execute_reply.started":"2022-07-07T13:07:42.397544Z","shell.execute_reply":"2022-07-07T13:07:44.719877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Now, we exame the model. Point 0 refers the same PoI to point 1 and 2 and Siamese successfully spots it out.**","metadata":{}},{"cell_type":"code","source":"train.iloc[[0,1,2,3,4]]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:07:46.063056Z","iopub.execute_input":"2022-07-07T13:07:46.063999Z","iopub.status.idle":"2022-07-07T13:07:46.080013Z","shell.execute_reply.started":"2022-07-07T13:07:46.063952Z","shell.execute_reply":"2022-07-07T13:07:46.079292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def difference_root(x,y):\n    x=np.array(x)\n    y=np.array(y)\n    diff=((x-y)**2)\n    output=np.sum(diff)\n    return output\n\ndef compare(k):\n    word=[]\n    lat_long=[]\n    state_city=[]\n    t0=time.time()\n    for j in range(k,k+1200): #using the window of 1200 rows\n        word.append(difference_root(output[0][k],output[0][j]))\n        lat_long.append(difference_root(output[1][k],output[1][j]))\n        state_city.append(difference_root(output[2][k],output[2][j]))\n    print(\"task finished:\",\"{:.4f}\".format(time.time()-t0))\n    return np.array(word), np.array(lat_long), np.array(state_city)\n\nk=0\nword, lat_long, state_city=compare(k)\nprint(\"the most likely points for no.%s:\"%k,np.where(Bias+word*distant[0]+lat_long*distant[1]+state_city*distant[2]>0)[0]+k)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:07:48.760739Z","iopub.execute_input":"2022-07-07T13:07:48.761333Z","iopub.status.idle":"2022-07-07T13:07:51.352488Z","shell.execute_reply.started":"2022-07-07T13:07:48.761293Z","shell.execute_reply":"2022-07-07T13:07:51.351558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axe=plt.subplots(4,1,figsize=(18,17))\n\naxe[0].plot(np.arange(len(word))+k,(word*distant[0]))\naxe[0].plot([0,len(word)],[0,0],\"--\",c='k',lw=0.75)\naxe[0].set_xlim([k,k+100])\naxe[0].set_title(\"Square Difference of Name and Catagories\")\n\naxe[1].plot(np.arange(len(lat_long))+k,(lat_long*distant[1]),c=\"c\")\naxe[1].plot([0,len(lat_long)],[0,0],\"--\",c='k',lw=0.75)\naxe[1].set_xlim([k,k+100])\naxe[1].set_title(\"Square Difference of latitude and longitude\")\n\naxe[2].plot(np.arange(len(state_city))+k,(state_city*distant[2]),c=\"g\")\naxe[2].plot([0,len(state_city)],[0,0],\"--\",c='k',lw=0.75)\naxe[2].set_xlim([k,k+100])\naxe[2].set_title(\"Square Difference of Country, State and City\")\n\naxe[3].plot(np.arange(len(state_city))+k,(Bias+word*distant[0]+lat_long*distant[1]+state_city*distant[2]),c=\"m\")\naxe[3].plot([0,len(lat_long)],[0,0],\"--\",c='k',lw=0.75)\naxe[3].set_xlim([k,k+100])\naxe[3].set_title(\"The Score of likelihood\")\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:07:55.033490Z","iopub.execute_input":"2022-07-07T13:07:55.033793Z","iopub.status.idle":"2022-07-07T13:07:55.509360Z","shell.execute_reply.started":"2022-07-07T13:07:55.033761Z","shell.execute_reply":"2022-07-07T13:07:55.508647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del output, ID, word, lat_long, state_city\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:07:11.538740Z","iopub.execute_input":"2022-07-07T13:07:11.539144Z","iopub.status.idle":"2022-07-07T13:07:12.889734Z","shell.execute_reply.started":"2022-07-07T13:07:11.539107Z","shell.execute_reply":"2022-07-07T13:07:12.889012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Submission","metadata":{}},{"cell_type":"code","source":"def Siamese_model():\n    word_anchor=tf.keras.Input(shape=(70))\n    lat_long_anchor=tf.keras.Input(shape=(2))\n    country_city_anchor=tf.keras.Input(shape=(8))\n    \n    word_feature=tf.keras.Input(shape=(70))\n    lat_long_feature=tf.keras.Input(shape=(2))\n    country_city_feature=tf.keras.Input(shape=(8))\n    \n    word=tf.keras.layers.Lambda(diff_of_square,name=\"cross_all\")([word_anchor,word_feature])\n    lat_long=tf.keras.layers.Lambda(diff_of_square,name=\"lat_long\")([lat_long_anchor,lat_long_feature])\n    country_city=tf.keras.layers.Lambda(diff_of_square,name=\"cross_geography\")([country_city_anchor,country_city_feature])\n    \n    x=tf.keras.layers.Concatenate()([word,lat_long,country_city])\n    output=tf.keras.layers.Dense(1,activation=\"sigmoid\",name=\"score\")(x)\n\n    diff_model=tf.keras.Model(\n            inputs=[word_anchor,lat_long_anchor,country_city_anchor,word_feature,lat_long_feature,country_city_feature],\n        outputs=output)\n    \n    return diff_model\n\nsiamese_model=Siamese_model()\nsiamese_model.layers[-1].set_weights([distant,Bias])\nif (siamese_model.weights[0].numpy()==distant).all() and (siamese_model.weights[1].numpy()==Bias).all():\n    print(\"everything has done\")\n    print(\"Siamese is ready to go\")","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:07:12.891010Z","iopub.execute_input":"2022-07-07T13:07:12.891280Z","iopub.status.idle":"2022-07-07T13:07:12.938248Z","shell.execute_reply.started":"2022-07-07T13:07:12.891243Z","shell.execute_reply":"2022-07-07T13:07:12.937548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test=pd.read_csv(\"../input/foursquare-location-matching/test.csv\")\n\ndef split_eastern_asian_language(s):\n    pattern=\"[\\u3040-\\u30ff\\u3400-\\u4dbf\\u4e00-\\u9fff\\uf900-\\ufaff\\uff66-\\uff9f]\"\n    lst=[]\n    for i in re.finditer(pattern,s):\n        lst.append((i.span(0)[0]))\n    if len(lst)>0:\n        new_s=\"\"\n        for i in range(len(s)):\n            if i in lst:\n                new_s=new_s+s[i]+\" \"\n            else:new_s=new_s+s[i]\n    else:new_s=s\n    #remove unnecessary spaces\n    new_s=new_s.split(\" \")\n    if '' in new_s: new_s.remove('')\n    new_s=\" \".join(new_s)\n\n    return new_s\n    \n\ndef clean_test_data(test):\n    col=[\"name\",\"city\",\"state\",\"country\",\"categories\",\"address\"]\n    split_number_vocab=r\"(?i)(?<=\\d)(?=[a-z])|(?<=[a-z])(?=\\d)\"\n    split_upper_lower=r\"([a-z](?=[A-Z])|[A-Z](?=[A-Z][a-z]))\"\n    \n    for i in col:\n        #filling nan/blank values and removing punctuation\n        test[i]=test[i].str.replace(\"[{}]\".format(string.punctuation),'',regex=True).fillna(\"nan\")\n        test[i][test[i]==\"\"]=0\n        test[i][test[i]==\"ERROR\"]=0\n        test[i]=test[i].astype(\"string\")\n        #splitting vacabulary and number\n        test[i]=test[i].map(lambda x: re.sub(split_number_vocab,\" \", x))\n        #splitting upper and lower words\n        test[i]=test[i].map(lambda x: re.sub(split_upper_lower,\" \", x))\n        test[i]=test[i].str.lower()\n        \n    for i in col[:-1]:\n        #test[i]=test[i].str.replace(\"0\",\"zero \")\n        #test[i]=test[i].str.replace(\"1\",\"one \")\n        #test[i]=test[i].str.replace(\"2\",\"two \")\n        #test[i]=test[i].str.replace(\"3\",\"three \")\n        #test[i]=test[i].str.replace(\"4\",\"four \")\n        #test[i]=test[i].str.replace(\"5\",\"five \")\n        #test[i]=test[i].str.replace(\"6\",\"six \")\n        #test[i]=test[i].str.replace(\"7\",\"seven \")\n        #test[i]=test[i].str.replace(\"8\",\"eight \")\n        #test[i]=test[i].str.replace(\"9\",\"nigh \")\n        test[i]=test[i].map(split_eastern_asian_language)\n    \n    #noise because addresses comprised of these words can be hardly finded in real map.\n    noise=dict()\n    x=test[test[\"address\"].str.contains(\"高仿|微信|精仿\")][\"id\"].to_numpy()\n    if len(x)>0:\n        for i in x:\n            noise[i]=[i]\n            \n    Deleted=dict()\n    x=test[test[\"name\"].str.contains(\"deleted venue\")][\"id\"].to_numpy()\n    if len(x)>0:\n        lst=\" \".join(x)\n        for i in x:\n            Deleted[i]=lst\n    \n    test=test[~((test[\"address\"].str.contains(\"高仿|微信|精仿\")) | (test[\"name\"].str.contains(\"Deleted Venue SEO\")))]\n    test=test[[\"id\",\"latitude\",\"longitude\",\"name\",\"city\",\"state\",\"country\",\"categories\"]]\n    test=test.sort_values(by=[\"longitude\"])\n    test=test.reset_index(drop=True)\n    \n    return test, noise, Deleted\n\ntest, noise, Deleted=clean_test_data(test)\ntest","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:07:12.942146Z","iopub.execute_input":"2022-07-07T13:07:12.942338Z","iopub.status.idle":"2022-07-07T13:07:13.013593Z","shell.execute_reply.started":"2022-07-07T13:07:12.942315Z","shell.execute_reply":"2022-07-07T13:07:13.012741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def matching_data(test,div=20000,window=1200):\n    #20k for each batch to avoid the system crashed\n    submit_dict=dict()\n    num=len(test)//div\n    mod=len(test)%div\n    \n    for j in range(num+1):\n        \n        if j<num:\n            K0=j*div\n            K0_1=(j+1)*div\n        elif mod>0:\n            K0=num*div+0\n            K0_1=num*div+mod\n        else:break #if there is nothing left, stop the whole loop\n\n        if j<num-1:check=window\n        elif j==num-1:check=min(window,mod)\n        else:check=0\n        \n        output, ID=get_data_feature(max(0,K0-window),K0_1+check,test,glove_twitter,baseline_model,PoI=False)\n        \n        t0=time.time()\n        for i in range(K0,K0_1):\n            start=max(i-window,0)-max(0,K0-window)\n            end=min(window+i,K0_1+check)-max(0,K0-window)\n\n            token_anchor=tf.repeat([output[0][i-max(0,K0-window)]],end-start,axis=0)\n            lat_long_anchor=tf.repeat([output[1][i-max(0,K0-window)]],end-start,axis=0)\n            country_state_anchor=tf.repeat([output[2][i-max(0,K0-window)]],end-start,axis=0)\n\n            token_feature=output[0][start:end]\n            lat_long_feature=output[1][start:end]\n            country_state_feature=output[2][start:end]\n            \n            \n            inputs=[token_anchor,lat_long_anchor,country_state_anchor,\n                   token_feature,lat_long_feature,country_state_feature]\n            score=siamese_model(inputs)\n            \n            q=np.where(score.numpy().reshape(end-start)>0.5)[0]+start\n            matched_id=[\" \".join([ID[k] for k in q])]\n            \n            submit_dict[ID[i-max(0,K0-window)]]=matched_id\n            \n        print(\"%sth run is finished:\"%K0_1,\"{:.4f}\".format(time.time()-t0),\"\\n\")\n    return submit_dict","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:07:13.014908Z","iopub.execute_input":"2022-07-07T13:07:13.015230Z","iopub.status.idle":"2022-07-07T13:07:13.034598Z","shell.execute_reply.started":"2022-07-07T13:07:13.015194Z","shell.execute_reply":"2022-07-07T13:07:13.033908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit_dict=matching_data(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:07:13.035619Z","iopub.execute_input":"2022-07-07T13:07:13.035810Z","iopub.status.idle":"2022-07-07T13:07:13.124961Z","shell.execute_reply.started":"2022-07-07T13:07:13.035786Z","shell.execute_reply":"2022-07-07T13:07:13.124221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def submission_output(submit_dict,noise,Deleted):\n    submit_dict.update(noise)\n    submit_dict.update(Deleted)\n    submission=pd.DataFrame(submit_dict).T.reset_index()\n    submission.columns=[\"id\",\"matches\"]\n    submission.sort_values(\"id\").reset_index(drop=True)\n    return submission\n\nsubmission=submission_output(submit_dict,noise,Deleted)\nsubmission.to_csv(\"submission.csv\", index=False)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:07:13.126112Z","iopub.execute_input":"2022-07-07T13:07:13.126831Z","iopub.status.idle":"2022-07-07T13:07:13.152521Z","shell.execute_reply.started":"2022-07-07T13:07:13.126791Z","shell.execute_reply":"2022-07-07T13:07:13.151749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}