{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport re\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport plotly.express as px\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nfrom tensorflow.keras.preprocessing.text import Tokenizer,one_hot\nfrom sklearn.metrics.pairwise import cosine_similarity,cosine_distances\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom tensorflow.keras import Sequential,Model,Input\nfrom tensorflow.keras.layers import LSTM,Concatenate,Dense,Lambda,Add,Embedding,Flatten,Dropout,GRU,Bidirectional,Dot\nfrom tensorflow.keras import regularizers\nfrom tensorflow.keras.callbacks import EarlyStopping\ntf.data.experimental.enable_debug_mode()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-05T04:36:18.890279Z","iopub.execute_input":"2022-08-05T04:36:18.890966Z","iopub.status.idle":"2022-08-05T04:36:27.149729Z","shell.execute_reply.started":"2022-08-05T04:36:18.890880Z","shell.execute_reply":"2022-08-05T04:36:27.148531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(\"../input/us-patent-phrase-to-phrase-matching/train.csv\")\ntest_data = pd.read_csv(\"../input/us-patent-phrase-to-phrase-matching/test.csv\")\nprint('TRAIN SIZE \\t: {}\\nTEST SIZE \\t: {}'.format(train_data.shape,test_data.shape))","metadata":{"execution":{"iopub.status.busy":"2022-08-05T04:36:27.151777Z","iopub.execute_input":"2022-08-05T04:36:27.152849Z","iopub.status.idle":"2022-08-05T04:36:27.264078Z","shell.execute_reply.started":"2022-08-05T04:36:27.152803Z","shell.execute_reply":"2022-08-05T04:36:27.262971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = train_data['score']\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T04:36:27.265504Z","iopub.execute_input":"2022-08-05T04:36:27.266072Z","iopub.status.idle":"2022-08-05T04:36:27.290038Z","shell.execute_reply.started":"2022-08-05T04:36:27.266034Z","shell.execute_reply":"2022-08-05T04:36:27.288805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.context.str[0].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T04:36:27.292302Z","iopub.execute_input":"2022-08-05T04:36:27.292769Z","iopub.status.idle":"2022-08-05T04:36:27.327153Z","shell.execute_reply.started":"2022-08-05T04:36:27.292733Z","shell.execute_reply":"2022-08-05T04:36:27.326146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = labels.value_counts().index.tolist()\ny = labels.value_counts().values.tolist()\nfig = px.bar(x = x,y = y,color = y,labels = {'x':'Labels','y':'Label count'},title = 'Labels Count ')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T04:36:55.125632Z","iopub.execute_input":"2022-08-05T04:36:55.125990Z","iopub.status.idle":"2022-08-05T04:36:56.056155Z","shell.execute_reply.started":"2022-08-05T04:36:55.125959Z","shell.execute_reply":"2022-08-05T04:36:56.055236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for idx in range(50):\n#     print(train_data['score'][idx],\"----->\",train_data['anchor'][idx],\"----->\",train_data['target'][idx],\"---->\",train_data['context_text'][idx])","metadata":{"execution":{"iopub.status.busy":"2022-08-05T04:36:57.115336Z","iopub.execute_input":"2022-08-05T04:36:57.116287Z","iopub.status.idle":"2022-08-05T04:36:57.120984Z","shell.execute_reply.started":"2022-08-05T04:36:57.116242Z","shell.execute_reply":"2022-08-05T04:36:57.120013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pearson_r(true,pred):\n    return np.corrcoef(true,pred)[0][1]\n\ndef get_score(y_true, y_pred):\n    score = scipy.stats.pearsonr(y_true, y_pred)[0]\n    return score\n\ndef euclideanDistance(layers):\n    dist = tf.sqrt(tf.reduce_sum(tf.square(layers[0] - layers[1]), 1))\n    return dist\n\n\nearly_stop = EarlyStopping(monitor = 'val_loss',\n                          min_delta = 0,\n                          patience = 3,\n                          verbose = 1,\n                          restore_best_weights = True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T04:36:58.767663Z","iopub.execute_input":"2022-08-05T04:36:58.768035Z","iopub.status.idle":"2022-08-05T04:36:58.774686Z","shell.execute_reply.started":"2022-08-05T04:36:58.768003Z","shell.execute_reply":"2022-08-05T04:36:58.773734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# bit of code stolen from Y.NAKAMA\ndef get_cpc_texts():\n    contexts = []\n    pattern = '[A-Z]\\d+'\n    for file_name in os.listdir('../input/cpc-data/CPCSchemeXML202105'):\n        result = re.findall(pattern, file_name)\n        if result:\n            contexts.append(result)\n    contexts = sorted(set(sum(contexts, [])))\n    results = {}\n    for cpc in ['A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'Y']:\n        with open(f'../input/cpc-data/CPCTitleList202202/cpc-section-{cpc}_20220201.txt') as f:\n            s = f.read()\n        pattern = f'{cpc}\\t\\t.+'\n        result = re.findall(pattern, s)\n        cpc_result = result[0].lstrip(pattern)\n        for context in [c for c in contexts if c[0] == cpc]:\n            pattern = f'{context}\\t\\t.+'\n            result = re.findall(pattern, s)\n            results[context] = cpc_result + \". \" + result[0].lstrip(pattern)\n    return results\n\ncpc_texts = get_cpc_texts()\ntrain_data['context_text'] = train_data['context'].map(cpc_texts)\ntest_data['context_text'] = test_data['context'].map(cpc_texts)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T04:37:01.088129Z","iopub.execute_input":"2022-08-05T04:37:01.089169Z","iopub.status.idle":"2022-08-05T04:37:02.256803Z","shell.execute_reply.started":"2022-08-05T04:37:01.089121Z","shell.execute_reply":"2022-08-05T04:37:02.255853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_text = list(set(\" \".join(train_data['anchor']).split())) + list(set(\" \".join(train_data['target']).split()))   #+ list(set(\" \".join(train_data['context_text']).split()))\nvocab_size = len(all_text) + 1\nmax_sentence_size = max([len(sen.split()) for sen in train_data['context_text']])\nembedding_size = 5\nlr = 0.001 #8e-5","metadata":{"execution":{"iopub.status.busy":"2022-08-05T04:37:02.343570Z","iopub.execute_input":"2022-08-05T04:37:02.343914Z","iopub.status.idle":"2022-08-05T04:37:02.394786Z","shell.execute_reply.started":"2022-08-05T04:37:02.343885Z","shell.execute_reply":"2022-08-05T04:37:02.393839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_sentence_size,lr,embedding_size,vocab_size","metadata":{"execution":{"iopub.status.busy":"2022-08-05T04:37:06.186388Z","iopub.execute_input":"2022-08-05T04:37:06.186756Z","iopub.status.idle":"2022-08-05T04:37:06.194246Z","shell.execute_reply.started":"2022-08-05T04:37:06.186727Z","shell.execute_reply":"2022-08-05T04:37:06.193025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = train_data[['anchor','target','context','context_text']]\ny = train_data[['score']]\n\nx_train,x_test,y_train,y_test = train_test_split(x,y,test_size = 0.2,shuffle = True)\nprint(x_train.shape,x_test.shape,y_train.shape,y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T04:37:08.538773Z","iopub.execute_input":"2022-08-05T04:37:08.539427Z","iopub.status.idle":"2022-08-05T04:37:08.559121Z","shell.execute_reply.started":"2022-08-05T04:37:08.539387Z","shell.execute_reply":"2022-08-05T04:37:08.557910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = Tokenizer(oov_token=\"<OOV>\")\ntokenizer.fit_on_texts(all_text)\n\ndef preprocessing(data,max_len):\n    anchor_sequences = tokenizer.texts_to_sequences(data['anchor'])\n    anchor_pad = pad_sequences(anchor_sequences, padding='post',maxlen = max_len)\n\n    context_sequences = tokenizer.texts_to_sequences(data['context_text'])\n    context_pad = pad_sequences(context_sequences, padding='post',maxlen = max_len)\n    \n    target_sequences = tokenizer.texts_to_sequences(data['target'])\n    target_pad = pad_sequences(target_sequences, padding='post',maxlen = max_len)\n\n    return anchor_pad,context_pad,target_pad\n\n# anchor_pad,context_pad,target_pad = preprocessing(train_data,max_len = max_sentence_size)\nx_train_anchor_pad,x_train_context_pad,x_train_target_pad = preprocessing(x_train,max_len = max_sentence_size)\nx_test_anchor_pad,x_test_context_pad,x_test_target_pad = preprocessing(x_test,max_len = max_sentence_size)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T04:37:10.321819Z","iopub.execute_input":"2022-08-05T04:37:10.322828Z","iopub.status.idle":"2022-08-05T04:37:11.635182Z","shell.execute_reply.started":"2022-08-05T04:37:10.322782Z","shell.execute_reply":"2022-08-05T04:37:11.634201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = y_train['score']\ny_test = y_test['score']","metadata":{"execution":{"iopub.status.busy":"2022-08-05T04:37:12.236605Z","iopub.execute_input":"2022-08-05T04:37:12.236955Z","iopub.status.idle":"2022-08-05T04:37:12.242995Z","shell.execute_reply.started":"2022-08-05T04:37:12.236927Z","shell.execute_reply":"2022-08-05T04:37:12.241501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CustomModel(Model):\n    def __init__(self):\n        super(CustomModel,self).__init__()\n\n        self.embedding = Embedding(input_dim=vocab_size,output_dim = embedding_size,input_length=max_sentence_size,name = 'embedding')\n        self.anchor_lstm = Bidirectional(LSTM(50,name = 'anchor_layer'))\n        self.context_lstm = Bidirectional(LSTM(50,name = 'context_layer'))\n        self.target_lstm = Bidirectional(LSTM(50,name = 'target_layer'))\n        self.add = Add(name = 'anchor + context layer')\n#         self.distance  = Lambda(cosine_similarity, name=\"output_layer\")\n        self.dropout = Dropout(0.5)\n        self.cosine = tf.keras.layers.Dot(axes=-1, normalize=True)\n#         self.cosine_sim  = tf.keras.losses.cosine_similarity()\n        \n    def call(self,inputs):\n        # Forward pass\n        anchor = self.embedding(inputs[0])\n        context = self.embedding(inputs[1])\n        target = self.embedding(inputs[2])\n        \n        anchor_lstm_out = self.anchor_lstm(anchor)  \n#         anchor_lstm_out = self.dropout(anchor_lstm_out)\n        \n        context_lstm_out = self.context_lstm(context)\n#         context_lstm_out = self.dropout(context_lstm_out)\n        \n        target_lstm_out = self.target_lstm(target)\n#         target_lstm_out = self.dropout(target_lstm_out)\n\n        \n        sum_layer = self.add([anchor_lstm_out,context_lstm_out])\n        output = self.cosine([target_lstm_out,sum_layer])\n\n        out = tf.reshape(output,shape = (output.shape[0],))\n        return (out + 1)/2","metadata":{"execution":{"iopub.status.busy":"2022-08-05T04:37:13.701340Z","iopub.execute_input":"2022-08-05T04:37:13.702150Z","iopub.status.idle":"2022-08-05T04:37:13.712478Z","shell.execute_reply.started":"2022-08-05T04:37:13.702115Z","shell.execute_reply":"2022-08-05T04:37:13.711265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def step_decay(epoch):\n    initial_lrate = 0.01\n    drop = 0.1\n    epochs_drop = 20\n    if epoch > 1:\n        if epoch % epochs_drop ==0:\n            lrate = initial_lrate * drop\n            return lrate\n        return initial_lrate\n\nlrate = tf.keras.callbacks.LearningRateScheduler(step_decay,verbose = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T04:38:14.367569Z","iopub.execute_input":"2022-08-05T04:38:14.368601Z","iopub.status.idle":"2022-08-05T04:38:14.375096Z","shell.execute_reply.started":"2022-08-05T04:38:14.368564Z","shell.execute_reply":"2022-08-05T04:38:14.373609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def scheduler(epoch,lr):\n    print(epoch,\"----------->\",lr)\n    if (epoch > 1) and (epoch % 3 == 0):\n        return lr*0.1\n    return lr\nlr_scheduler = tf.keras.callbacks.LearningRateScheduler(scheduler,verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T04:52:52.618035Z","iopub.execute_input":"2022-08-05T04:52:52.619026Z","iopub.status.idle":"2022-08-05T04:52:52.624918Z","shell.execute_reply.started":"2022-08-05T04:52:52.618988Z","shell.execute_reply":"2022-08-05T04:52:52.623877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inputs = [anchor_pad,context_pad,target_pad]\ntrain_inputs  = [x_train_anchor_pad,x_train_context_pad,x_train_target_pad]\ntest_inputs = [x_test_anchor_pad,x_test_context_pad,x_test_target_pad]\nadam = tf.keras.optimizers.Adam(learning_rate = 0.001)\nmodel = CustomModel()\nls = tf.keras.losses.CategoricalCrossentropy()\nmodel.compile(optimizer=adam,loss=ls,metrics = pearson_r,run_eagerly=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T04:53:01.701741Z","iopub.execute_input":"2022-08-05T04:53:01.702584Z","iopub.status.idle":"2022-08-05T04:53:01.745114Z","shell.execute_reply.started":"2022-08-05T04:53:01.702546Z","shell.execute_reply":"2022-08-05T04:53:01.744231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_anchor_pad,test_context_pad,test_target_pad = preprocessing(test_data,max_len = max_sentence_size)\ntesting = [test_anchor_pad,test_context_pad,test_target_pad]\nsubmission = pd.read_csv(\"/kaggle/input/us-patent-phrase-to-phrase-matching/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-05T04:52:59.598751Z","iopub.execute_input":"2022-08-05T04:52:59.599101Z","iopub.status.idle":"2022-08-05T04:52:59.614696Z","shell.execute_reply.started":"2022-08-05T04:52:59.599070Z","shell.execute_reply":"2022-08-05T04:52:59.613725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\ndef step_decay(epoch):\n    initial_lrate = 0.001\n    drop = 0.1\n    epochs_drop = 2.0\n    lrate = initial_lrate * 0.1\n    return lrate\nlr_scheduler = tf.keras.callbacks.LearningRateScheduler(step_decay,verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T18:16:56.981645Z","iopub.execute_input":"2022-08-04T18:16:56.982021Z","iopub.status.idle":"2022-08-04T18:16:56.988391Z","shell.execute_reply.started":"2022-08-04T18:16:56.981988Z","shell.execute_reply":"2022-08-04T18:16:56.986890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_folds = 4\nfrom sklearn.model_selection import StratifiedGroupKFold,KFold\ncv = StratifiedGroupKFold(n_splits=n_folds)\naccuracy = []\n\nkf = KFold(n_splits=n_folds, shuffle=True)\nfor trainIndices, testIndices in kf.split(train_inputs[0], np.array(y_train)):\n    x4 = test_inputs[0][0:testIndices[-1]]\n    x5 = test_inputs[1][0:testIndices[-1]]\n    x6 = test_inputs[2][0:testIndices[-1]]\n    \n#     print(trainIndices.shape)\n    x1 = train_inputs[0][trainIndices]\n    x2 = train_inputs[1][trainIndices]\n    x3 = train_inputs[2][trainIndices]\n\n    assert (len(x4) == len(x5) == len(x6))\n    assert (len(x1) == len(x2) == len(x3))\n    \n    inputs = [x1,x2,x3]\n    test_inputs = [x4,x5,x6]\n    \n    outs = np.array(y_train)[trainIndices]\n    test_outs = np.array(y_test)[0:testIndices[-1]]\n    history = model.fit(inputs, outs,\n                    batch_size=128,\n                    epochs= 50,\n                    verbose=1,\n                   validation_data=(test_inputs,test_outs),callbacks =[lr_scheduler,early_stop])\n\n    prediction = model.predict(testing,verbose = 1)\n    accuracy.append(prediction)\n    plt.plot(history.history['loss'])\n    plt.plot(history.history['val_loss'])\n    plt.title('model loss')\n    plt.ylabel('loss')\n    plt.xlabel('epoch')\n    plt.legend(['train', 'val'], loc='upper left')\n    plt.show()\n    \n    plt.plot(history.history['pearson_r'])\n    plt.plot(history.history['val_pearson_r'])\n    plt.title('model accuracy')\n    plt.ylabel('accuracy')\n    plt.xlabel('epoch')\n    plt.legend(['train_pearson_r', 'val_pearson_r'], loc='upper left')\n    plt.show()\n    \n    plt.plot(history.history['lr'])\n    plt.title('epochs vs learning rate')\n    plt.ylabel('learning rate')\n    plt.xlabel('epoch')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T04:53:05.255901Z","iopub.execute_input":"2022-08-05T04:53:05.256290Z","iopub.status.idle":"2022-08-05T04:57:32.636011Z","shell.execute_reply.started":"2022-08-05T04:53:05.256257Z","shell.execute_reply":"2022-08-05T04:57:32.635126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history.history['lr']","metadata":{"execution":{"iopub.status.busy":"2022-08-05T05:03:28.702404Z","iopub.execute_input":"2022-08-05T05:03:28.702754Z","iopub.status.idle":"2022-08-05T05:03:28.710547Z","shell.execute_reply.started":"2022-08-05T05:03:28.702725Z","shell.execute_reply":"2022-08-05T05:03:28.709542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-08-05T05:04:25.217080Z","iopub.execute_input":"2022-08-05T05:04:25.218031Z","iopub.status.idle":"2022-08-05T05:04:25.225159Z","shell.execute_reply.started":"2022-08-05T05:04:25.217991Z","shell.execute_reply":"2022-08-05T05:04:25.224200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = np.mean(accuracy,axis = 0)\npreds.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-05T05:05:16.721941Z","iopub.execute_input":"2022-08-05T05:05:16.722349Z","iopub.status.idle":"2022-08-05T05:05:16.729528Z","shell.execute_reply.started":"2022-08-05T05:05:16.722317Z","shell.execute_reply":"2022-08-05T05:05:16.728487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['score'] = preds\nsubmission.to_csv('submission.csv',index=False)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T05:05:20.215438Z","iopub.execute_input":"2022-08-05T05:05:20.216163Z","iopub.status.idle":"2022-08-05T05:05:20.233252Z","shell.execute_reply.started":"2022-08-05T05:05:20.216126Z","shell.execute_reply":"2022-08-05T05:05:20.232240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}