{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<center><h1><b style=\"text-align:center;\">U.S. Patent Phrase to Phrase Matching Using LSTM </b></h1></center>\n\n------------------------------------------------------------------------------------------------------","metadata":{}},{"cell_type":"code","source":"from IPython.display import Image\nImage(\"../input/patent/dataset-cover.jpg\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T21:02:29.638641Z","iopub.execute_input":"2022-08-03T21:02:29.639077Z","iopub.status.idle":"2022-08-03T21:02:29.697718Z","shell.execute_reply.started":"2022-08-03T21:02:29.638992Z","shell.execute_reply":"2022-08-03T21:02:29.696742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Can you build a model to match phrases in order to extract contextual information, thereby helping the patent community connect the dots between millions of patent documents?**","metadata":{}},{"cell_type":"markdown","source":"## **Evaluation**\n\n**Submissions are evaluated on the Pearson correlation coefficient between the predicted and actual similarity score**","metadata":{}},{"cell_type":"code","source":"Image(\"../input/patent/pearson-coefficient.png\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T21:02:38.855579Z","iopub.execute_input":"2022-08-03T21:02:38.855949Z","iopub.status.idle":"2022-08-03T21:02:38.872118Z","shell.execute_reply.started":"2022-08-03T21:02:38.855905Z","shell.execute_reply":"2022-08-03T21:02:38.871162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1> Columns </h1>\n\n> <b>id - a unique identifier for a pair of phrases</b>\n\n> <b>anchor - the first phrase</b>\n\n> <b>target - the second phrase</b>\n\n> <b>context - the CPC classification (version 2021.05), which indicates the subject within which the similarity is to be scored</b>\n\n> <b>score - the similarity. This is sourced from a combination of one or more manual expert ratings.</b>","metadata":{}},{"cell_type":"markdown","source":"# **Imports**","metadata":{}},{"cell_type":"code","source":"import os\nimport re\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport plotly.express as px\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nfrom tensorflow.keras.preprocessing.text import Tokenizer,one_hot\nfrom sklearn.metrics.pairwise import cosine_similarity,cosine_distances\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom tensorflow.keras import Sequential,Model,Input\nfrom tensorflow.keras.layers import LSTM,Concatenate,Dense,Lambda,Add,Embedding,Flatten,Dropout,GRU,Bidirectional,Dot\nfrom tensorflow.keras import regularizers\nfrom tensorflow.keras.callbacks import EarlyStopping\ntf.data.experimental.enable_debug_mode()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:54:05.766127Z","iopub.execute_input":"2022-08-04T03:54:05.767041Z","iopub.status.idle":"2022-08-04T03:54:05.775107Z","shell.execute_reply.started":"2022-08-04T03:54:05.766990Z","shell.execute_reply":"2022-08-04T03:54:05.773968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Get data**","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(\"../input/us-patent-phrase-to-phrase-matching/train.csv\")\ntest_data = pd.read_csv(\"../input/us-patent-phrase-to-phrase-matching/test.csv\")\nprint('TRAIN SIZE \\t: {}\\nTEST SIZE \\t: {}'.format(train_data.shape,test_data.shape))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:54:07.197173Z","iopub.execute_input":"2022-08-04T03:54:07.197521Z","iopub.status.idle":"2022-08-04T03:54:07.258609Z","shell.execute_reply.started":"2022-08-04T03:54:07.197491Z","shell.execute_reply":"2022-08-04T03:54:07.257594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = train_data['score']\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:54:07.450517Z","iopub.execute_input":"2022-08-04T03:54:07.451591Z","iopub.status.idle":"2022-08-04T03:54:07.465010Z","shell.execute_reply.started":"2022-08-04T03:54:07.451545Z","shell.execute_reply":"2022-08-04T03:54:07.464061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.context.str[0].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:54:07.679048Z","iopub.execute_input":"2022-08-04T03:54:07.679740Z","iopub.status.idle":"2022-08-04T03:54:07.709736Z","shell.execute_reply.started":"2022-08-04T03:54:07.679703Z","shell.execute_reply":"2022-08-04T03:54:07.708557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = labels.value_counts().index.tolist()\ny = labels.value_counts().values.tolist()\nfig = px.bar(x = x,y = y,color = y,labels = {'x':'Labels','y':'Label count'},title = 'Labels Count ')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:54:07.928664Z","iopub.execute_input":"2022-08-04T03:54:07.929466Z","iopub.status.idle":"2022-08-04T03:54:07.988420Z","shell.execute_reply.started":"2022-08-04T03:54:07.929427Z","shell.execute_reply":"2022-08-04T03:54:07.987045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pearson_r(true,pred):\n    return np.corrcoef(true,pred)[0][1]\n\ndef get_score(y_true, y_pred):\n    score = sp.stats.pearsonr(y_true, y_pred)[0]\n    return score\n\ndef euclideanDistance(layers):\n    dist = tf.sqrt(tf.reduce_sum(tf.square(layers[0] - layers[1]), 1))\n    return dist","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:54:08.220769Z","iopub.execute_input":"2022-08-04T03:54:08.221849Z","iopub.status.idle":"2022-08-04T03:54:08.228271Z","shell.execute_reply.started":"2022-08-04T03:54:08.221802Z","shell.execute_reply":"2022-08-04T03:54:08.227093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def schedule(epoch,lr):\n    if epoch % 3 == 0:\n        lr = lr - (lr*.05)\n        return lr\n    return lr\nlr_scheduler = tf.keras.callbacks.LearningRateScheduler(schedule,verbose=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:54:08.445745Z","iopub.execute_input":"2022-08-04T03:54:08.446943Z","iopub.status.idle":"2022-08-04T03:54:08.455871Z","shell.execute_reply.started":"2022-08-04T03:54:08.446894Z","shell.execute_reply":"2022-08-04T03:54:08.454176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# bit of code stolen from Y.NAKAMA\ndef get_cpc_texts():\n    contexts = []\n    pattern = '[A-Z]\\d+'\n    for file_name in os.listdir('../input/cpc-data/CPCSchemeXML202105'):\n        result = re.findall(pattern, file_name)\n        if result:\n            contexts.append(result)\n    contexts = sorted(set(sum(contexts, [])))\n    results = {}\n    for cpc in ['A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'Y']:\n        with open(f'../input/cpc-data/CPCTitleList202202/cpc-section-{cpc}_20220201.txt') as f:\n            s = f.read()\n        pattern = f'{cpc}\\t\\t.+'\n        result = re.findall(pattern, s)\n        cpc_result = result[0].lstrip(pattern)\n        for context in [c for c in contexts if c[0] == cpc]:\n            pattern = f'{context}\\t\\t.+'\n            result = re.findall(pattern, s)\n            results[context] = cpc_result + \". \" + result[0].lstrip(pattern)\n    return results\n\ncpc_texts = get_cpc_texts()\ntrain_data['context_text'] = train_data['context'].map(cpc_texts)\ntest_data['context_text'] = test_data['context'].map(cpc_texts)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:54:08.626984Z","iopub.execute_input":"2022-08-04T03:54:08.627680Z","iopub.status.idle":"2022-08-04T03:54:09.030846Z","shell.execute_reply.started":"2022-08-04T03:54:08.627643Z","shell.execute_reply":"2022-08-04T03:54:09.029912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_text = list(set(\" \".join(train_data['anchor']).split())) + list(set(\" \".join(train_data['target']).split()))   + list(set(\" \".join(train_data['context_text']).split()))\nvocab_size = len(all_text) + 1\nmax_sentence_size = max([len(sen.split()) for sen in train_data['context_text']])\nembedding_size = 5\nlr = 0.0001 #8e-5","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:54:09.032635Z","iopub.execute_input":"2022-08-04T03:54:09.032987Z","iopub.status.idle":"2022-08-04T03:54:09.123432Z","shell.execute_reply.started":"2022-08-04T03:54:09.032959Z","shell.execute_reply":"2022-08-04T03:54:09.122453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:54:14.972301Z","iopub.execute_input":"2022-08-04T03:54:14.973309Z","iopub.status.idle":"2022-08-04T03:54:14.981652Z","shell.execute_reply.started":"2022-08-04T03:54:14.973249Z","shell.execute_reply":"2022-08-04T03:54:14.980778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:54:15.404818Z","iopub.execute_input":"2022-08-04T03:54:15.405196Z","iopub.status.idle":"2022-08-04T03:54:15.420217Z","shell.execute_reply.started":"2022-08-04T03:54:15.405160Z","shell.execute_reply":"2022-08-04T03:54:15.419100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = train_data[['anchor','target','context','context_text']]\ny = train_data[['score']]\n\nx_train,x_test,y_train,y_test = train_test_split(x,y,test_size = 0.2,shuffle = True)\nprint(x_train.shape,x_test.shape,y_train.shape,y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:54:15.658820Z","iopub.execute_input":"2022-08-04T03:54:15.659501Z","iopub.status.idle":"2022-08-04T03:54:15.679819Z","shell.execute_reply.started":"2022-08-04T03:54:15.659464Z","shell.execute_reply":"2022-08-04T03:54:15.678804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Preprocessing**","metadata":{}},{"cell_type":"code","source":"tokenizer = Tokenizer(oov_token=\"<OOV>\")","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:54:16.590379Z","iopub.execute_input":"2022-08-04T03:54:16.591379Z","iopub.status.idle":"2022-08-04T03:54:16.597361Z","shell.execute_reply.started":"2022-08-04T03:54:16.591329Z","shell.execute_reply":"2022-08-04T03:54:16.596050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer.fit_on_texts(all_text)\n\ndef preprocessing(data,max_len):\n    anchor_sequences = tokenizer.texts_to_sequences(data['anchor'])\n    anchor_pad = pad_sequences(anchor_sequences, padding='post',maxlen = max_len)\n\n    context_sequences = tokenizer.texts_to_sequences(data['context_text'])\n    context_pad = pad_sequences(context_sequences, padding='post',maxlen = max_len)\n    \n    target_sequences = tokenizer.texts_to_sequences(data['target'])\n    target_pad = pad_sequences(target_sequences, padding='post',maxlen = max_len)\n\n    return anchor_pad,context_pad,target_pad\n\n# anchor_pad,context_pad,target_pad = preprocessing(train_data,max_len = max_sentence_size)\nx_train_anchor_pad,x_train_context_pad,x_train_target_pad = preprocessing(x_train,max_len = max_sentence_size)\nx_test_anchor_pad,x_test_context_pad,x_test_target_pad = preprocessing(x_test,max_len = max_sentence_size)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:54:17.115933Z","iopub.execute_input":"2022-08-04T03:54:17.116299Z","iopub.status.idle":"2022-08-04T03:54:18.829484Z","shell.execute_reply.started":"2022-08-04T03:54:17.116267Z","shell.execute_reply":"2022-08-04T03:54:18.828464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = y_train['score']\ny_test = y_test['score']","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:54:18.831205Z","iopub.execute_input":"2022-08-04T03:54:18.831667Z","iopub.status.idle":"2022-08-04T03:54:18.837271Z","shell.execute_reply.started":"2022-08-04T03:54:18.831628Z","shell.execute_reply":"2022-08-04T03:54:18.836141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Modelling**","metadata":{}},{"cell_type":"code","source":"early_stop = EarlyStopping(monitor = 'val_loss',\n                          min_delta = 0,\n                          patience = 3,\n                          verbose = 1,\n                          restore_best_weights = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:54:19.287215Z","iopub.execute_input":"2022-08-04T03:54:19.287583Z","iopub.status.idle":"2022-08-04T03:54:19.292711Z","shell.execute_reply.started":"2022-08-04T03:54:19.287552Z","shell.execute_reply":"2022-08-04T03:54:19.291587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CustomModel(Model):\n    def __init__(self):\n        super(CustomModel,self).__init__()\n\n        self.embedding = Embedding(input_dim=vocab_size,output_dim = embedding_size,input_length=max_sentence_size,name = 'embedding')\n        self.anchor_lstm = Bidirectional(LSTM(50,name = 'anchor_layer'))\n        self.context_lstm = Bidirectional(LSTM(50,name = 'context_layer'))\n        self.target_lstm = Bidirectional(LSTM(50,name = 'target_layer'))\n        self.add = Add(name = 'anchor + context layer')\n#         self.distance  = Lambda(cosine_similarity, name=\"output_layer\")\n        self.dropout = Dropout(0.15)\n        self.dense = Dense(50,activation = 'softmax')\n        self.cosine = tf.keras.layers.Dot(axes=-1, normalize=True)\n#         self.cosine_sim  = tf.keras.losses.cosine_similarity()\n        \n    def call(self,inputs):\n        # Forward pass\n        anchor = self.embedding(inputs[0])\n        context = self.embedding(inputs[1])\n        target = self.embedding(inputs[2])\n        \n        anchor_lstm_out = self.anchor_lstm(anchor)       \n        context_lstm_out = self.context_lstm(context)\n        target_lstm_out = self.target_lstm(target)\n        \n        sum_layer = self.add([anchor_lstm_out,context_lstm_out])\n        output = self.cosine([target_lstm_out,sum_layer])\n\n        out = tf.reshape(output,shape = (output.shape[0],))\n        return (out + 1)/2","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:54:20.775331Z","iopub.execute_input":"2022-08-04T03:54:20.776467Z","iopub.status.idle":"2022-08-04T03:54:20.787029Z","shell.execute_reply.started":"2022-08-04T03:54:20.776415Z","shell.execute_reply":"2022-08-04T03:54:20.786085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Training**","metadata":{}},{"cell_type":"code","source":"# inputs = [anchor_pad,context_pad,target_pad]\ntrain_inputs  = [x_train_anchor_pad,x_train_context_pad,x_train_target_pad]\ntest_inputs = [x_test_anchor_pad,x_test_context_pad,x_test_target_pad]\nadam = tf.keras.optimizers.Adam(learning_rate = 8e-05)\nmodel = CustomModel()\nls = tf.keras.losses.CategoricalCrossentropy()\nmodel.compile(optimizer='adam',loss=ls,metrics = pearson_r,run_eagerly=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:57:31.334399Z","iopub.execute_input":"2022-08-04T03:57:31.335377Z","iopub.status.idle":"2022-08-04T03:57:31.376771Z","shell.execute_reply.started":"2022-08-04T03:57:31.335338Z","shell.execute_reply":"2022-08-04T03:57:31.375838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# history =  model.fit(x=train_inputs,y =np.array(y_train),validation_data=(test_inputs,y_test), epochs=20,verbose = 1,batch_size  =256,callbacks = [lr_scheduler,early_stop])","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:59:18.358466Z","iopub.execute_input":"2022-08-04T03:59:18.358841Z","iopub.status.idle":"2022-08-04T03:59:18.364672Z","shell.execute_reply.started":"2022-08-04T03:59:18.358809Z","shell.execute_reply":"2022-08-04T03:59:18.363186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plt.plot(history.history['loss'])\n# plt.plot(history.history['val_loss'])\n# plt.title('model loss')\n# plt.ylabel('loss')\n# plt.xlabel('epoch')\n# plt.legend(['train', 'val'], loc='upper left')\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T19:44:30.785046Z","iopub.status.idle":"2022-08-03T19:44:30.786448Z","shell.execute_reply.started":"2022-08-03T19:44:30.786140Z","shell.execute_reply":"2022-08-03T19:44:30.786175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n# plt.plot(history.history['pearson_r'])\n# plt.plot(history.history['val_pearson_r'])\n# plt.title('model accuracy')\n# plt.ylabel('accuracy')\n# plt.xlabel('epoch')\n# plt.legend(['train_pearson_r', 'val_pearson_r'], loc='upper left')\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T19:44:30.789074Z","iopub.status.idle":"2022-08-03T19:44:30.790213Z","shell.execute_reply.started":"2022-08-03T19:44:30.789938Z","shell.execute_reply":"2022-08-03T19:44:30.789963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_anchor_pad,test_context_pad,test_target_pad = preprocessing(test_data,max_len = max_sentence_size)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:57:53.765307Z","iopub.execute_input":"2022-08-04T03:57:53.766318Z","iopub.status.idle":"2022-08-04T03:57:53.774661Z","shell.execute_reply.started":"2022-08-04T03:57:53.766267Z","shell.execute_reply":"2022-08-04T03:57:53.773733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testing = [test_anchor_pad,test_context_pad,test_target_pad]\n# preds = model.predict(testing,verbose = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:59:25.164609Z","iopub.execute_input":"2022-08-04T03:59:25.165569Z","iopub.status.idle":"2022-08-04T03:59:25.169907Z","shell.execute_reply.started":"2022-08-04T03:59:25.165532Z","shell.execute_reply":"2022-08-04T03:59:25.168851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Submission**","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv(\"/kaggle/input/us-patent-phrase-to-phrase-matching/sample_submission.csv\")\n# submission['score'] = preds\n# submission.to_csv('submission.csv',index=False)\n# submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:59:26.774848Z","iopub.execute_input":"2022-08-04T03:59:26.775766Z","iopub.status.idle":"2022-08-04T03:59:26.788261Z","shell.execute_reply.started":"2022-08-04T03:59:26.775729Z","shell.execute_reply":"2022-08-04T03:59:26.786941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"K-Fold","metadata":{}},{"cell_type":"code","source":"n_folds = 4\nfrom sklearn.model_selection import StratifiedGroupKFold,KFold\ncv = StratifiedGroupKFold(n_splits=n_folds)\naccuracy = []","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:59:28.506782Z","iopub.execute_input":"2022-08-04T03:59:28.507160Z","iopub.status.idle":"2022-08-04T03:59:28.512316Z","shell.execute_reply.started":"2022-08-04T03:59:28.507106Z","shell.execute_reply":"2022-08-04T03:59:28.511172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kf = KFold(n_splits=n_folds, shuffle=True, random_state=1)\nfor trainIndices, testIndices in kf.split(train_inputs[0], np.array(y_train)):\n    print(testIndices.shape)\n    x4 = test_inputs[0][0:testIndices[-1]]\n    x5 = test_inputs[1][0:testIndices[-1]]\n    x6 = test_inputs[2][0:testIndices[-1]]\n    \n#     print(trainIndices.shape)\n    x1 = train_inputs[0][trainIndices]\n    x2 = train_inputs[1][trainIndices]\n    x3 = train_inputs[2][trainIndices]\n\n    assert (len(x4) == len(x5) == len(x6))\n    assert (len(x1) == len(x2) == len(x3))\n    \n    inputs = [x1,x2,x3]\n    test_inputs = [x4,x5,x6]\n    \n    outs = np.array(y_train)[trainIndices]\n    test_outs = np.array(y_test)[0:testIndices[-1]]\n    history = model.fit(inputs, outs,\n                    batch_size=128,\n                    epochs= 10,\n                    verbose=1,\n                   validation_data=(test_inputs,test_outs),callbacks =[lr_scheduler,early_stop])\n\n    prediction = model.predict(testing,verbose = 1)\n    accuracy.append(prediction)\n#     accuracy.append(accuracy_score(y[trainIndices], prediction))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T04:00:02.676952Z","iopub.execute_input":"2022-08-04T04:00:02.677636Z","iopub.status.idle":"2022-08-04T04:03:49.029683Z","shell.execute_reply.started":"2022-08-04T04:00:02.677598Z","shell.execute_reply":"2022-08-04T04:03:49.028735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preds = list()\n# for i in list(zip(accuracy[0],accuracy[1],accuracy[2])):\n#     preds.append(sum(i)/3)\npreds =np.mean(accuracy,axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T04:04:35.140024Z","iopub.execute_input":"2022-08-04T04:04:35.140782Z","iopub.status.idle":"2022-08-04T04:04:35.145896Z","shell.execute_reply.started":"2022-08-04T04:04:35.140742Z","shell.execute_reply":"2022-08-04T04:04:35.144784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-04T04:04:35.895021Z","iopub.execute_input":"2022-08-04T04:04:35.895638Z","iopub.status.idle":"2022-08-04T04:04:35.902089Z","shell.execute_reply.started":"2022-08-04T04:04:35.895603Z","shell.execute_reply":"2022-08-04T04:04:35.901157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['score'] = preds\nsubmission.to_csv('submission.csv',index=False)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T04:04:36.828273Z","iopub.execute_input":"2022-08-04T04:04:36.829461Z","iopub.status.idle":"2022-08-04T04:04:36.851096Z","shell.execute_reply.started":"2022-08-04T04:04:36.829420Z","shell.execute_reply":"2022-08-04T04:04:36.850247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}