{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T17:07:38.897587Z","iopub.execute_input":"2022-07-12T17:07:38.898326Z","iopub.status.idle":"2022-07-12T17:07:38.926486Z","shell.execute_reply.started":"2022-07-12T17:07:38.898222Z","shell.execute_reply":"2022-07-12T17:07:38.925518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/contradictory-my-dear-watson/train.csv')\ndf_test = pd.read_csv('/kaggle/input/contradictory-my-dear-watson/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T17:07:38.983526Z","iopub.execute_input":"2022-07-12T17:07:38.984096Z","iopub.status.idle":"2022-07-12T17:07:39.373387Z","shell.execute_reply.started":"2022-07-12T17:07:38.984055Z","shell.execute_reply":"2022-07-12T17:07:39.372418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2022-07-12T17:07:39.375710Z","iopub.execute_input":"2022-07-12T17:07:39.376119Z","iopub.status.idle":"2022-07-12T17:07:39.398949Z","shell.execute_reply.started":"2022-07-12T17:07:39.376081Z","shell.execute_reply":"2022-07-12T17:07:39.398034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() \n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    \n    strategy = tf.distribute.experimental.TPUStrategy\nexcept ValueError:\n    strategy = tf.distribute.get_strategy() \n    print('Number of replicas:', strategy.num_replicas_in_sync) ","metadata":{"execution":{"iopub.status.busy":"2022-07-12T17:07:39.402211Z","iopub.execute_input":"2022-07-12T17:07:39.402481Z","iopub.status.idle":"2022-07-12T17:07:44.111881Z","shell.execute_reply.started":"2022-07-12T17:07:39.402458Z","shell.execute_reply":"2022-07-12T17:07:44.110867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() # TPU detection\nexcept ValueError:\n    tpu = None\n    gpus = tf.config.experimental.list_logical_devices(\"GPU\")\n    \nif tpu:\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu,) \n    print('Running on TPU ', tpu.cluster_spec().as_dict()['worker'])\nelif len(gpus) > 1:\n    strategy = tf.distribute.MirroredStrategy([gpu.name for gpu in gpus])\n    print('Running on multiple GPUs ', [gpu.name for gpu in gpus])\nelif len(gpus) == 1:\n    strategy = tf.distribute.get_strategy() \n    print('Running on single GPU ', gpus[0].name)\nelse:\n    strategy = tf.distribute.get_strategy() \n    print('Running on CPU')\nprint(\"Number of accelerators: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T17:07:44.114707Z","iopub.execute_input":"2022-07-12T17:07:44.115345Z","iopub.status.idle":"2022-07-12T17:07:46.500335Z","shell.execute_reply.started":"2022-07-12T17:07:44.115316Z","shell.execute_reply":"2022-07-12T17:07:46.499195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import TFAutoModel, AutoTokenizer","metadata":{"execution":{"iopub.status.busy":"2022-07-12T17:07:46.502031Z","iopub.execute_input":"2022-07-12T17:07:46.503153Z","iopub.status.idle":"2022-07-12T17:07:48.406924Z","shell.execute_reply.started":"2022-07-12T17:07:46.503111Z","shell.execute_reply":"2022-07-12T17:07:48.405885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def input_convert(data):\n        # -in- data - list of dict\n        # -out- inputs - dict of <key + list>\n        \n        inputs = {\n            'input_word_ids': [],\n            'input_mask': [],\n            'input_type_ids': []\n        }\n        \n        for i in data:\n            inputs['input_word_ids'].append(i['input_ids'])\n            inputs['input_mask'].append(i['attention_mask'])\n            inputs['input_type_ids'].append(i['token_type_ids'])\n            \n        inputs['input_word_ids'] = tf.ragged.constant(inputs['input_word_ids']).to_tensor()\n        inputs['input_mask'] = tf.ragged.constant(inputs['input_mask']).to_tensor()\n        inputs['input_type_ids'] = tf.ragged.constant(inputs['input_type_ids']).to_tensor()\n           \n        return inputs","metadata":{"execution":{"iopub.status.busy":"2022-07-12T17:07:48.408357Z","iopub.execute_input":"2022-07-12T17:07:48.408708Z","iopub.status.idle":"2022-07-12T17:07:48.417055Z","shell.execute_reply.started":"2022-07-12T17:07:48.408673Z","shell.execute_reply":"2022-07-12T17:07:48.416165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = df_train.pop('label')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T17:07:48.418638Z","iopub.execute_input":"2022-07-12T17:07:48.419290Z","iopub.status.idle":"2022-07-12T17:07:48.432125Z","shell.execute_reply.started":"2022-07-12T17:07:48.419242Z","shell.execute_reply":"2022-07-12T17:07:48.431056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T17:07:48.433737Z","iopub.execute_input":"2022-07-12T17:07:48.434258Z","iopub.status.idle":"2022-07-12T17:07:48.447333Z","shell.execute_reply.started":"2022-07-12T17:07:48.434219Z","shell.execute_reply":"2022-07-12T17:07:48.445929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmodel_name = 'bert-base-multilingual-uncased'\ndf = pd.concat([df_train, df_test], ignore_index = True)\n # tokenizing\ntokenizer = AutoTokenizer.from_pretrained(model_name)\nprint(type(tokenizer))\n\nmask = []\nfor i in range(len(df)):\n    padded_seq = tokenizer(df['premise'][i], df['hypothesis'][i], padding = True, \n                           add_special_tokens = True, return_token_type_ids = True)\n    mask.append(padded_seq)\n\ninputs = input_convert(mask)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T17:07:48.448478Z","iopub.execute_input":"2022-07-12T17:07:48.449037Z","iopub.status.idle":"2022-07-12T17:08:06.205323Z","shell.execute_reply.started":"2022-07-12T17:07:48.449001Z","shell.execute_reply":"2022-07-12T17:08:06.204257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ninputs_train = {}\ninputs_test = {}\n\nfor key in inputs.keys():\n    inputs_train[key] = inputs[key][:len(y), :]\n    inputs_test[key] = inputs[key][len(y):, :]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T17:08:06.208578Z","iopub.execute_input":"2022-07-12T17:08:06.208992Z","iopub.status.idle":"2022-07-12T17:08:06.246671Z","shell.execute_reply.started":"2022-07-12T17:08:06.208952Z","shell.execute_reply":"2022-07-12T17:08:06.245701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import Input, Model\nfrom tensorflow.keras.layers import Dense, Dropout, GlobalAveragePooling1D\nfrom tensorflow.keras.optimizers import Adam\n\nwith strategy.scope():\n    max_len = inputs['input_word_ids'].shape[1]\n    \n    encoder = TFAutoModel.from_pretrained(model_name)\n    \n    input_word_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\")\n    input_mask = Input(shape=(max_len,), dtype=tf.int32, name=\"input_mask\")\n    input_type_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_type_ids\")\n\n    embedding = encoder([input_word_ids, input_mask, input_type_ids])[0]\n    dense1 = Dense(256, activation='relu')(embedding[:,0,:])\n    dense2 = Dense(32, activation='relu')(dense1)\n    output = Dense(3, activation='softmax')(dense2)\n    model = Model(inputs=[input_word_ids, input_mask, input_type_ids], outputs = output)\n    model.compile(Adam(lr=1e-5), loss='sparse_categorical_crossentropy', metrics=['accuracy'], steps_per_execution = 100)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T17:08:06.248150Z","iopub.execute_input":"2022-07-12T17:08:06.248518Z","iopub.status.idle":"2022-07-12T17:08:46.324009Z","shell.execute_reply.started":"2022-07-12T17:08:06.248482Z","shell.execute_reply":"2022-07-12T17:08:46.323027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"early_stop = tf.keras.callbacks.EarlyStopping(patience = 3, restore_best_weights = True)\nmodel.fit(inputs_train, y.values, epochs = 2, verbose = 1, validation_split = 0.1,\n                    batch_size = 16 * strategy.num_replicas_in_sync, callbacks = [early_stop])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T17:38:33.531851Z","iopub.execute_input":"2022-07-12T17:38:33.532421Z","iopub.status.idle":"2022-07-12T17:50:55.526035Z","shell.execute_reply.started":"2022-07-12T17:38:33.532385Z","shell.execute_reply":"2022-07-12T17:50:55.525124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = [np.argmax(i) for i in model.predict(inputs_test)]\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T17:51:41.531704Z","iopub.execute_input":"2022-07-12T17:51:41.532065Z","iopub.status.idle":"2022-07-12T17:52:38.000820Z","shell.execute_reply.started":"2022-07-12T17:51:41.532036Z","shell.execute_reply":"2022-07-12T17:52:37.999663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/contradictory-my-dear-watson/sample_submission.csv')\nsubmission['prediction'] = predictions\nsubmission.to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T17:53:20.780717Z","iopub.execute_input":"2022-07-12T17:53:20.781073Z","iopub.status.idle":"2022-07-12T17:53:20.814546Z","shell.execute_reply.started":"2022-07-12T17:53:20.781043Z","shell.execute_reply":"2022-07-12T17:53:20.813655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}