{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-19T11:27:06.330614Z","iopub.execute_input":"2022-07-19T11:27:06.331184Z","iopub.status.idle":"2022-07-19T11:27:06.363215Z","shell.execute_reply.started":"2022-07-19T11:27:06.331083Z","shell.execute_reply":"2022-07-19T11:27:06.362226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pycodestyle\n!pip install --index-url https://test.pypi.org/simple/ nbpep8","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:17:59.103752Z","iopub.execute_input":"2022-07-15T15:17:59.104090Z","iopub.status.idle":"2022-07-15T15:18:23.525053Z","shell.execute_reply.started":"2022-07-15T15:17:59.104056Z","shell.execute_reply":"2022-07-15T15:18:23.523870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nbpep8.nbpep8 import pep8","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:18:23.526969Z","iopub.execute_input":"2022-07-15T15:18:23.527355Z","iopub.status.idle":"2022-07-15T15:18:23.537369Z","shell.execute_reply.started":"2022-07-15T15:18:23.527319Z","shell.execute_reply":"2022-07-15T15:18:23.536272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\n#from bs4 import BeautifulSoup\n#import re\nimport nltk\nimport os\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom tensorflow.keras import callbacks, models, layers\nimport matplotlib.pyplot as plt\n\n# tokenization\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:27:45.063328Z","iopub.execute_input":"2022-07-19T11:27:45.063784Z","iopub.status.idle":"2022-07-19T11:27:54.032289Z","shell.execute_reply.started":"2022-07-19T11:27:45.063746Z","shell.execute_reply":"2022-07-19T11:27:54.031024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE = '/kaggle/input/word2vec-nlp-tutorial'\nMAX_WORDS = 25_000\ntrain = pd.read_csv(os.path.join(BASE,'labeledTrainData.tsv.zip'),\n                    header=0,\n                    delimiter=\"\\t\",\n                    quoting=3)\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:28:03.671397Z","iopub.execute_input":"2022-07-19T11:28:03.674613Z","iopub.status.idle":"2022-07-19T11:28:04.623291Z","shell.execute_reply.started":"2022-07-19T11:28:03.674566Z","shell.execute_reply":"2022-07-19T11:28:04.622327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(os.path.join(BASE,'testData.tsv.zip'),\n                   header=0,\n                   delimiter=\"\\t\",\n                   quoting=3)\ntest.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:28:06.686097Z","iopub.execute_input":"2022-07-19T11:28:06.686578Z","iopub.status.idle":"2022-07-19T11:28:07.957958Z","shell.execute_reply.started":"2022-07-19T11:28:06.686539Z","shell.execute_reply":"2022-07-19T11:28:07.956715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x = 'sentiment', data = train, palette = 'Set3')\nplt.xticks(ticks = [0,1], labels = ['Negative','Positive'])\nplt.ylabel(\"Count\")\nplt.xlabel(\"Target\")\nplt.title(\"Distribution of sentiment label\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:18:30.960665Z","iopub.execute_input":"2022-07-15T15:18:30.961152Z","iopub.status.idle":"2022-07-15T15:18:31.156895Z","shell.execute_reply.started":"2022-07-15T15:18:30.961111Z","shell.execute_reply":"2022-07-15T15:18:31.156001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import nltk\nnltk.download('punkt')\nfrom nltk.tokenize import sent_tokenize, word_tokenize\n\ndef tokenizer_fct(sentence) :\n    # print(sentence)\n    sentence_clean = sentence.replace('-', ' ').replace('+', ' ').replace('/', ' ').replace('#', ' ')\n    word_tokens = word_tokenize(sentence_clean)\n    return word_tokens\n\ndef split_tokenizer_fct(sentence) :\n    # print(sentence)\n    sentence_clean = sentence.replace('-', ' ').replace('+', ' ').replace('/', ' ').replace('#', ' ')\n    word_tokens = sentence_clean.split(' ')\n    return word_tokens\n\n\n\n# lower case et alpha\ndef lower_start_fct(list_words) :\n    lw = [w.lower() for w in list_words if (not w.startswith(\"@\")) \n    #                                   and (not w.startswith(\"#\"))\n                                       and (not w.startswith(\"http\"))]\n    return lw\n\n# Fonction de préparation du texte pour le Deep learning (USE et BERT)\ndef transform_dl_fct(desc_text) :\n    word_tokens = split_tokenizer_fct(desc_text)\n#    sw = stop_word_filter_fct(word_tokens)\n    lw = lower_start_fct(word_tokens)\n    # lem_w = lemma_fct(lw)    \n    transf_desc_text = ' '.join(lw)\n    return transf_desc_text","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:28:15.409441Z","iopub.execute_input":"2022-07-19T11:28:15.409793Z","iopub.status.idle":"2022-07-19T11:28:16.118193Z","shell.execute_reply.started":"2022-07-19T11:28:15.409763Z","shell.execute_reply":"2022-07-19T11:28:16.117222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['sentence_dl'] = train['review'].apply(lambda x : transform_dl_fct(x))\ntrain[['review','sentence_dl']]","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:28:21.749102Z","iopub.execute_input":"2022-07-19T11:28:21.749749Z","iopub.status.idle":"2022-07-19T11:28:24.231915Z","shell.execute_reply.started":"2022-07-19T11:28:21.749713Z","shell.execute_reply":"2022-07-19T11:28:24.230089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!pip install transformers\n#!pip install pytorch-transformers\n# Import des librairies\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nfrom sklearn import metrics\nfrom sklearn.model_selection import train_test_split\n\nimport tensorflow as tf\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom transformers import *\nimport time\n# tf.compat.v1.disable_eager_execution()\npath = \"/content/\"\nprint(tf.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:28:28.969838Z","iopub.execute_input":"2022-07-19T11:28:28.970410Z","iopub.status.idle":"2022-07-19T11:28:43.329091Z","shell.execute_reply.started":"2022-07-19T11:28:28.970373Z","shell.execute_reply":"2022-07-19T11:28:43.327298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweets_raw = train['sentence_dl']\ny = train['sentiment'].values","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:28:46.648329Z","iopub.execute_input":"2022-07-19T11:28:46.649005Z","iopub.status.idle":"2022-07-19T11:28:46.655733Z","shell.execute_reply.started":"2022-07-19T11:28:46.648971Z","shell.execute_reply":"2022-07-19T11:28:46.654683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet_raw_train, tweet_raw_test, y_train, y_test = train_test_split(\n    tweets_raw.values, y,\n    test_size=0.2,\n    stratify=y,\n    random_state=7\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:28:50.366503Z","iopub.execute_input":"2022-07-19T11:28:50.366950Z","iopub.status.idle":"2022-07-19T11:28:50.397867Z","shell.execute_reply.started":"2022-07-19T11:28:50.366900Z","shell.execute_reply":"2022-07-19T11:28:50.396857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_label = y_train\ntest_label  = y_test\ntrain_label","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:28:54.720560Z","iopub.execute_input":"2022-07-19T11:28:54.720900Z","iopub.status.idle":"2022-07-19T11:28:54.727943Z","shell.execute_reply.started":"2022-07-19T11:28:54.720870Z","shell.execute_reply":"2022-07-19T11:28:54.727086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preparing the sentences\ndef data_prep_fct(bert_tokenizer, sentences, max_length) :\n    \n    input_ids=[]\n    attention_masks=[]\n    token_type_ids=[]\n    segment_ids=[]\n\n    for sent in sentences:\n        bert_inp = bert_tokenizer.encode_plus(sent,\n                                              add_special_tokens = True,\n                                              max_length = max_length,\n                                              padding='max_length',\n                                              truncation=True,\n                                              return_attention_mask = True, \n                                              return_token_type_ids=True)\n        input_ids.append(bert_inp['input_ids'])\n        attention_masks.append(bert_inp['attention_mask'])\n        token_type_ids.append(bert_inp['token_type_ids'])\n        segment_id = [0] * max_length\n        segment_ids.append(segment_id)\n\n    input_ids = np.asarray(input_ids)\n    attention_masks = np.array(attention_masks)\n    token_type_ids = np.array(token_type_ids)\n    segment_ids = np.array(segment_ids)\n    \n#     return input_ids, attention_masks\n    return input_ids, attention_masks, token_type_ids, segment_ids","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:28:57.153248Z","iopub.execute_input":"2022-07-19T11:28:57.154280Z","iopub.status.idle":"2022-07-19T11:28:57.163871Z","shell.execute_reply.started":"2022-07-19T11:28:57.154227Z","shell.execute_reply":"2022-07-19T11:28:57.162797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_test_prep_fct(bert_tokenizer) :\n\n    print(\"Train tweets preparation ...\")\n    sentences = tweet_raw_train\n    start = time.time()\n    train_inp, train_mask, train_token, train_seg = data_prep_fct(\n        bert_tokenizer,\n        sentences,\n        max_length=max_length\n    )\n    print(\"duration: \", time.time()-start)\n    print()\n\n    print(\"Test tweets preparation ...\")\n    sentences = tweet_raw_test\n    start = time.time()\n    test_inp, test_mask, test_token, test_seg = data_prep_fct(\n        bert_tokenizer,\n        sentences,\n        max_length=max_length\n    )\n    print(\"duration: \", time.time()-start)\n    \n    return train_inp, train_mask,train_token,train_seg, test_inp, test_mask,test_token, test_seg","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:29:02.429455Z","iopub.execute_input":"2022-07-19T11:29:02.429886Z","iopub.status.idle":"2022-07-19T11:29:02.442812Z","shell.execute_reply.started":"2022-07-19T11:29:02.429850Z","shell.execute_reply":"2022-07-19T11:29:02.440542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_name = 'bert'\n\nmodel_type   = 'fabriceyhc/bert-base-uncased-imdb'\n\nmax_length   = 512\nbert_tokenizer = BertTokenizer.from_pretrained(model_type)\n#AutoTokenizer.from_pretrained(model_type)\n\nmt = (model_type.split('-')[0][0] + model_type.split('-')[1][0] + model_type.split('-')[2][0]).upper()\nml = 'ML' + str(max_length)\ntw = 'T' + str(len(tweets_raw))\nmodel_file_name = model_name + '_' + mt + '_' + ml + '_' + tw\nmodel_save_path = model_file_name + '.h5'\n# = path + 'models/' + model_file_name + '.h5'\nprint(model_save_path)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:18:46.760567Z","iopub.execute_input":"2022-07-15T15:18:46.761668Z","iopub.status.idle":"2022-07-15T15:18:55.075366Z","shell.execute_reply.started":"2022-07-15T15:18:46.761629Z","shell.execute_reply":"2022-07-15T15:18:55.074301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_inp, train_mask, train_token, train_seg, test_inp, test_mask, test_token, test_seg = \\\n                                                                train_test_prep_fct(bert_tokenizer)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:18:55.076940Z","iopub.execute_input":"2022-07-15T15:18:55.077388Z","iopub.status.idle":"2022-07-15T15:22:38.175494Z","shell.execute_reply.started":"2022-07-15T15:18:55.077351Z","shell.execute_reply":"2022-07-15T15:22:38.174359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bert_model = TFBertForSequenceClassification.from_pretrained(model_type, num_labels=2, from_pt=True)\nloss = tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True)\noptimizer = tf.keras.optimizers.Adam(learning_rate=5e-5, epsilon=1e-08)\nmetric = tf.keras.metrics.SparseCategoricalAccuracy('accuracy')\n\n\nbert_model.compile(loss=loss, optimizer=optimizer, metrics=[metric])\nprint(bert_model.summary())","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:22:38.180267Z","iopub.execute_input":"2022-07-15T15:22:38.182677Z","iopub.status.idle":"2022-07-15T15:23:13.366006Z","shell.execute_reply.started":"2022-07-15T15:22:38.182635Z","shell.execute_reply":"2022-07-15T15:23:13.364994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checkpoint_path = \"bert_FBU_ML512_T25000.{epoch:02d}-{val_loss:.2f}.h5\"\nes = tf.keras.callbacks.EarlyStopping(monitor='val_loss', mode='min', verbose=1, patience=5)\nmc = tf.keras.callbacks.ModelCheckpoint(filepath=model_save_path,\n                                                    save_weights_only=True,\n                                                    monitor='val_loss',mode='min',\n                                                    save_best_only=True, verbose=1)\ncallbacks = [es, mc]\n\nhistory = bert_model.fit(\n                         [train_inp, train_mask, train_seg], train_label,                         \n                         batch_size=4, epochs=1 ,\n                         validation_data=([test_inp, test_mask, test_seg],test_label),\n                         callbacks=callbacks, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:23:13.367557Z","iopub.execute_input":"2022-07-15T15:23:13.367909Z","iopub.status.idle":"2022-07-15T15:50:02.712362Z","shell.execute_reply.started":"2022-07-15T15:23:13.367874Z","shell.execute_reply":"2022-07-15T15:50:02.711281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trained_model = bert_model\ntrained_model.save_weights(model_save_path)\ntrained_model.load_weights(model_save_path)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:50:02.714172Z","iopub.execute_input":"2022-07-15T15:50:02.714575Z","iopub.status.idle":"2022-07-15T15:50:04.270606Z","shell.execute_reply.started":"2022-07-15T15:50:02.714538Z","shell.execute_reply":"2022-07-15T15:50:04.269529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_proba = trained_model.predict([test_inp, test_mask, test_seg], batch_size=4)[0][:,1]\n# print(y_pred_proba)\ny_pred = np.where(y_pred_proba>0,1,0)\nprint(\"accuracy : \", metrics.accuracy_score(y_test,y_pred))\nprint(\"auc      : \", metrics.roc_auc_score(y_test,y_pred_proba))\nprint()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:50:04.272242Z","iopub.execute_input":"2022-07-15T15:50:04.272601Z","iopub.status.idle":"2022-07-15T15:51:58.319218Z","shell.execute_reply.started":"2022-07-15T15:50:04.272564Z","shell.execute_reply":"2022-07-15T15:51:58.318284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# RoBERTa\nmodel_name = 'roberta'\nmodel_type = 'aychang/roberta-base-imdb'\nmax_length   = 512\nRoberta_tokenizer = AutoTokenizer.from_pretrained(model_type)\nmt = (model_type.split('-')[0][0] + model_type.split('-')[1][0] + model_type.split('-')[2][0]).upper()\nml = 'ML' + str(max_length)\ntw = 'T' + str(len(tweets_raw))\nmodel_file_name = model_name + '_' + mt + '_' + ml + '_' + tw\nmodel_save_path = model_file_name + '.h5'\nprint(model_save_path)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:29:11.072269Z","iopub.execute_input":"2022-07-19T11:29:11.072665Z","iopub.status.idle":"2022-07-19T11:29:19.807010Z","shell.execute_reply.started":"2022-07-19T11:29:11.072633Z","shell.execute_reply":"2022-07-19T11:29:19.805112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_inp, train_mask, train_token, train_seg, test_inp, test_mask, test_token, test_seg = \\\n                                                                train_test_prep_fct(Roberta_tokenizer)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:29:25.590361Z","iopub.execute_input":"2022-07-19T11:29:25.590961Z","iopub.status.idle":"2022-07-19T11:30:03.315036Z","shell.execute_reply.started":"2022-07-19T11:29:25.590928Z","shell.execute_reply":"2022-07-19T11:30:03.313834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Roberta_model = TFRobertaForSequenceClassification.from_pretrained(\n    model_type,\n    num_labels=2,\n    from_pt=True\n)\nloss = tf.keras.losses.SparseCategoricalCrossentropy(\n    from_logits=True\n)\noptimizer = tf.keras.optimizers.Adam(\n    learning_rate=2e-5,\n    epsilon=1e-08\n)\nmetric = tf.keras.metrics.SparseCategoricalAccuracy('accuracy')\nRoberta_model.compile(loss=loss,\n                      optimizer=optimizer,\n                      metrics=[metric])\nprint(Roberta_model.summary())","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:30:07.117925Z","iopub.execute_input":"2022-07-19T11:30:07.118893Z","iopub.status.idle":"2022-07-19T11:30:55.197676Z","shell.execute_reply.started":"2022-07-19T11:30:07.118856Z","shell.execute_reply":"2022-07-19T11:30:55.196659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checkpoint_path = \"roberta_WBF_ML512_T25000.{epoch:02d}-{val_loss:.2f}.h5\"\nes = tf.keras.callbacks.EarlyStopping(monitor='val_loss',\n                                      mode='min',\n                                      verbose=1,\n                                      patience=5)\nmc = tf.keras.callbacks.ModelCheckpoint(filepath=model_save_path,\n                                        save_weights_only=True,\n                                        monitor='val_loss',mode='min',\n                                        save_best_only=True, verbose=1)\ncallbacks = [es, mc]\n\nhistory = Roberta_model.fit(\n                         [train_inp, train_mask, train_seg], train_label,                         \n                         batch_size=4, epochs=1,\n                         validation_data=([test_inp, test_mask, test_seg],test_label),\n                         callbacks=callbacks, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:31:08.332564Z","iopub.execute_input":"2022-07-19T11:31:08.332927Z","iopub.status.idle":"2022-07-19T11:58:43.877222Z","shell.execute_reply.started":"2022-07-19T11:31:08.332896Z","shell.execute_reply":"2022-07-19T11:58:43.876167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trained_model = Roberta_model\ntrained_model.save_weights(model_save_path)\ntrained_model.load_weights(model_save_path)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:58:43.879957Z","iopub.execute_input":"2022-07-19T11:58:43.880347Z","iopub.status.idle":"2022-07-19T11:58:45.850825Z","shell.execute_reply.started":"2022-07-19T11:58:43.880313Z","shell.execute_reply":"2022-07-19T11:58:45.849865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_proba = trained_model.predict([test_inp, test_mask, test_seg],\n                                     batch_size=4)[0][:,1]\ny_pred = np.where(y_pred_proba>0,1,0)\nprint(\"Accuracy : \", metrics.accuracy_score(y_test,y_pred))\nprint(\"AUC      : \", metrics.roc_auc_score(y_test,y_pred_proba))\nprint()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:59:23.797288Z","iopub.execute_input":"2022-07-19T11:59:23.798343Z","iopub.status.idle":"2022-07-19T12:01:17.512338Z","shell.execute_reply.started":"2022-07-19T11:59:23.798296Z","shell.execute_reply":"2022-07-19T12:01:17.510211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_sentences = test['review']\npred_sentences","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:01:24.153133Z","iopub.execute_input":"2022-07-19T12:01:24.153467Z","iopub.status.idle":"2022-07-19T12:01:24.164156Z","shell.execute_reply.started":"2022-07-19T12:01:24.153438Z","shell.execute_reply":"2022-07-19T12:01:24.163030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_input = data_prep_fct(Roberta_tokenizer, pred_sentences, max_length)\n\ntf_output = trained_model.predict(predict_input)[0]\ntf_prediction = tf.nn.softmax(tf_output, axis=1)\n\nlabels = ['Negative','Positive'] #(0:negative, 1:positive)\nlabel = tf.argmax(tf_prediction, axis=1)\nlabel = label.numpy()\nprint(labels[label[0]])","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:01:39.094321Z","iopub.execute_input":"2022-07-19T12:01:39.094670Z","iopub.status.idle":"2022-07-19T12:11:42.165558Z","shell.execute_reply.started":"2022-07-19T12:01:39.094642Z","shell.execute_reply":"2022-07-19T12:11:42.164476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf_output_1 = tf_output[:,1]\ntf_output_2 = np.where(tf_output_1>0,1,0)\ntf_output_2","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:20:44.354087Z","iopub.execute_input":"2022-07-19T12:20:44.354434Z","iopub.status.idle":"2022-07-19T12:20:44.363398Z","shell.execute_reply.started":"2022-07-19T12:20:44.354399Z","shell.execute_reply":"2022-07-19T12:20:44.362183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame(data={\"id\":test.id,\"sentiment\":label})\noutput","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:12:12.960424Z","iopub.execute_input":"2022-07-19T12:12:12.960798Z","iopub.status.idle":"2022-07-19T12:12:12.977516Z","shell.execute_reply.started":"2022-07-19T12:12:12.960766Z","shell.execute_reply":"2022-07-19T12:12:12.976410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.to_csv('submission.csv', index=False, quoting=3)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:23:01.672811Z","iopub.execute_input":"2022-07-19T12:23:01.673186Z","iopub.status.idle":"2022-07-19T12:23:01.706007Z","shell.execute_reply.started":"2022-07-19T12:23:01.673154Z","shell.execute_reply":"2022-07-19T12:23:01.705103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:23:07.594315Z","iopub.execute_input":"2022-07-19T12:23:07.594667Z","iopub.status.idle":"2022-07-19T12:23:07.622026Z","shell.execute_reply.started":"2022-07-19T12:23:07.594637Z","shell.execute_reply":"2022-07-19T12:23:07.621064Z"},"trusted":true},"execution_count":null,"outputs":[]}]}