{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Implementing the magic plot and time methods","metadata":{}},{"cell_type":"code","source":"!pip install ipython-autotime\n%matplotlib inline\n%load_ext autotime","metadata":{"execution":{"iopub.status.busy":"2023-05-09T08:57:30.052239Z","iopub.execute_input":"2023-05-09T08:57:30.052621Z","iopub.status.idle":"2023-05-09T08:57:42.60292Z","shell.execute_reply.started":"2023-05-09T08:57:30.05259Z","shell.execute_reply":"2023-05-09T08:57:42.601913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  ***Importing the Required Dependecies***","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os,re\nimport unicodedata\nimport gc\nimport time\n\nfrom tqdm.notebook import tqdm\n\nimport matplotlib.pyplot as plt\n\nimport nltk\nfrom nltk.corpus import stopwords\n\nimport transformers\nfrom transformers import TFAutoModel, AutoTokenizer\nfrom transformers import RobertaTokenizerFast, TFRobertaModel\nfrom tokenizers import Tokenizer, models, pre_tokenizers, decoders, processors\nfrom tokenizers import BertWordPieceTokenizer\n\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom tensorflow.keras.layers import Dense, Input\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.losses import BinaryCrossentropy\n\nfrom numba import jit, cuda ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-09T08:57:42.605071Z","iopub.execute_input":"2023-05-09T08:57:42.605494Z","iopub.status.idle":"2023-05-09T08:57:56.878509Z","shell.execute_reply.started":"2023-05-09T08:57:42.605457Z","shell.execute_reply":"2023-05-09T08:57:56.87737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Detect hardware, return appropriate distribution strategy\n","metadata":{}},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-05-09T08:57:56.879618Z","iopub.execute_input":"2023-05-09T08:57:56.880231Z","iopub.status.idle":"2023-05-09T08:58:01.675415Z","shell.execute_reply.started":"2023-05-09T08:57:56.8802Z","shell.execute_reply":"2023-05-09T08:58:01.674633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reading the required datasets using pandas dataframe\n\n**Downlading the translated Datasets**\n\n*Link* - https://www.kaggle.com/kashnitsky/jigsaw-multilingual-toxic-test-translated/","metadata":{}},{"cell_type":"code","source":"ls","metadata":{"execution":{"iopub.status.busy":"2023-05-09T08:58:01.677569Z","iopub.execute_input":"2023-05-09T08:58:01.67811Z","iopub.status.idle":"2023-05-09T08:58:02.729038Z","shell.execute_reply.started":"2023-05-09T08:58:01.678077Z","shell.execute_reply":"2023-05-09T08:58:02.727224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the data paths\ndata_path = '/kaggle/input/cvedata/'\n# loading all the train datasets\n\ntrain_data = pd.read_csv(data_path + 'final_training.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-09T08:58:02.731166Z","iopub.execute_input":"2023-05-09T08:58:02.731627Z","iopub.status.idle":"2023-05-09T08:58:03.256018Z","shell.execute_reply.started":"2023-05-09T08:58:02.731575Z","shell.execute_reply":"2023-05-09T08:58:03.254883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparation of Data","metadata":{}},{"cell_type":"markdown","source":"1.1. Gathering all the data and spliting into train, val and test","metadata":{}},{"cell_type":"code","source":"train_data.drop(columns=['Unnamed: 0'],inplace=True)\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-09T08:58:03.257269Z","iopub.execute_input":"2023-05-09T08:58:03.257618Z","iopub.status.idle":"2023-05-09T08:58:03.300233Z","shell.execute_reply.started":"2023-05-09T08:58:03.257589Z","shell.execute_reply":"2023-05-09T08:58:03.299012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Consequences","metadata":{}},{"cell_type":"code","source":"Description = np.array(train_data['Description'])\nprint('Description shape = ',Description.shape)\n\n#train_data1['toxic'] = np.where(train_data1['toxic'] > 0.5, 1, 0)\n#train_data3['toxic'] = np.where(train_data3['toxic'] > 0.5, 1, 0)\n#valid_translated['toxic'] = np.where(valid_translated['toxic'] > 0.5, 1, 0)\n\n\n\nConsequences = np.array(train_data['Consequences'])\nprint('Consequences labels shape = ',Consequences.shape)\n\ndata = pd.DataFrame(columns=['Description','Consequences'])\ndata['Description'] = Description\ndata['Consequences'] = Consequences","metadata":{"execution":{"iopub.status.busy":"2023-05-09T08:58:22.713037Z","iopub.execute_input":"2023-05-09T08:58:22.713468Z","iopub.status.idle":"2023-05-09T08:58:22.742505Z","shell.execute_reply.started":"2023-05-09T08:58:22.713433Z","shell.execute_reply":"2023-05-09T08:58:22.740854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(2048)\ntrain, valid, test = np.split(data.sample(frac=1), [int(.98*len(data)), int(.99*len(data))])\n\n# train, valid= np.split(data.sample(frac=1), [int(.8*len(data))])\n\nprint(\"Train rows = \", train.shape[0])\nprint(\"validate rows = \", valid.shape[0])\nprint(\"Test rows = \", test.shape[0])","metadata":{"execution":{"iopub.status.busy":"2023-05-09T08:58:23.612665Z","iopub.execute_input":"2023-05-09T08:58:23.613648Z","iopub.status.idle":"2023-05-09T08:58:23.642627Z","shell.execute_reply.started":"2023-05-09T08:58:23.613607Z","shell.execute_reply":"2023-05-09T08:58:23.641417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Train, Validation and Test data and labels**","metadata":{}},{"cell_type":"code","source":"x_train,y_train = train['Description'],np.array(train.iloc[:,1:])\nx_valid,y_valid = valid['Description'],np.array(valid.iloc[:,1:])\nx_test, y_test = test['Description'],np.array(test.iloc[:,1:])\n\nprint('Comment_text shapes')\nprint('x_train shape = ',x_train.shape)\nprint('x_valid shape = ',x_valid.shape)\nprint('x_test shape = ',x_test.shape)\nprint('-'*35)\nprint('Labels shapes')\nprint('y_train shape = ',y_train.shape)\nprint('y_valid shape = ',y_valid.shape)\nprint('y_test shape = ',y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-09T08:58:24.28246Z","iopub.execute_input":"2023-05-09T08:58:24.282848Z","iopub.status.idle":"2023-05-09T08:58:24.292907Z","shell.execute_reply.started":"2023-05-09T08:58:24.282818Z","shell.execute_reply":"2023-05-09T08:58:24.291766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2023-05-09T08:58:24.892755Z","iopub.execute_input":"2023-05-09T08:58:24.893775Z","iopub.status.idle":"2023-05-09T08:58:24.900485Z","shell.execute_reply.started":"2023-05-09T08:58:24.893735Z","shell.execute_reply":"2023-05-09T08:58:24.899406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_data\ndel data\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-09T08:58:37.994672Z","iopub.execute_input":"2023-05-09T08:58:37.995846Z","iopub.status.idle":"2023-05-09T08:58:38.365562Z","shell.execute_reply.started":"2023-05-09T08:58:37.995801Z","shell.execute_reply":"2023-05-09T08:58:38.364362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Final Data to predict","metadata":{}},{"cell_type":"code","source":"#final_test_data = np.array(test_translated['translated'])\n#print(\"Shape of final_test_data: \", final_test_data.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-09T08:58:39.573134Z","iopub.execute_input":"2023-05-09T08:58:39.57355Z","iopub.status.idle":"2023-05-09T08:58:39.578509Z","shell.execute_reply.started":"2023-05-09T08:58:39.573518Z","shell.execute_reply":"2023-05-09T08:58:39.577374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Processing and Tokenizing the Data","metadata":{}},{"cell_type":"markdown","source":"**Functions to clean the data**","metadata":{}},{"cell_type":"code","source":"# Stopword list\npattern = re.compile(r'\\b('+r'|'.join(stopwords.words('english'))+r')\\b\\s*')\n\n# @cuda.jit(device=True)\ndef unicode_to_ascii(s):\n  return ''.join(c for c in unicodedata.normalize('NFD', s)\n      if unicodedata.category(c) != 'Mn')\n\n# @tf.function()\ndef preprocess_sentence(w):\n    print(w)\n    w = unicode_to_ascii(w.lower().strip())\n    \n    #replacing email addresses with blank space\n    w = re.sub(r\"[a-zA-Z0-9_\\-\\.]+@[a-zA-Z0-9_\\-\\.]+\\.[a-zA-Z]{2,5}\",\" \",w)\n    \n    #replacing urls with blank space\n    w = re.sub(r\"\\bhttp:\\/\\/([^\\/]*)\\/([^\\s]*)|https:\\/\\/([^\\/]*)\\/([^\\s]*)\",\" \",w)\n    \n    # creating a space between a word and the punctuation following it\n    w = re.sub(r\"([?.!,¿])\", r\" \\1 \", w)\n    w = re.sub(r'[\" \"]+', \" \", w)\n    \n    # replacing all the stopwords\n    w = pattern.sub('',w)\n    \n    # removes all the punctuations\n    w = re.sub(r\"[^a-zA-Z]+\", \" \", w)\n    \n    w = w.strip()\n\n    # adding a start and an end token to the sentence so that the model know when to start and stop predicting.\n#     w = '<start> ' + w + ' <end>'\n    \n    return w\n\npreprocess_sentence_vect = np.vectorize(preprocess_sentence)","metadata":{"execution":{"iopub.status.busy":"2023-05-09T09:01:54.997296Z","iopub.execute_input":"2023-05-09T09:01:54.99775Z","iopub.status.idle":"2023-05-09T09:01:55.007627Z","shell.execute_reply.started":"2023-05-09T09:01:54.997717Z","shell.execute_reply":"2023-05-09T09:01:55.006478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Processing and Tokenizing training, validation and testing data","metadata":{}},{"cell_type":"code","source":"def fast_clean(array,chunk_size=256):\n    cleaned_array = []\n    \n    for i in tqdm(range(0, len(array), chunk_size)):\n        print(type(array[i:i+chunk_size]))\n        text_chunk = preprocess_sentence_vect(array[i:i+chunk_size])\n        cleaned_array.extend(text_chunk)\n\n    return np.array(cleaned_array)","metadata":{"execution":{"iopub.status.busy":"2023-05-09T09:01:56.352949Z","iopub.execute_input":"2023-05-09T09:01:56.353406Z","iopub.status.idle":"2023-05-09T09:01:56.361063Z","shell.execute_reply.started":"2023-05-09T09:01:56.353368Z","shell.execute_reply":"2023-05-09T09:01:56.359566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(x_train[0])","metadata":{"execution":{"iopub.status.busy":"2023-05-09T09:01:56.867436Z","iopub.execute_input":"2023-05-09T09:01:56.867831Z","iopub.status.idle":"2023-05-09T09:01:56.875272Z","shell.execute_reply.started":"2023-05-09T09:01:56.867797Z","shell.execute_reply":"2023-05-09T09:01:56.874169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train","metadata":{"execution":{"iopub.status.busy":"2023-05-09T09:01:57.373433Z","iopub.execute_input":"2023-05-09T09:01:57.37385Z","iopub.status.idle":"2023-05-09T09:01:57.38312Z","shell.execute_reply.started":"2023-05-09T09:01:57.373817Z","shell.execute_reply":"2023-05-09T09:01:57.38193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = fast_clean(x_train)","metadata":{"execution":{"iopub.status.busy":"2023-05-09T09:01:59.712965Z","iopub.execute_input":"2023-05-09T09:01:59.713654Z","iopub.status.idle":"2023-05-09T09:01:59.804683Z","shell.execute_reply.started":"2023-05-09T09:01:59.71358Z","shell.execute_reply":"2023-05-09T09:01:59.80319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_valid = fast_clean(x_valid)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = fast_clean(x_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_test_data = fast_clean(final_test_data)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ii. Tokenizing all the data","metadata":{}},{"cell_type":"code","source":"vocab_size = 50000\nMAX_LEN = 128\ntrunc_type='post'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def regular_encode(texts, tokenizer, maxlen=512):\n    tokenizer.pad_token = tokenizer.pad_token\n    tokenizer.unk_token = tokenizer.unk_token\n    \n    enc_di = tokenizer.batch_encode_plus(\n        list(texts), \n        return_attention_masks=False, \n        return_token_type_ids=False,\n        pad_to_max_length=True,\n        max_length=maxlen\n    )\n    \n    return np.array(enc_di['input_ids'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def chunk_encode(texts, tokenizer, maxlen=512, chunk_size=256):\n    tokenizer.pad_token = tokenizer.pad_token\n    tokenizer.unk_token = tokenizer.unk_token\n    \n    all_ids = []\n    for i in tqdm(range(0, len(texts), chunk_size)):\n        text_chunk = texts[i:i+chunk_size].tolist()\n        encs = tokenizer.encode_plus(text_chunk)\n        all_ids.extend([enc.ids for enc in encs])\n    \n    return np.array(all_ids)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fast_encode(texts, tokenizer, chunk_size=256, maxlen=512):\n\n    tokenizer.enable_truncation(max_length=maxlen)\n    tokenizer.enable_padding(max_length=maxlen)\n    all_ids = []\n    \n    for i in tqdm(range(0, len(texts), chunk_size)):\n        text_chunk = texts[i:i+chunk_size].tolist()\n        encs = tokenizer.encode_batch(text_chunk)\n        all_ids.extend([enc.ids for enc in encs])\n    \n    return np.array(all_ids)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defining the tokenizer function\n\ntokenizer = RobertaTokenizerFast.from_pretrained('roberta-large')\nprint(tokenizer.save_pretrained('.'))\nprint(tokenizer)\n\n# fast_tokenizer = BertWordPieceTokenizer('vocab.txt', lowercase=False)\n# fast_tokenizer\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_valid=regular_encode(x_valid, tokenizer, maxlen=MAX_LEN)\nx_test=regular_encode(x_test, tokenizer, maxlen=MAX_LEN)\nfinal_test_data = regular_encode(final_test_data, tokenizer, maxlen=MAX_LEN)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train=regular_encode(x_train, tokenizer, maxlen=MAX_LEN)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('New shape of comments and labels after TOKENIZATION and PROCESSING:-')\nprint('-'*50)\nprint('Data for Training and Evaluation:\\n')\nprint('x_train shape = ',x_train.shape)\nprint('x_valid shape = ',x_valid.shape)\nprint('x_test shape = ',x_test.shape)\nprint('-'*35)\nprint('Labels shapes')\nprint('y_train shape = ',y_train.shape)\nprint('y_valid shape = ',y_valid.shape)\nprint('y_test shape = ',y_test.shape)\nprint('-'*50)\nprint('The Final data for Predication\\n')\nprint('final_test_data shape = ',final_test_data.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Converting all the data to Tensors**","metadata":{}},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\nEPOCHS = 2\nBATCH_SIZE = 32 * strategy.num_replicas_in_sync","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train, y_train))\n    .repeat()\n    .shuffle(4096)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid = (\n    tf.data.Dataset\n    .from_tensor_slices((x_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\ntest = (\n    tf.data.Dataset\n    .from_tensor_slices((x_test,y_test))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\nfinal_test_data = (\n    tf.data.Dataset\n    .from_tensor_slices(final_test_data)\n    .batch(BATCH_SIZE)\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Creating a Model**","metadata":{}},{"cell_type":"code","source":"from transformers import RobertaConfig, RobertaForMaskedLM","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roberta = 'roberta-large'\n\nconfig = RobertaConfig(hidden_dropout_prob=0.2,attention_probs_dropout_prob=0.2)\nconfig","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = x_train.shape[0] // BATCH_SIZE\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(transformer, max_len=512):\n    \n    input_word_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\")\n#     input_masks_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_masks_ids\")\n    \n    embedding_layer = transformer(input_word_ids)[0]\n#     cls_token = sequence_output[:, 0, :]\n            \n    X = tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(50, return_sequences=True, dropout=0.1, recurrent_dropout=0.1))(embedding_layer)\n    X = tf.keras.layers.GlobalMaxPool1D()(X)\n    X = tf.keras.layers.Dense(50, activation='relu')(X)\n    X = tf.keras.layers.Dropout(0.2)(X)\n    X = tf.keras.layers.Dense(1, activation='sigmoid')(X)\n    model = tf.keras.Model(inputs=[input_word_ids], outputs = X)\n\n    for layer in model.layers[:3]:\n      layer.trainable = False\n    \n    model.compile(Adam(lr=1e-5),\n                  loss='binary_crossentropy',\n                  metrics=['accuracy'])\n    \n    return model","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"configs = {\"attention_probs_dropout_prob\": 0.2,\n  \"hidden_act\": \"gelu\",\n  \"hidden_dropout_prob\": 0.2,\n  \"hidden_size\": 1024,\n  \"initializer_range\": 0.02,\n  \"intermediate_size\": 4096,\n  \"layer_norm_eps\": 1e-05,\n  \"max_position_embeddings\": 514,\n  \"model_type\": \"roberta\",\n  \"num_attention_heads\": 16,\n  \"num_hidden_layers\": 24,\n  \"pad_token_id\": 1,\n  \"type_vocab_size\": 1,\n  \"vocab_size\": 50265}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    transformer_layer = TFAutoModel.from_pretrained('roberta-large',config=configs)\n\n    model = build_model(transformer_layer, max_len=MAX_LEN)\nmodel.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Stage 1","metadata":{}},{"cell_type":"code","source":"train_history = model.fit(\n    train,\n    steps_per_epoch=n_steps,\n    validation_data=valid,\n    epochs=4\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs_range = range(4)\n\nplt.figure(figsize=(8, 5))\n\n# plt.subplot(1,2,1)\nplt.plot(epochs_range,train_history.history['accuracy'], label='accuracy')\nplt.plot(epochs_range,train_history.history['val_accuracy'], label = 'val_accuracy')\nplt.plot(epochs_range,train_history.history['loss'], label='loss')\nplt.plot(epochs_range,train_history.history['val_loss'], label = 'val_loss')\n\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy/Loss')\nplt.legend(loc='center right')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss,accuracy = model.evaluate(test,verbose=1)\nprint(loss,accuracy)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Submission**","metadata":{}},{"cell_type":"code","source":"sub1 = pd.read_csv(data_path + 'sample_submission.csv')\nsub1['toxic'] = model.predict(final_test_data, verbose=1)\nsub1.to_csv('submission.csv', index=False)\nsub1.head(15)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # submission dataset\n# sub1 = pd.read_csv(data_path + 'sample_submission.csv')\n# sub1['toxic'] = model.predict(final_test_data, verbose=1)\n# sub1.to_csv('submission.csv', index=False)\n# sub1.head(15)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub1.to_csv('/kaggle/working/submission_0.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Stage 2","metadata":{}},{"cell_type":"code","source":"# n_steps = x_valid.shape[0] // BATCH_SIZE\n# train_history_2 = model.fit(\n#     valid.repeat(),\n#     steps_per_epoch=n_steps,\n#     epochs=3\n# )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*Evaluating the model*","metadata":{}},{"cell_type":"code","source":"# loss,accuracy = model.evaluate(test,verbose=1)\n# print(loss,accuracy)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Writing the predicted values to a .csv file","metadata":{}},{"cell_type":"code","source":"# # submission dataset\n# sub2 = pd.read_csv(data_path + 'sample_submission.csv')\n# sub2['toxic'] = model.predict(final_test_data, verbose=1)\n# sub2.to_csv('submission.csv', index=False)\n# sub2.head(15)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thank You Hardware!!!!!","metadata":{"trusted":true}}]}