{"metadata":{"colab":{"provenance":[],"gpuType":"T4"},"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"accelerator":"GPU","language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"},{"sourceId":2644,"sourceType":"modelInstanceVersion","modelInstanceId":1910},{"sourceId":2938,"sourceType":"modelInstanceVersion","modelInstanceId":2180}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h2 align=center> <b><u> Fine-Tuned BERT for Quora Insincere Questions Classification </u></b>\n</h2>","metadata":{"id":"zGCJYkQj_Uu2"}},{"cell_type":"markdown","source":"### 1. Check GPU Availability and install dependencies","metadata":{"id":"mpe6GhLuBJWB"}},{"cell_type":"code","source":"!nvidia-smi","metadata":{"id":"8V9c8vzSL3aj","outputId":"cc54afaa-d2fa-429f-8228-863c8423d1b6","execution":{"iopub.status.busy":"2024-03-02T15:35:18.530040Z","iopub.execute_input":"2024-03-02T15:35:18.530545Z","iopub.status.idle":"2024-03-02T15:35:19.524079Z","shell.execute_reply.started":"2024-03-02T15:35:18.530516Z","shell.execute_reply":"2024-03-02T15:35:19.522998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow_text\nimport tensorflow_text as text  # Registers the ops.\n\n\n# After running this cell, we have to restart the Kernel and after restarting run this cell once again!","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-03-02T15:35:24.705008Z","iopub.execute_input":"2024-03-02T15:35:24.706006Z","iopub.status.idle":"2024-03-02T15:35:41.552274Z","shell.execute_reply.started":"2024-03-02T15:35:24.705961Z","shell.execute_reply":"2024-03-02T15:35:41.551258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2. Import the Quora Insincere Questions Dataset","metadata":{"id":"IMsEoT3Fg4Wg"}},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nimport tensorflow_hub as hub","metadata":{"id":"GmqEylyFYTdP","outputId":"8e9f0646-e8d5-4279-9da5-2e66298f6762","execution":{"iopub.status.busy":"2024-03-02T15:35:41.554793Z","iopub.execute_input":"2024-03-02T15:35:41.555338Z","iopub.status.idle":"2024-03-02T15:35:41.560074Z","shell.execute_reply.started":"2024-03-02T15:35:41.555308Z","shell.execute_reply":"2024-03-02T15:35:41.558921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"TF Version: \", tf.__version__)\nprint(\"Eager mode: \", tf.executing_eagerly())\nprint(\"Hub version: \", hub.__version__)\nprint(\"GPU is\", \"available\" if tf.config.experimental.list_physical_devices(\"GPU\") else \"NOT AVAILABLE\")","metadata":{"id":"ZuX1lB8pPJ-W","outputId":"9d0eb66e-32c2-480f-ac6d-8f97b92838e3","execution":{"iopub.status.busy":"2024-03-02T15:35:42.489139Z","iopub.execute_input":"2024-03-02T15:35:42.489504Z","iopub.status.idle":"2024-03-02T15:35:42.657955Z","shell.execute_reply.started":"2024-03-02T15:35:42.489476Z","shell.execute_reply":"2024-03-02T15:35:42.656833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\n\ndf = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\")\ndf.shape","metadata":{"id":"0nI-9itVwCCQ","outputId":"fec5f1a4-561e-4302-ae08-aa1e1411165b","execution":{"iopub.status.busy":"2024-03-02T15:35:44.389907Z","iopub.execute_input":"2024-03-02T15:35:44.390638Z","iopub.status.idle":"2024-03-02T15:35:48.732566Z","shell.execute_reply.started":"2024-03-02T15:35:44.390603Z","shell.execute_reply":"2024-03-02T15:35:48.731550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.tail(20)","metadata":{"id":"yeHE98KiMvDd","outputId":"ffbfa52a-b91b-4e05-b223-ec88c8156c0c","execution":{"iopub.status.busy":"2024-03-02T15:35:48.734004Z","iopub.execute_input":"2024-03-02T15:35:48.734309Z","iopub.status.idle":"2024-03-02T15:35:48.749588Z","shell.execute_reply.started":"2024-03-02T15:35:48.734283Z","shell.execute_reply":"2024-03-02T15:35:48.748739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.target.plot(kind='hist', title = 'Target Distribution');","metadata":{"id":"leRFRWJMocVa","outputId":"531296b7-0bc2-418d-ac01-461af40e8510","execution":{"iopub.status.busy":"2024-03-02T15:35:55.095269Z","iopub.execute_input":"2024-03-02T15:35:55.095641Z","iopub.status.idle":"2024-03-02T15:35:55.506390Z","shell.execute_reply.started":"2024-03-02T15:35:55.095609Z","shell.execute_reply":"2024-03-02T15:35:55.505267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3. Preparing Input Data for Training and Evaluation","metadata":{"id":"ELjswHcFHfp3"}},{"cell_type":"markdown","source":"Since the given training set has 1 million samples, it would a huge time to train the model on the entire training data. So, a better approach would be to go for a pretrained model and fine tune using just 0.75% of the training samples.\n\nAlso to counter the huge class imbalance, we use downsampling of the 0 class.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.utils import resample\n\n# Splitting the dataset into train and remaining (90% and 10% respectively)\ntrain_df, remaining = train_test_split(df, random_state=42, train_size=0.0075, stratify=df.target.values)\n\n# Splitting the remaining dataset into validation and test sets (90% and 10% respectively)\nvalid_df, _ = train_test_split(remaining, random_state=42, train_size=0.00075, stratify=remaining.target.values)\n\n# Separating majority and minority classes in the training dataset\nmajority_class = train_df[train_df.target == 0]\nminority_class = train_df[train_df.target == 1]\n\n# Downsampling the majority class to match the size of the minority class\ndownsampled_majority = resample(majority_class,\n                                replace=False,  # sample without replacement\n                                n_samples=len(minority_class),  # match minority class\n                                random_state=42)  # for reproducible results\n\n# Combining the minority class with the downsampled majority class\ndownsampled_train_df = pd.concat([downsampled_majority, minority_class])\n\n# Shuffle the downsampled training dataset\ndownsampled_train_df = downsampled_train_df.sample(frac=1, random_state=42)\n\n# Display the shapes of the downsampled training and validation datasets\ndownsampled_train_df.shape, valid_df.shape","metadata":{"id":"fScULIGPwuWk","outputId":"f3f0dc5f-c7a9-4901-b3fe-dcff630b0eaa","execution":{"iopub.status.busy":"2024-03-02T15:39:20.486283Z","iopub.execute_input":"2024-03-02T15:39:20.486704Z","iopub.status.idle":"2024-03-02T15:39:21.896045Z","shell.execute_reply.started":"2024-03-02T15:39:20.486666Z","shell.execute_reply":"2024-03-02T15:39:21.895068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"downsampled_train_df.target.plot(kind='hist', title = 'Target Distribution');","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:40:00.400884Z","iopub.execute_input":"2024-03-02T15:40:00.401283Z","iopub.status.idle":"2024-03-02T15:40:00.696599Z","shell.execute_reply.started":"2024-03-02T15:40:00.401255Z","shell.execute_reply":"2024-03-02T15:40:00.695706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = downsampled_train_df","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:40:20.882522Z","iopub.execute_input":"2024-03-02T15:40:20.882917Z","iopub.status.idle":"2024-03-02T15:40:20.888076Z","shell.execute_reply.started":"2024-03-02T15:40:20.882885Z","shell.execute_reply":"2024-03-02T15:40:20.887211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = train_df[\"question_text\"]\ny_train = train_df[\"target\"]\n\nX_valid = valid_df[\"question_text\"]\ny_valid = valid_df[\"target\"]","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:40:24.374607Z","iopub.execute_input":"2024-03-02T15:40:24.375474Z","iopub.status.idle":"2024-03-02T15:40:24.380620Z","shell.execute_reply.started":"2024-03-02T15:40:24.375438Z","shell.execute_reply":"2024-03-02T15:40:24.379678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_train.shape)\nprint(y_train.shape)\n\nprint(X_valid.shape)\nprint(y_valid.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:40:26.611322Z","iopub.execute_input":"2024-03-02T15:40:26.612033Z","iopub.status.idle":"2024-03-02T15:40:26.616892Z","shell.execute_reply.started":"2024-03-02T15:40:26.612000Z","shell.execute_reply":"2024-03-02T15:40:26.615895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4. Input Format for BERT","metadata":{"id":"9QinzNq6OsP1"}},{"cell_type":"markdown","source":"**Token IDs** - This corresponds to the tokenized strings padded with 0s\n                upto the max sequence length and beginning with CLS and ending with SEP. <br><br>\n**Input Mask** - Note that BERT uses Self-Attention Networks to provide\n                 contextualised embeddings corresponding to each token in the token string i.e., for each word in the string BERT looks to the left and right of it in the sentence so as to find contextual meaning of the word in the sentence (say, if there is a \"the\", then look at the noun to which it points). Now, note that we have padded our token strings with 0s upto the max seq length, but we do not want the padding 0s to influence the contextual information to be derived.  The Input Mask is a list of same length as the length of Token Ids (ie the max seq length) where there is a 0 for a padding and 1 for a valid token. The 0s will cancel out the internal multiplications that we perform for capturing the Self Attention for contextual information. <br><br>\n\n**Input Type IDs** - Note that originally BERT was pretrained on two   tasks, Masked Language Modelling (where random words from the sentence would be masked and it would be the task for the BERT to predict what those masked words are) and the other task was Next Sentence Prediction or NSP (Given two sentences, the BERT has to predict which came first and which came after. The first sentence was given the value 0 and the next sentence was given the value 1).<br><br>\n**In Text classification, we are dealing with only 1 sequence at a time, so our input type IDs would just be a vector with all values 0. **\n","metadata":{"id":"shyvv_0JaIzj"}},{"cell_type":"markdown","source":"### 5. Checking the tokenization process","metadata":{}},{"cell_type":"code","source":"preprocessor = hub.KerasLayer(\n    \"https://kaggle.com/models/tensorflow/bert/frameworks/TensorFlow2/variations/en-uncased-preprocess/versions/3\")\n\n# Tokenize the input text\ninput_text = [\"Hello, how are you?\"]\ntokenized_output = preprocessor(input_text)\n\n# Print token IDs\nprint(tokenized_output['input_word_ids'])\nprint(tokenized_output['input_mask'])\nprint(tokenized_output['input_type_ids'])","metadata":{"execution":{"iopub.status.busy":"2024-03-02T15:41:08.436391Z","iopub.execute_input":"2024-03-02T15:41:08.437089Z","iopub.status.idle":"2024-03-02T15:41:12.933907Z","shell.execute_reply.started":"2024-03-02T15:41:08.437050Z","shell.execute_reply":"2024-03-02T15:41:12.932894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Upon checking the input word ids, we find that the tokens are --> \"Hello\", \"#,\", \"how\", \"are\", \"you\" and \"#?\" {'#' signifies that the succeeding character ',' is attached to characters before] and they are encoded as [7592 1010 2129 2024 2017 1029]. \n\nNote that this is not the whole story. We also have to make sure that each sequence is initiated with the CLS token (which signifies start of sequence and has a token_id 101) and ended with SEP (separator) which means end of sequence and has a token_id 102. Also we have to make sure that all tensors have sequence size equal to the max_sequence_length by using padding.","metadata":{}},{"cell_type":"markdown","source":"### 6. Add a Classification Head to the BERT Layer","metadata":{"id":"GZxe-7yhPyQe"}},{"cell_type":"markdown","source":"We only need the pooled_output that represents the whole sentence (using the CLS token that contains the contextual information of the whole sentence) and not the sequence_output.","metadata":{"id":"eED7TDu0vQ2Y"}},{"cell_type":"code","source":"# Building the model\n\ntext_input = tf.keras.layers.Input(shape=(), dtype=tf.string)\npreprocessor = hub.KerasLayer(\n    \"https://kaggle.com/models/tensorflow/bert/frameworks/TensorFlow2/variations/en-uncased-preprocess/versions/3\")\nencoder_inputs = preprocessor(text_input)\nencoder = hub.KerasLayer(\n    \"https://www.kaggle.com/models/tensorflow/bert/frameworks/TensorFlow2/variations/bert-en-uncased-l-12-h-768-a-12/versions/2\",\n    trainable=True)\noutputs = encoder(encoder_inputs)\npooled_output = outputs[\"pooled_output\"]      # [batch_size, 768].\nsequence_output = outputs[\"sequence_output\"]  # [batch_size, seq_length, 768].\n\n\n\n\n\n# Classification\n# Add dropout layer\ndrop1 = tf.keras.layers.Dropout(0.2)(pooled_output)\n\n# Add hidden dense layers\nhidden1 = tf.keras.layers.Dense(128, activation='relu')(drop1)\ndrop2 = tf.keras.layers.Dropout(0.2)(hidden1)\nhidden2 = tf.keras.layers.Dense(32, activation='relu')(drop2)\ndrop3 = tf.keras.layers.Dropout(0.2)(hidden2)\n\n# Output layer\noutput_layer = tf.keras.layers.Dense(1, activation='sigmoid', name='output')(drop3)\n\n\nmodel=tf.keras.Model(inputs=[text_input],outputs=[output_layer])","metadata":{"id":"G9il4gtlADcp","execution":{"iopub.status.busy":"2024-03-02T17:00:27.169284Z","iopub.execute_input":"2024-03-02T17:00:27.169546Z","iopub.status.idle":"2024-03-02T17:00:42.041301Z","shell.execute_reply.started":"2024-03-02T17:00:27.169523Z","shell.execute_reply":"2024-03-02T17:00:42.040432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 7. Fine-Tune BERT for Text Classification","metadata":{"id":"S6maM-vr7YaJ"}},{"cell_type":"code","source":"model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=2e-5),\n              loss = tf.keras.losses.BinaryCrossentropy(),\n              metrics = tf.keras.metrics.BinaryAccuracy())\nmodel.summary()","metadata":{"id":"ptCtiiONsBgo","outputId":"5586d260-0153-4659-e47e-a5420023debb","execution":{"iopub.status.busy":"2024-03-02T17:00:42.042828Z","iopub.execute_input":"2024-03-02T17:00:42.043136Z","iopub.status.idle":"2024-03-02T17:00:42.103091Z","shell.execute_reply.started":"2024-03-02T17:00:42.043110Z","shell.execute_reply":"2024-03-02T17:00:42.102235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define callbacks\ncallbacks = [\n    tf.keras.callbacks.ModelCheckpoint(\n        filepath='best_model.h5',  # Path to save the best model\n        save_best_only=True,  # Save only the best model\n        monitor='val_binary_accuracy',  # Quantity to be monitored\n        save_weights_only=True,  # Do not save the entire model\n        verbose=1,  # Verbosity mode. 0 or 1.\n        save_freq='epoch'  # Save the model at the end of every epoch\n    ),\n    tf.keras.callbacks.EarlyStopping(\n        patience=10,  # Number of epochs with no improvement after which training will be stopped\n        monitor='val_binary_accuracy',  # Quantity to be monitored\n        restore_best_weights=True  # Restore model weights from the epoch with the best value of the monitored quantity\n    ),\n    tf.keras.callbacks.ReduceLROnPlateau(\n        monitor='val_binary_accuracy',  # Quantity to be monitored\n        factor=0.5,  # Factor by which the learning rate will be reduced. new_lr = lr * factor\n        patience=5,  # Number of epochs with no improvement after which learning rate will be reduced\n        min_lr=1e-7  # Lower bound on the learning rate\n    )\n]","metadata":{"execution":{"iopub.status.busy":"2024-03-02T17:01:23.459229Z","iopub.execute_input":"2024-03-02T17:01:23.459592Z","iopub.status.idle":"2024-03-02T17:01:23.466747Z","shell.execute_reply.started":"2024-03-02T17:01:23.459564Z","shell.execute_reply":"2024-03-02T17:01:23.465673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the model architecture\ntf.keras.utils.plot_model(model, show_shapes=True, dpi=76)","metadata":{"id":"6GJaFnkbMtPL","outputId":"83dad392-eb4c-481b-fd54-5955c91c1197","execution":{"iopub.status.busy":"2024-03-02T17:01:25.085095Z","iopub.execute_input":"2024-03-02T17:01:25.085468Z","iopub.status.idle":"2024-03-02T17:01:25.293638Z","shell.execute_reply.started":"2024-03-02T17:01:25.085442Z","shell.execute_reply":"2024-03-02T17:01:25.292766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm \"/kaggle/working/model.png\"","metadata":{"execution":{"iopub.status.busy":"2024-03-02T17:01:27.375071Z","iopub.execute_input":"2024-03-02T17:01:27.375448Z","iopub.status.idle":"2024-03-02T17:01:28.540208Z","shell.execute_reply.started":"2024-03-02T17:01:27.375417Z","shell.execute_reply":"2024-03-02T17:01:28.538860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train model\nepochs = 100\nhistory = model.fit(X_train, \n                    y_train,\n                    validation_data = (X_valid, y_valid),\n                    epochs=epochs,\n                    verbose=1,\n                    callbacks=callbacks\n                   )","metadata":{"id":"OcREcgPUHr9O","outputId":"c9fa3758-6c20-4158-d22c-18da8aa40dac","execution":{"iopub.status.busy":"2024-03-02T17:01:28.730213Z","iopub.execute_input":"2024-03-02T17:01:28.730569Z","iopub.status.idle":"2024-03-02T17:10:11.851546Z","shell.execute_reply.started":"2024-03-02T17:01:28.730537Z","shell.execute_reply":"2024-03-02T17:10:11.850661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 8. Evaluate the BERT Text Classification Model","metadata":{"id":"kNZl1lx_cA5Y"}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndef plot_graphs(history, metric):\n    plt.plot(history.history[metric])\n    plt.plot(history.history['val_'+metric], '')\n    plt.xlabel(\"Epochs\")\n    plt.ylabel(metric)\n    plt.legend([metric, 'val_'+metric])\n    plt.show()","metadata":{"id":"dCjgrUYH_IsE","execution":{"iopub.status.busy":"2024-03-02T17:10:11.853259Z","iopub.execute_input":"2024-03-02T17:10:11.853552Z","iopub.status.idle":"2024-03-02T17:10:11.859380Z","shell.execute_reply.started":"2024-03-02T17:10:11.853527Z","shell.execute_reply":"2024-03-02T17:10:11.858411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_graphs(history, 'loss')","metadata":{"id":"v6lrFRra_KmA","outputId":"f419e4a5-f935-4cf1-8eba-2cd30f515410","execution":{"iopub.status.busy":"2024-03-02T17:10:11.860416Z","iopub.execute_input":"2024-03-02T17:10:11.860799Z","iopub.status.idle":"2024-03-02T17:11:11.881079Z","shell.execute_reply.started":"2024-03-02T17:10:11.860735Z","shell.execute_reply":"2024-03-02T17:11:11.880000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_graphs(history, 'binary_accuracy')","metadata":{"id":"opu9neBA_98R","outputId":"e1b12c79-ff53-4f5a-c8b6-92e0b52a5b54","execution":{"iopub.status.busy":"2024-03-02T17:11:11.883646Z","iopub.execute_input":"2024-03-02T17:11:11.884327Z","iopub.status.idle":"2024-03-02T17:11:12.166590Z","shell.execute_reply.started":"2024-03-02T17:11:11.884298Z","shell.execute_reply":"2024-03-02T17:11:12.165598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 9. Sanity Checking","metadata":{}},{"cell_type":"code","source":"sample_example = [\"why are americans racist?\", \"why are indians so black?\", \"have a great day!\"]\npreds = model.predict(sample_example)\nthreshold = 0.5 #between 0 and 1\n\n\n[\"Insincere\" if pred>=threshold else \"Sincere\" for pred in preds]","metadata":{"id":"hkhtCCgnUbY6","outputId":"e0d850f3-7162-4b20-8411-f21cf02fe08e","execution":{"iopub.status.busy":"2024-03-02T17:11:12.167710Z","iopub.execute_input":"2024-03-02T17:11:12.168008Z","iopub.status.idle":"2024-03-02T17:11:13.004079Z","shell.execute_reply.started":"2024-03-02T17:11:12.167983Z","shell.execute_reply":"2024-03-02T17:11:13.003104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 10. Trying the Trained BERT Model on Test Dataset provided in the contest","metadata":{"id":"t1wgRroGD4xf"}},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/sample_submission.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/test.csv\")","metadata":{"id":"ELn3hNHyG-wK","outputId":"fa887206-f0ac-4cba-992b-7e420765a679","execution":{"iopub.status.busy":"2024-03-02T17:11:13.005217Z","iopub.execute_input":"2024-03-02T17:11:13.005563Z","iopub.status.idle":"2024-03-02T17:11:13.914927Z","shell.execute_reply.started":"2024-03-02T17:11:13.005535Z","shell.execute_reply":"2024-03-02T17:11:13.914108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-03-02T17:11:13.916032Z","iopub.execute_input":"2024-03-02T17:11:13.916342Z","iopub.status.idle":"2024-03-02T17:11:13.926394Z","shell.execute_reply.started":"2024-03-02T17:11:13.916316Z","shell.execute_reply":"2024-03-02T17:11:13.925501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-03-02T17:11:13.927619Z","iopub.execute_input":"2024-03-02T17:11:13.927984Z","iopub.status.idle":"2024-03-02T17:11:13.939711Z","shell.execute_reply.started":"2024-03-02T17:11:13.927960Z","shell.execute_reply":"2024-03-02T17:11:13.938802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_df[\"question_text\"]","metadata":{"execution":{"iopub.status.busy":"2024-03-02T17:11:13.940865Z","iopub.execute_input":"2024-03-02T17:11:13.941196Z","iopub.status.idle":"2024-03-02T17:11:13.949792Z","shell.execute_reply.started":"2024-03-02T17:11:13.941169Z","shell.execute_reply":"2024-03-02T17:11:13.948963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-02T17:11:13.953097Z","iopub.execute_input":"2024-03-02T17:11:13.953720Z","iopub.status.idle":"2024-03-02T17:11:13.961334Z","shell.execute_reply.started":"2024-03-02T17:11:13.953694Z","shell.execute_reply":"2024-03-02T17:11:13.960473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_predict=model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-02T17:11:13.962647Z","iopub.execute_input":"2024-03-02T17:11:13.963102Z","iopub.status.idle":"2024-03-02T17:49:46.218581Z","shell.execute_reply.started":"2024-03-02T17:11:13.963070Z","shell.execute_reply":"2024-03-02T17:49:46.217512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_predict.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-02T17:49:46.220141Z","iopub.execute_input":"2024-03-02T17:49:46.220538Z","iopub.status.idle":"2024-03-02T17:49:46.227515Z","shell.execute_reply.started":"2024-03-02T17:49:46.220507Z","shell.execute_reply":"2024-03-02T17:49:46.226494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold = 0.5\npreds = [1 if pred[0]>=threshold else 0 for pred in y_predict]","metadata":{"execution":{"iopub.status.busy":"2024-03-02T17:49:46.229048Z","iopub.execute_input":"2024-03-02T17:49:46.229705Z","iopub.status.idle":"2024-03-02T17:49:47.356652Z","shell.execute_reply.started":"2024-03-02T17:49:46.229657Z","shell.execute_reply":"2024-03-02T17:49:47.355632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 11. Submission File","metadata":{"id":"rEWbDMrvUKRH"}},{"cell_type":"code","source":"sample_submission['prediction'] = preds","metadata":{"id":"nr9McMkGUNuC","execution":{"iopub.status.busy":"2024-03-02T17:49:47.358382Z","iopub.execute_input":"2024-03-02T17:49:47.358793Z","iopub.status.idle":"2024-03-02T17:49:47.533725Z","shell.execute_reply.started":"2024-03-02T17:49:47.358738Z","shell.execute_reply":"2024-03-02T17:49:47.532586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.to_csv('submission.csv', index=False)","metadata":{"id":"7TccmUU5UyjP","execution":{"iopub.status.busy":"2024-03-02T17:49:47.535213Z","iopub.execute_input":"2024-03-02T17:49:47.535540Z","iopub.status.idle":"2024-03-02T17:49:48.163373Z","shell.execute_reply.started":"2024-03-02T17:49:47.535513Z","shell.execute_reply":"2024-03-02T17:49:48.162389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm \"/kaggle/working/best_model.h5\"","metadata":{"execution":{"iopub.status.busy":"2024-03-02T17:50:37.539348Z","iopub.execute_input":"2024-03-02T17:50:37.540018Z","iopub.status.idle":"2024-03-02T17:50:38.777000Z","shell.execute_reply.started":"2024-03-02T17:50:37.539986Z","shell.execute_reply":"2024-03-02T17:50:38.775672Z"},"trusted":true},"execution_count":null,"outputs":[]}]}