{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-18T15:56:00.984561Z","iopub.execute_input":"2021-08-18T15:56:00.985134Z","iopub.status.idle":"2021-08-18T15:56:00.998069Z","shell.execute_reply.started":"2021-08-18T15:56:00.985031Z","shell.execute_reply":"2021-08-18T15:56:00.997311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_hub as hub\nimport sys\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!git clone --depth 1 -b v2.3.0 https://github.com/tensorflow/models.git","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -Uqr models/official/requirements.txt","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nprint(tf.version.VERSION)\n!pip install -q tensorflow==2.3.0\nsys.path.append('models')\nfrom official.nlp.data import classifier_data_lib\nfrom official.nlp.bert import tokenization\nfrom official.nlp import optimization\nfrom sklearn.model_selection import train_test_split","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\")\ntrain_df,remaining=train_test_split(df,random_state=42,train_size=0.0075,stratify=df.target.values)\ntest_df=pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/test.csv\")\n \nvalid_df=train_test_split(remaining,random_state=42,train_size=0.00075,stratify=remaining.target.values)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with tf.device('/cpu:0'):\n    train_data=tf.data.Dataset.from_tensor_slices({train_df['questions_text'].values,train_df['target'].values})\n    valid_data=tf.data.Dataset.from_tensor_slices({valid_df['questions_text'].values,valid_df['target'].values})\n    test_data=tf.data.Dataset.from_tensor_slices({test_df['questions_text'].values,test_df['target'].values})\n    for text,label in train_data.take(1):\n      print(text)\n      print(label)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_list=[0,1]\n max_seq_length=128\n train_batch_size=32","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bert_layer=hub.KerasLayer(\"https://tfhub.dev/tensorflow/bert_en_uncased_L-12_H-768_A-12/2\",trainable=True)\nvocab_file = bert_layer.resolved_object.vocab_file.asset_path.numpy()\ndo_lower_case = bert_layer.resolved_object.do_lower_case.numpy()\ntokenizer = tokenization.FullTokenizer(vocab_file, do_lower_case)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef to_feature(text, label, label_list=label_list, max_seq_length=max_seq_length, tokenizer=tokenizer):\n  example=classifier_data.lib.InputExample(guid=None,text_a=text.numpy(),text_b=None,label=label.numpy())\n  feature=classifier_data_lib.convert_single-example(0,example,label_list,max_seq_length,tokenizer)\n  return (feature.input_ids,feature.input_mask,feature.segment_ids,feature.label_id)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_feature_map(text, label):\n  input_ids, input_mask, segment_ids, label_id=tf.py_function(to_feature,inp=[text,label],Tout=[tf.int32, tf.int32 ,tf.int32, tf.int32])\n  input_ids.set_shape([max_seq_length])\n  input-mask.set_shape([max_seq_length])\n  segment_ids.set_shape([max_seq_length])\n  x={\n      'input_word_ids': input_ids,\n      'input_mask': input_mask,\n      'input_type_ids': segment_ids\n  }\n  return (x,label_id)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with tf.device('/cpu:0'):\n  # train\n  train_data=(train_data.map(to_feature_map,num_parallel_calls=tf.data.experimental.AUTOTUNE).shuffle(1000).batch(32,drop-remainder=True).prefetch(tf.data.experiment.AUTOTUNE))\n\n  # valid\n  valid_data=(valid_data.map(to_feature_map,num_parallel_calls=tf.data.experimental.AUTOTUNE).shuffle(1000).batch(32,drop-remainder=True).prefetch(tf.data.experiment.AUTOTUNE))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.element_spec","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_data.element_spec","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model():\n  input_word_ids = tf.keras.layers.Input(shape=(max_seq_length,), dtype=tf.int32,\n                                       name=\"input_word_ids\")\n input_mask = tf.keras.layers.Input(shape=(max_seq_length,), dtype=tf.int32,\n                                   name=\"input_mask\")\n segment_ids = tf.keras.layers.Input(shape=(max_seq_length,), dtype=tf.int32,\n                                    name=\"segment_ids\")\n bert_layer = hub.KerasLayer(\"https://tfhub.dev/tensorflow/bert_en_uncased_L-12_H-768_A-12/2\",\n                            trainable=True)\n pooled_output, sequence_output = bert_layer([input_word_ids, input_mask, segment_ids])\n drop=tf.keras.layers.Dropout(0.4)(pooled_output)\n output=tf.keras.layers.Dense(1,activation='sigmoid'name='output')(drop)\n model=tf.keras.Model(inputs={\n    'input_word_ids': input_word_ids,\n    'input_mask': input_mask,\n    'input_type_ids': input_type_ids    \n\n\n },\n outputs=output)\n  return model","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model=create_model()\nmodel.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=2e-5),loss='binary_crossentropy',metrics=[tf.keras.metrics.BinaryAccuracy()])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train model\nepochs=4\nhistory=model.fit(train_data,validation_data=valid_data,epochs=epochs,verbose=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(model=model,show_shapes=True,dpi=76)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs=4\nhistory=model.fit(train_data,validation_data=valid_data,epochs=epochs,verbose=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndef plot_graphs(history, metric):\n  plt.plot(history.history[metric])\n  plt.plot(history.history['val_'+metric], '')\n  plt.xlabel(\"Epochs\")\n  plt.ylabel(metric)\n  plt.legend([metric, 'val_'+metric])\n  plt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\npreds=model.predict(test_data)\nthreshold=#between 0 and 1\n['Insinciere' if pred>=threshold else 'Sincer']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit=pd.DataFrame()\nsubmit[\"qid\"]=test_data.qid\nsubmit[\"prediction\"]=preds\nsubmit.to_csv(\"submission.csv\",index=False)","metadata":{},"execution_count":null,"outputs":[]}]}