{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nfrom nltk.corpus import stopwords\nfrom nltk.util import ngrams\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom collections import defaultdict\nfrom collections import  Counter\nplt.style.use('ggplot')\nstop=set(stopwords.words('english'))\nimport re\nfrom nltk.tokenize import word_tokenize\nimport gensim\nimport string\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom tqdm import tqdm\nfrom keras.models import Sequential\nfrom keras.layers import Embedding,LSTM,Dense,SpatialDropout1D\nfrom keras.initializers import Constant\nfrom sklearn.model_selection import train_test_split\n# from keras.optimizers import Adam\nimport os ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-12T13:45:25.882379Z","iopub.execute_input":"2022-08-12T13:45:25.882865Z","iopub.status.idle":"2022-08-12T13:45:25.897637Z","shell.execute_reply.started":"2022-08-12T13:45:25.882827Z","shell.execute_reply":"2022-08-12T13:45:25.896210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(action  = 'ignore')\n\n%config Completer.use_jedi = False","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:45:25.902509Z","iopub.execute_input":"2022-08-12T13:45:25.903705Z","iopub.status.idle":"2022-08-12T13:45:25.922034Z","shell.execute_reply.started":"2022-08-12T13:45:25.903656Z","shell.execute_reply":"2022-08-12T13:45:25.920763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/nlp-getting-started/train.csv')\ndf_test = pd.read_csv('../input/nlp-getting-started/test.csv')\ndf_sample = pd.read_csv('../input/nlp-getting-started/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:45:25.926542Z","iopub.execute_input":"2022-08-12T13:45:25.927035Z","iopub.status.idle":"2022-08-12T13:45:25.975847Z","shell.execute_reply.started":"2022-08-12T13:45:25.927007Z","shell.execute_reply":"2022-08-12T13:45:25.974607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:45:25.978770Z","iopub.execute_input":"2022-08-12T13:45:25.979553Z","iopub.status.idle":"2022-08-12T13:45:25.994752Z","shell.execute_reply.started":"2022-08-12T13:45:25.979510Z","shell.execute_reply":"2022-08-12T13:45:25.993319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"fig,(ax1,ax2)=plt.subplots(1,2,figsize=(10,5))\nlen_tweets = df_train[df_train['target']==1]['text'].str.len()\nsns.distplot(len_tweets, ax = ax1)\nn_len_tweets = df_train[df_train['target']==0]['text'].str.len()\nsns.distplot(n_len_tweets, ax = ax2)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:45:25.997232Z","iopub.execute_input":"2022-08-12T13:45:25.998311Z","iopub.status.idle":"2022-08-12T13:45:26.548553Z","shell.execute_reply.started":"2022-08-12T13:45:25.998264Z","shell.execute_reply":"2022-08-12T13:45:26.547001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Cleaning","metadata":{}},{"cell_type":"code","source":"target = df_train.target\n\ncdata = pd.concat([df_train, df_test])","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:45:26.552193Z","iopub.execute_input":"2022-08-12T13:45:26.553569Z","iopub.status.idle":"2022-08-12T13:45:26.566222Z","shell.execute_reply.started":"2022-08-12T13:45:26.553521Z","shell.execute_reply":"2022-08-12T13:45:26.564513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:45:26.570222Z","iopub.execute_input":"2022-08-12T13:45:26.571641Z","iopub.status.idle":"2022-08-12T13:45:26.581953Z","shell.execute_reply.started":"2022-08-12T13:45:26.571581Z","shell.execute_reply":"2022-08-12T13:45:26.579918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import emoji\n\ndef text_preproccessing(df):  \n    \n    df = df.copy()\n    \n    def remove_URL(text):\n        url = re.compile(r'https?://\\S+|www\\.\\S+')\n        return url.sub(r'',text)\n\n    def remove_html(text):\n        html=re.compile(r'<.*?>')\n        return html.sub(r'',text)\n\n    def remove_punct(text):\n        table=str.maketrans('','',string.punctuation)\n        return text.translate(table)\n\n    # Reference : https://gist.github.com/slowkow/7a7f61f495e3dbb7e3d767f97bd7304b\n    def remove_emoji(text):\n        emoji_pattern = re.compile(\"[\"\n                               u\"\\U0001F600-\\U0001F64F\"  # emoticons\n                               u\"\\U0001F300-\\U0001F5FF\"  # symbols & pictographs\n                               u\"\\U0001F680-\\U0001F6FF\"  # transport & map symbols\n                               u\"\\U0001F1E0-\\U0001F1FF\"  # flags (iOS)\n                               u\"\\U00002702-\\U000027B0\"\n                               u\"\\U000024C2-\\U0001F251\"\n                               \"]+\", flags=re.UNICODE)\n        return emoji_pattern.sub(r'', text)\n\n    df['text']=df['text'].apply(lambda x : remove_URL(x))\n    df['text']=df['text'].apply(lambda x : remove_html(x))\n    df['text']=df['text'].apply(lambda x : remove_punct(x))\n    # cdata['text']=cdata['text'].apply(lambda x : remove_emoji(x))()\n\n\n    df['text'] = df['text'].apply(lambda x : emoji.demojize(x))\n    \n    return df\n\ncdata = text_preproccessing(cdata)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:45:26.584015Z","iopub.execute_input":"2022-08-12T13:45:26.584904Z","iopub.status.idle":"2022-08-12T13:45:26.994700Z","shell.execute_reply.started":"2022-08-12T13:45:26.584858Z","shell.execute_reply":"2022-08-12T13:45:26.993286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import AutoTokenizer, TFBertModel\ntokenizer = AutoTokenizer.from_pretrained('bert-base-uncased')\nbert = TFBertModel.from_pretrained('bert-base-uncased')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:45:26.996247Z","iopub.execute_input":"2022-08-12T13:45:26.996635Z","iopub.status.idle":"2022-08-12T13:45:33.254271Z","shell.execute_reply.started":"2022-08-12T13:45:26.996593Z","shell.execute_reply":"2022-08-12T13:45:33.252741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_length = max([len(sent.split()) for sent in cdata.text ])\nprint(max_length)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:45:33.256671Z","iopub.execute_input":"2022-08-12T13:45:33.257454Z","iopub.status.idle":"2022-08-12T13:45:33.308943Z","shell.execute_reply.started":"2022-08-12T13:45:33.257411Z","shell.execute_reply":"2022-08-12T13:45:33.306342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = cdata.iloc[:7613,:]\ndf_text = cdata.iloc[7613:,:]","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:45:33.315306Z","iopub.execute_input":"2022-08-12T13:45:33.316676Z","iopub.status.idle":"2022-08-12T13:45:33.324551Z","shell.execute_reply.started":"2022-08-12T13:45:33.316633Z","shell.execute_reply":"2022-08-12T13:45:33.323473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = tokenizer(\ntext = df_train.text.tolist(),\n    add_special_tokens = True,\n    max_length = 34,\n    truncation = True,\n    padding = True,\n    return_tensors = 'tf',\n    return_token_type_ids = False,\n    return_attention_mask = True,\n    verbose = True\n    \n)\n\ntarget = df_train.target.values","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:45:33.328383Z","iopub.execute_input":"2022-08-12T13:45:33.330021Z","iopub.status.idle":"2022-08-12T13:45:34.368884Z","shell.execute_reply.started":"2022-08-12T13:45:33.329976Z","shell.execute_reply":"2022-08-12T13:45:34.367597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.initializers import TruncatedNormal\nfrom tensorflow.keras.losses import CategoricalCrossentropy,BinaryCrossentropy\nfrom tensorflow.keras.metrics import CategoricalAccuracy,BinaryAccuracy\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.utils import plot_model","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:45:34.370743Z","iopub.execute_input":"2022-08-12T13:45:34.372521Z","iopub.status.idle":"2022-08-12T13:45:34.381643Z","shell.execute_reply.started":"2022-08-12T13:45:34.372475Z","shell.execute_reply":"2022-08-12T13:45:34.379979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_len = 34\nimport tensorflow as tf \nfrom tensorflow.keras.layers import Input, Dense \n\ninput_ids = Input(shape = (max_len,), dtype = tf.int32, name = 'input_ids')\ninput_mask = Input(shape = (max_len,), dtype = tf.int32, name = 'input_mask')\n\nembeddings = bert([input_ids, input_mask])[1]\n\nout = tf.keras.layers.Dropout(0.1)(embeddings)\n\nout = Dense(128, activation='relu')(out)\nout = tf.keras.layers.Dropout(0.1)(out)\nout = Dense(32,activation = 'relu')(out)\n\ny = Dense(1,activation = 'sigmoid')(out)\n    \nmodel = tf.keras.Model(inputs=[input_ids, input_mask], outputs=y)\nmodel.layers[2].trainable = True","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:45:34.384022Z","iopub.execute_input":"2022-08-12T13:45:34.384625Z","iopub.status.idle":"2022-08-12T13:45:36.261599Z","shell.execute_reply.started":"2022-08-12T13:45:34.384582Z","shell.execute_reply":"2022-08-12T13:45:36.260237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optimizer = Adam(\n    learning_rate=6e-06, # this learning rate is for bert model , taken from huggingface website \n    epsilon=1e-08,\n    decay=0.01,\n    clipnorm=1.0)\n\n# Set loss and metrics\nloss = BinaryCrossentropy(from_logits = True)\nmetric = BinaryAccuracy('accuracy'),\n# Compile the model\nmodel.compile(\n    optimizer = optimizer,\n    loss = loss, \n    metrics = metric)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:45:36.263717Z","iopub.execute_input":"2022-08-12T13:45:36.264233Z","iopub.status.idle":"2022-08-12T13:45:36.287904Z","shell.execute_reply.started":"2022-08-12T13:45:36.264173Z","shell.execute_reply":"2022-08-12T13:45:36.286538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_history = model.fit(\n    x ={'input_ids':x_train['input_ids'],'input_mask':x_train['attention_mask']} ,\n    y = df_train.target,\n#     validation_split = 0.1,\n  epochs=6,\n    batch_size=10\n)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T13:45:36.289518Z","iopub.execute_input":"2022-08-12T13:45:36.290275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = tokenizer(\n    text=df_test.text.tolist(),\n    add_special_tokens=True,\n    max_length=34,\n    truncation=True,\n    padding=True, \n    return_tensors='tf',\n    return_token_type_ids = False,\n    return_attention_mask = True,\n    verbose = True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test['attention_mask']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted = model.predict({'input_ids':x_test['input_ids'],'input_mask':x_test['attention_mask']})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_predicted = np.where(predicted>0.5,1,0)\ny_predicted = y_predicted.reshape((1,3263))[0]\ndf_sample['id'] = df_test.id\ndf_sample['target'] = y_predicted\ndf_sample.to_csv('submission.csv',index = False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}