{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport zipfile \nimport gensim\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom tensorflow.keras.layers import Dense, Embedding, GRU, LSTM,Dropout,Input,Bidirectional,GlobalMaxPool1D,Reshape,Conv1D,Reshape\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import EarlyStopping\nimport tensorflow_addons as tfa\nimport tensorflow as tf\ntrn = pd.read_csv('/kaggle/input/nlp-getting-started/train.csv')\ntst = pd.read_csv('/kaggle/input/nlp-getting-started/test.csv')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-15T19:51:25.809884Z","iopub.execute_input":"2021-06-15T19:51:25.810269Z","iopub.status.idle":"2021-06-15T19:51:31.544385Z","shell.execute_reply.started":"2021-06-15T19:51:25.810169Z","shell.execute_reply":"2021-06-15T19:51:31.543544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number of Positive Cases:',trn.target.sum())\nprint('Number of Negative Cases:',trn.shape[0]-trn.target.sum())","metadata":{"execution":{"iopub.status.busy":"2021-06-15T19:51:31.545782Z","iopub.execute_input":"2021-06-15T19:51:31.546173Z","iopub.status.idle":"2021-06-15T19:51:31.560756Z","shell.execute_reply.started":"2021-06-15T19:51:31.546103Z","shell.execute_reply":"2021-06-15T19:51:31.559661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_chars = trn.text.str.len()\nnum_words = trn.text.str.split().str.len()\nfig,axs = plt.subplots(1,2)\naxs[0].hist(num_words,edgecolor = 'black',bins = 10)\naxs[0].set_xlabel('Number of Words per Case')\naxs[0].set_ylabel('Frequency')\n\naxs[1].hist(num_chars,edgecolor = 'black',bins = 25)\naxs[1].set_xlabel('Number of Words per Case')\naxs[1].set_ylabel('Frequency')\nplt.subplots_adjust(bottom=0, left = 0,right=1.5, top=0.66,wspace = 0.25 );\n","metadata":{"execution":{"iopub.status.busy":"2021-06-15T19:51:31.564043Z","iopub.execute_input":"2021-06-15T19:51:31.564325Z","iopub.status.idle":"2021-06-15T19:51:32.055534Z","shell.execute_reply.started":"2021-06-15T19:51:31.564301Z","shell.execute_reply":"2021-06-15T19:51:32.054755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Tokenize Things","metadata":{}},{"cell_type":"code","source":"X = trn.text\ny = trn.target\nX_trn,X_hld,y_trn,y_hld = train_test_split(X,y,test_size = 0.1,random_state = 123123)\nX_tst = tst.text","metadata":{"execution":{"iopub.status.busy":"2021-06-15T19:51:32.057249Z","iopub.execute_input":"2021-06-15T19:51:32.057692Z","iopub.status.idle":"2021-06-15T19:51:32.068265Z","shell.execute_reply.started":"2021-06-15T19:51:32.057650Z","shell.execute_reply":"2021-06-15T19:51:32.067247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_vocab_size = 10000\ntokenized_vocab = Tokenizer(num_words = max_vocab_size)\ntokenized_vocab.fit_on_texts(X_trn)\ntokenized_vocab_trn = tokenized_vocab.texts_to_sequences(X_trn)\ntokenized_vocab_hld = tokenized_vocab.texts_to_sequences(X_hld)\ntokenized_vocab_tst = tokenized_vocab.texts_to_sequences(X_tst)\nV = len(tokenized_vocab.word_index)\nmax_seq_len = 145\nprint(f'number of unique tokens is {V}')\n\ntrn_padded = pad_sequences(tokenized_vocab_trn,maxlen = max_seq_len,padding = 'post')\nhld_padded = pad_sequences(tokenized_vocab_hld,maxlen = max_seq_len,padding = 'post')\ntst_padded = pad_sequences(tokenized_vocab_tst,maxlen = max_seq_len,padding = 'post')","metadata":{"execution":{"iopub.status.busy":"2021-06-15T19:51:32.069914Z","iopub.execute_input":"2021-06-15T19:51:32.070324Z","iopub.status.idle":"2021-06-15T19:51:32.502462Z","shell.execute_reply.started":"2021-06-15T19:51:32.070286Z","shell.execute_reply":"2021-06-15T19:51:32.501561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load Embedding Matrix","metadata":{}},{"cell_type":"code","source":"archive = zipfile.ZipFile('/kaggle/input/quora-insincere-questions-classification/embeddings.zip', 'r') \narchive.namelist()\nnews_path=archive.open('GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin', 'r')\nword2vec_embeddings = gensim.models.KeyedVectors.load_word2vec_format(news_path, binary=True)\n\nnb_words = V+1\nembed_dim = 300\nembedding_matrix = np.zeros((nb_words, embed_dim))\nword_index = tokenized_vocab.word_index\nfor word, i in word_index.items():\n    if word in word2vec_embeddings.key_to_index:\n        embedding_matrix[i] = word2vec_embeddings.get_vector(word)","metadata":{"execution":{"iopub.status.busy":"2021-06-15T19:51:32.503893Z","iopub.execute_input":"2021-06-15T19:51:32.504387Z","iopub.status.idle":"2021-06-15T19:52:48.101864Z","shell.execute_reply.started":"2021-06-15T19:51:32.504347Z","shell.execute_reply":"2021-06-15T19:52:48.101037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Metrics to Measure","metadata":{}},{"cell_type":"code","source":"model_metrics = [tf.keras.metrics.AUC(),\n                 tf.keras.metrics.BinaryAccuracy(),\n                 tfa.metrics.F1Score(num_classes=1, average='macro',threshold=0.5)]","metadata":{"execution":{"iopub.status.busy":"2021-06-15T19:52:48.109680Z","iopub.execute_input":"2021-06-15T19:52:48.110280Z","iopub.status.idle":"2021-06-15T19:52:50.277511Z","shell.execute_reply.started":"2021-06-15T19:52:48.110241Z","shell.execute_reply":"2021-06-15T19:52:50.276417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Test Set Generation","metadata":{}},{"cell_type":"code","source":"def get_tst_preds(model,output_name):\n    predictions = model.predict(tst_padded)\n    predictions=predictions.reshape(predictions.shape[0],)\n    final_predictions = pd.concat([tst.id,pd.Series(predictions)],axis = 1)\n    final_predictions = final_predictions.rename(columns = {'id':'id',0:'target'})\n    final_predictions.loc[final_predictions.target>0.5,'target'] = 1\n    final_predictions.loc[final_predictions.target<=0.5,'target'] = 0\n    final_predictions.prediction = final_predictions.target.astype('int')\n    final_predictions.to_csv(output_name,sep = ',',index = False)\n    return(final_predictions)","metadata":{"execution":{"iopub.status.busy":"2021-06-15T20:08:32.924933Z","iopub.execute_input":"2021-06-15T20:08:32.925283Z","iopub.status.idle":"2021-06-15T20:08:32.931880Z","shell.execute_reply.started":"2021-06-15T20:08:32.925251Z","shell.execute_reply":"2021-06-15T20:08:32.930584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Basic Model","metadata":{}},{"cell_type":"code","source":"input_layer = Input(shape = trn_padded.shape[1])\nembed_layer = Embedding(nb_words,embed_dim,weights = [embedding_matrix],input_length = max_seq_len,trainable = False) (input_layer)\nlstm_layer = Bidirectional(LSTM(50,dropout = 0.25,return_sequences = False))(embed_layer)\ndense_layer = Dense(1,activation = 'sigmoid') (lstm_layer)\n\nbuild_model = Model(inputs = input_layer, outputs = dense_layer)\nbuild_model.compile(optimizer = 'adam',loss = 'binary_crossentropy',metrics = model_metrics)\nbuild_model.fit(trn_padded,y_trn,batch_size = 128, validation_data = (hld_padded,y_hld),epochs = 10)","metadata":{"execution":{"iopub.status.busy":"2021-06-15T19:52:50.282144Z","iopub.execute_input":"2021-06-15T19:52:50.282711Z","iopub.status.idle":"2021-06-15T19:53:12.171183Z","shell.execute_reply.started":"2021-06-15T19:52:50.282672Z","shell.execute_reply":"2021-06-15T19:53:12.170289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,axs = plt.subplots(1,3,figsize=(20,5))\naxs[0].plot(build_model.history.history['val_f1_score'],label = 'Hold F1')\naxs[0].plot(build_model.history.history['f1_score'], label = 'Train F1')\n\naxs[1].plot(build_model.history.history['val_auc'],label = 'Hold AUC')\naxs[1].plot(build_model.history.history['auc'], label = 'Train AUC')\n\naxs[2].plot(build_model.history.history['val_loss'],label = 'Hold Loss')\naxs[2].plot(build_model.history.history['loss'], label = 'Train Loss')","metadata":{"execution":{"iopub.status.busy":"2021-06-15T19:53:12.172586Z","iopub.execute_input":"2021-06-15T19:53:12.172985Z","iopub.status.idle":"2021-06-15T19:53:12.678401Z","shell.execute_reply.started":"2021-06-15T19:53:12.172947Z","shell.execute_reply":"2021-06-15T19:53:12.677605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_preds = get_tst_preds(build_model,'submission.csv')","metadata":{"execution":{"iopub.status.busy":"2021-06-15T20:10:23.049208Z","iopub.execute_input":"2021-06-15T20:10:23.049539Z","iopub.status.idle":"2021-06-15T20:10:23.740200Z","shell.execute_reply.started":"2021-06-15T20:10:23.049508Z","shell.execute_reply":"2021-06-15T20:10:23.739078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-06-15T20:09:47.651824Z","iopub.execute_input":"2021-06-15T20:09:47.652137Z","iopub.status.idle":"2021-06-15T20:09:47.659229Z","shell.execute_reply.started":"2021-06-15T20:09:47.652106Z","shell.execute_reply":"2021-06-15T20:09:47.658210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"]","metadata":{"execution":{"iopub.status.busy":"2021-06-15T20:10:02.707842Z","iopub.execute_input":"2021-06-15T20:10:02.708174Z","iopub.status.idle":"2021-06-15T20:10:02.713551Z","shell.execute_reply.started":"2021-06-15T20:10:02.708142Z","shell.execute_reply":"2021-06-15T20:10:02.712721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}