{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"},{"sourceType":"datasetVersion","sourceId":11650,"datasetId":8327}],"dockerImageVersionId":25160,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.\nfrom keras.models import Sequential\nfrom keras.layers import LSTM,Dense,Dropout,Embedding,CuDNNLSTM,Bidirectional\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-08-12T11:40:56.523960Z","iopub.execute_input":"2024-08-12T11:40:56.524240Z","iopub.status.idle":"2024-08-12T11:40:58.189108Z","shell.execute_reply.started":"2024-08-12T11:40:56.524198Z","shell.execute_reply":"2024-08-12T11:40:58.188054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Reading the dataset","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2024-08-12T11:41:18.152312Z","iopub.execute_input":"2024-08-12T11:41:18.152630Z","iopub.status.idle":"2024-08-12T11:41:22.179226Z","shell.execute_reply.started":"2024-08-12T11:41:18.152570Z","shell.execute_reply":"2024-08-12T11:41:22.178482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-12T11:41:22.180441Z","iopub.execute_input":"2024-08-12T11:41:22.180680Z","iopub.status.idle":"2024-08-12T11:41:22.207481Z","shell.execute_reply.started":"2024-08-12T11:41:22.180640Z","shell.execute_reply":"2024-08-12T11:41:22.206791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Displaying the count of each class in Y label**","metadata":{}},{"cell_type":"code","source":"print(df['target'].value_counts())\nsns.countplot(df['target'])","metadata":{"execution":{"iopub.status.busy":"2024-08-12T11:41:22.654116Z","iopub.execute_input":"2024-08-12T11:41:22.654402Z","iopub.status.idle":"2024-08-12T11:41:22.999727Z","shell.execute_reply.started":"2024-08-12T11:41:22.654361Z","shell.execute_reply":"2024-08-12T11:41:22.998698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = df['question_text']\ny = df['target']","metadata":{"execution":{"iopub.status.busy":"2024-08-12T11:41:24.979878Z","iopub.execute_input":"2024-08-12T11:41:24.980164Z","iopub.status.idle":"2024-08-12T11:41:24.984400Z","shell.execute_reply.started":"2024-08-12T11:41:24.980119Z","shell.execute_reply":"2024-08-12T11:41:24.983420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"token = Tokenizer()","metadata":{"execution":{"iopub.status.busy":"2024-08-12T11:41:26.686903Z","iopub.execute_input":"2024-08-12T11:41:26.687277Z","iopub.status.idle":"2024-08-12T11:41:26.691344Z","shell.execute_reply.started":"2024-08-12T11:41:26.687219Z","shell.execute_reply":"2024-08-12T11:41:26.690329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Converting the text into sequence for processing in LSTM Layers**","metadata":{}},{"cell_type":"code","source":"token.fit_on_texts(x)\nseq = token.texts_to_sequences(x)","metadata":{"execution":{"iopub.status.busy":"2024-08-12T11:41:29.182639Z","iopub.execute_input":"2024-08-12T11:41:29.182912Z","iopub.status.idle":"2024-08-12T11:42:36.945698Z","shell.execute_reply.started":"2024-08-12T11:41:29.182869Z","shell.execute_reply":"2024-08-12T11:42:36.945038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seq[1:5] ","metadata":{"execution":{"iopub.status.busy":"2024-08-12T11:42:36.947071Z","iopub.execute_input":"2024-08-12T11:42:36.947301Z","iopub.status.idle":"2024-08-12T11:42:36.953114Z","shell.execute_reply.started":"2024-08-12T11:42:36.947255Z","shell.execute_reply":"2024-08-12T11:42:36.952473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pad_seq = pad_sequences(seq,maxlen=300)","metadata":{"execution":{"iopub.status.busy":"2024-08-12T11:42:36.954232Z","iopub.execute_input":"2024-08-12T11:42:36.954460Z","iopub.status.idle":"2024-08-12T11:42:48.769765Z","shell.execute_reply.started":"2024-08-12T11:42:36.954398Z","shell.execute_reply":"2024-08-12T11:42:48.768972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pad_seq[1:5]","metadata":{"execution":{"iopub.status.busy":"2024-08-12T11:42:48.771758Z","iopub.execute_input":"2024-08-12T11:42:48.771990Z","iopub.status.idle":"2024-08-12T11:42:48.778907Z","shell.execute_reply.started":"2024-08-12T11:42:48.771951Z","shell.execute_reply":"2024-08-12T11:42:48.778257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_size = len(token.word_index)+1","metadata":{"execution":{"iopub.status.busy":"2024-08-12T11:42:48.780154Z","iopub.execute_input":"2024-08-12T11:42:48.780406Z","iopub.status.idle":"2024-08-12T11:42:48.789046Z","shell.execute_reply.started":"2024-08-12T11:42:48.780346Z","shell.execute_reply":"2024-08-12T11:42:48.788342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_size","metadata":{"execution":{"iopub.status.busy":"2024-08-12T11:42:48.790309Z","iopub.execute_input":"2024-08-12T11:42:48.790557Z","iopub.status.idle":"2024-08-12T11:42:48.800129Z","shell.execute_reply.started":"2024-08-12T11:42:48.790505Z","shell.execute_reply":"2024-08-12T11:42:48.799467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = df['question_text']\ny = df['target']","metadata":{"execution":{"iopub.status.busy":"2024-08-12T11:42:48.801505Z","iopub.execute_input":"2024-08-12T11:42:48.801804Z","iopub.status.idle":"2024-08-12T11:42:48.808595Z","shell.execute_reply.started":"2024-08-12T11:42:48.801747Z","shell.execute_reply":"2024-08-12T11:42:48.808072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Using word embeddings so that words with similar words have similar representation in vector space. It represents every word as a vector. The words which have similar meaning are place close to each other.**","metadata":{}},{"cell_type":"markdown","source":"If the file is missing, download the required GloVe file from the official website (http://nlp.stanford.edu/data/glove.840B.300d.zip).","metadata":{}},{"cell_type":"code","source":"from tqdm import tqdm\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2024-08-12T11:42:48.810012Z","iopub.execute_input":"2024-08-12T11:42:48.810313Z","iopub.status.idle":"2024-08-12T11:42:48.817797Z","shell.execute_reply.started":"2024-08-12T11:42:48.810259Z","shell.execute_reply":"2024-08-12T11:42:48.817223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_vector = {}\nf = open('/kaggle/input/glove840b300dtxt/glove.840B.300d.txt')\nfor line in tqdm(f):\n    value = line.split(' ')\n    word = value[0]\n    coef = np.array(value[1:],dtype = 'float32')\n    embedding_vector[word] = coef\n    ","metadata":{"execution":{"iopub.status.busy":"2024-08-12T11:42:48.818759Z","iopub.execute_input":"2024-08-12T11:42:48.818981Z","iopub.status.idle":"2024-08-12T11:47:35.770988Z","shell.execute_reply.started":"2024-08-12T11:42:48.818931Z","shell.execute_reply":"2024-08-12T11:47:35.770342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Converting the words in our Vocabulary to their corresponding embeddings and placing them in a matrix.**","metadata":{}},{"cell_type":"code","source":"embedding_matrix = np.zeros((vocab_size,300))\nfor word,i in tqdm(token.word_index.items()):\n    embedding_value = embedding_vector.get(word)\n    if embedding_value is not None:\n        embedding_matrix[i] = embedding_value","metadata":{"execution":{"iopub.status.busy":"2024-08-12T11:47:35.772252Z","iopub.execute_input":"2024-08-12T11:47:35.772498Z","iopub.status.idle":"2024-08-12T11:47:36.611278Z","shell.execute_reply.started":"2024-08-12T11:47:35.772451Z","shell.execute_reply":"2024-08-12T11:47:36.610605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_matrix","metadata":{"execution":{"iopub.status.busy":"2024-08-12T11:50:07.725956Z","iopub.execute_input":"2024-08-12T11:50:07.726271Z","iopub.status.idle":"2024-08-12T11:50:07.732409Z","shell.execute_reply.started":"2024-08-12T11:50:07.726226Z","shell.execute_reply":"2024-08-12T11:50:07.731557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Building a LSTM model. LSTM networks are useful in sequence data as they are capable of remembering the past words which help them in understanding the meaning of the sentence which helps in text classification. \nBidirectional Layer is helpful as it helps in understanding thesentence from start to end and also from end to start. It works in both the direction. This is useful as the reverse order LSTM layer is capable of learning patterns which are not possible for the normal LSTM layers which goes from start to end of the sentence in the normal order. Hence Bidirectional layers are useful in text classification problems as different patterns can be captured from 2 directions.**","metadata":{}},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Embedding(vocab_size,300,weights = [embedding_matrix],input_length=300,trainable = False))\nmodel.add(Bidirectional(CuDNNLSTM(75)))\nmodel.add(Dense(32,activation = 'relu'))\nmodel.add(Dense(1,activation = 'sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2024-08-12T11:50:14.028414Z","iopub.execute_input":"2024-08-12T11:50:14.028677Z","iopub.status.idle":"2024-08-12T11:50:19.279100Z","shell.execute_reply.started":"2024-08-12T11:50:14.028635Z","shell.execute_reply":"2024-08-12T11:50:19.278389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='adam',loss='binary_crossentropy',metrics = ['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-08-12T11:50:24.016128Z","iopub.execute_input":"2024-08-12T11:50:24.016418Z","iopub.status.idle":"2024-08-12T11:50:24.068732Z","shell.execute_reply.started":"2024-08-12T11:50:24.016371Z","shell.execute_reply":"2024-08-12T11:50:24.067980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(pad_seq,y,epochs = 5,batch_size=256,validation_split=0.2)","metadata":{"execution":{"iopub.status.busy":"2024-08-12T11:50:25.988002Z","iopub.execute_input":"2024-08-12T11:50:25.988307Z","iopub.status.idle":"2024-08-12T12:11:07.522828Z","shell.execute_reply.started":"2024-08-12T11:50:25.988254Z","shell.execute_reply":"2024-08-12T12:11:07.521896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values = history.history","metadata":{"execution":{"iopub.status.busy":"2024-08-12T12:11:07.524105Z","iopub.execute_input":"2024-08-12T12:11:07.524396Z","iopub.status.idle":"2024-08-12T12:11:07.528114Z","shell.execute_reply.started":"2024-08-12T12:11:07.524349Z","shell.execute_reply":"2024-08-12T12:11:07.527105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_loss = values['val_loss']\ntraining_loss = values['loss']\ntraining_acc = values['acc']\nvalidation_acc = values['val_acc']\nepochs = range(5)\n\nplt.plot(epochs,val_loss,label = 'Validation Loss')\nplt.plot(epochs,training_loss,label = 'Training Loss')\nplt.title('Epochs vs Loss')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-12T12:11:07.529452Z","iopub.execute_input":"2024-08-12T12:11:07.529744Z","iopub.status.idle":"2024-08-12T12:11:07.832104Z","shell.execute_reply.started":"2024-08-12T12:11:07.529686Z","shell.execute_reply":"2024-08-12T12:11:07.830912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(epochs,validation_acc,label = 'Validation Accuracy')\nplt.plot(epochs,training_acc,label = 'Training Accuracy')\nplt.title('Epochs vs Accuracy')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-12T12:11:07.833984Z","iopub.execute_input":"2024-08-12T12:11:07.834687Z","iopub.status.idle":"2024-08-12T12:11:08.141925Z","shell.execute_reply.started":"2024-08-12T12:11:07.834615Z","shell.execute_reply":"2024-08-12T12:11:08.140925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testing = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2024-08-12T12:11:41.477342Z","iopub.execute_input":"2024-08-12T12:11:41.477627Z","iopub.status.idle":"2024-08-12T12:11:42.490953Z","shell.execute_reply.started":"2024-08-12T12:11:41.477584Z","shell.execute_reply":"2024-08-12T12:11:42.490331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testing.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-12T12:11:43.788727Z","iopub.execute_input":"2024-08-12T12:11:43.789005Z","iopub.status.idle":"2024-08-12T12:11:43.802747Z","shell.execute_reply.started":"2024-08-12T12:11:43.788962Z","shell.execute_reply":"2024-08-12T12:11:43.801753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = testing['question_text']","metadata":{"execution":{"iopub.status.busy":"2024-08-12T12:11:47.033682Z","iopub.execute_input":"2024-08-12T12:11:47.033996Z","iopub.status.idle":"2024-08-12T12:11:47.037864Z","shell.execute_reply.started":"2024-08-12T12:11:47.033936Z","shell.execute_reply":"2024-08-12T12:11:47.036976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = token.texts_to_sequences(x_test)","metadata":{"execution":{"iopub.status.busy":"2024-08-12T12:11:49.260949Z","iopub.execute_input":"2024-08-12T12:11:49.261281Z","iopub.status.idle":"2024-08-12T12:11:58.556468Z","shell.execute_reply.started":"2024-08-12T12:11:49.261228Z","shell.execute_reply":"2024-08-12T12:11:58.555682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testing_seq = pad_sequences(x_test,maxlen=300)","metadata":{"execution":{"iopub.status.busy":"2024-08-12T12:11:58.557814Z","iopub.execute_input":"2024-08-12T12:11:58.558140Z","iopub.status.idle":"2024-08-12T12:12:01.959724Z","shell.execute_reply.started":"2024-08-12T12:11:58.558081Z","shell.execute_reply":"2024-08-12T12:12:01.959087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict = model.predict_classes(testing_seq)","metadata":{"execution":{"iopub.status.busy":"2024-08-12T12:12:01.961115Z","iopub.execute_input":"2024-08-12T12:12:01.961374Z","iopub.status.idle":"2024-08-12T12:14:15.334486Z","shell.execute_reply.started":"2024-08-12T12:12:01.961323Z","shell.execute_reply":"2024-08-12T12:14:15.333657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testing['label'] = predict","metadata":{"execution":{"iopub.status.busy":"2024-08-12T12:14:15.335729Z","iopub.execute_input":"2024-08-12T12:14:15.335957Z","iopub.status.idle":"2024-08-12T12:14:15.340203Z","shell.execute_reply.started":"2024-08-12T12:14:15.335916Z","shell.execute_reply":"2024-08-12T12:14:15.339478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testing.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-12T12:14:15.341630Z","iopub.execute_input":"2024-08-12T12:14:15.341901Z","iopub.status.idle":"2024-08-12T12:14:15.361746Z","shell.execute_reply.started":"2024-08-12T12:14:15.341846Z","shell.execute_reply":"2024-08-12T12:14:15.361136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit_df = pd.DataFrame({\"qid\": testing[\"qid\"], \"prediction\": testing['label']})\nsubmit_df.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-08-12T12:14:15.362774Z","iopub.execute_input":"2024-08-12T12:14:15.363028Z","iopub.status.idle":"2024-08-12T12:14:16.951920Z","shell.execute_reply.started":"2024-08-12T12:14:15.362968Z","shell.execute_reply":"2024-08-12T12:14:16.951292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\n# Predict class probabilities\n#predictions = model.predict(testing_seq)\n\n# Convert probabilities to class labels\npredicted_classes = np.where(predict > 0.5, 1, 0)  # Since you have a sigmoid activation for binary classification\n","metadata":{"execution":{"iopub.status.busy":"2024-08-12T12:14:55.449442Z","iopub.execute_input":"2024-08-12T12:14:55.449742Z","iopub.status.idle":"2024-08-12T12:14:55.455106Z","shell.execute_reply.started":"2024-08-12T12:14:55.449699Z","shell.execute_reply":"2024-08-12T12:14:55.454132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}