{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"}],"dockerImageVersionId":30762,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd\nimport os\nimport gensim\nimport re\nimport string\nimport seaborn as sns\nfrom sklearn.metrics.pairwise import cosine_similarity,euclidean_distances\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import precision_recall_curve,log_loss,recall_score,precision_score,confusion_matrix\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\nfrom  tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.utils import pad_sequences\nfrom tensorflow.keras.layers import Dense, Input, Dropout, LSTM, Activation,Embedding, SimpleRNN, Bidirectional, BatchNormalization\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.metrics import F1Score\nfrom keras.optimizers import Adam\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nfrom tensorflow.keras.losses import CosineSimilarity\nimport nltk\nfrom nltk.corpus import stopwords,wordnet\nfrom nltk import pos_tag, word_tokenize\nfrom nltk.stem import WordNetLemmatizer,PorterStemmer,porter\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom nltk.corpus import sentiwordnet as swn\nfrom datasets import Dataset\nimport spacy\nnlp = spacy.load('en_core_web_sm')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-09T16:03:00.240738Z","iopub.execute_input":"2024-09-09T16:03:00.241121Z","iopub.status.idle":"2024-09-09T16:03:36.600481Z","shell.execute_reply.started":"2024-09-09T16:03:00.241082Z","shell.execute_reply":"2024-09-09T16:03:36.599236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-09-09T16:03:36.602793Z","iopub.execute_input":"2024-09-09T16:03:36.603736Z","iopub.status.idle":"2024-09-09T16:03:42.684947Z","shell.execute_reply.started":"2024-09-09T16:03:36.603685Z","shell.execute_reply":"2024-09-09T16:03:42.683900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape,test.shape","metadata":{"execution":{"iopub.status.busy":"2024-09-09T16:03:42.686141Z","iopub.execute_input":"2024-09-09T16:03:42.686516Z","iopub.status.idle":"2024-09-09T16:03:42.694335Z","shell.execute_reply.started":"2024-09-09T16:03:42.686477Z","shell.execute_reply":"2024-09-09T16:03:42.693217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_text(text):\n    text = str(text).lower()\n    text=re.sub(r\"http\\S+\", \"\", text)  ## remove website\n    # text=re.sub(r'[/(){}\\[\\]\\|@,;:\\.]','',text) ## remove brakcets and all\n    text=re.sub('[^A-Za-z0-9]+', ' ', text) ## Extract only alphanumeric\n    text=re.sub(r\"\\d{10}\", \"\",text)  ## Remove any number having 10 digits\n    text = re.sub(r'\\s+',' ',text) ## replace extra space with single space\n    return text.strip()\n\ndef clean_text_and_sw(text):\n    text = text.strip()\n    text = str(text).lower()\n    text=re.sub(r\"http\\S+\", \"\", text)\n    text=re.sub(r\"bit\\S+\", \"\", text)\n    text=re.sub(r'[/(){}\\[\\]\\|@,;:\\.]','',text)\n    text=re.sub('[^A-Za-z0-9]+', ' ', text)\n    text=re.sub(r\"\\d{10}\", \"\",text)\n    text = re.sub(r'\\s+','',text)\n    text = [w for w in text.split(' ') if w not in stop_words]\n    text= \" \".join(text)\n    return text.strip()\n\ndef lem(text):\n    # pos_tags = pos_tag(nltk.word_tokenize(text))\n    # text2 = [WordNetLemmatizer().lemmatize(text,pos_tags)]\n    doc = nlp(text)\n    lem_text = [token.lemma_ for token in doc]\n    return ' '.join(lem_text)\n\ndef cnt_wrds(text):\n    if not text:\n        return 0\n    return len(text.split(' '))\n","metadata":{"execution":{"iopub.status.busy":"2024-09-09T16:03:42.696795Z","iopub.execute_input":"2024-09-09T16:03:42.697204Z","iopub.status.idle":"2024-09-09T16:03:42.709337Z","shell.execute_reply.started":"2024-09-09T16:03:42.697158Z","shell.execute_reply":"2024-09-09T16:03:42.708142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['len'] = train['question_text'].apply(lambda x: len(x.split(' ')))\ntrain['len'].quantile([0.25,0.5,0.75,0.9,0.95,0.99,1])","metadata":{"execution":{"iopub.status.busy":"2024-09-09T16:03:42.710873Z","iopub.execute_input":"2024-09-09T16:03:42.711364Z","iopub.status.idle":"2024-09-09T16:03:44.942871Z","shell.execute_reply.started":"2024-09-09T16:03:42.711317Z","shell.execute_reply":"2024-09-09T16:03:44.941527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train[train['len'] < 36]\ntrain['clean_text'] = train['question_text'].apply(lambda x: clean_text(x))\ntest['clean_text'] = test['question_text'].apply(lambda x: clean_text(x))","metadata":{"execution":{"iopub.status.busy":"2024-09-09T16:03:44.944194Z","iopub.execute_input":"2024-09-09T16:03:44.944568Z","iopub.status.idle":"2024-09-09T16:04:20.752762Z","shell.execute_reply.started":"2024-09-09T16:03:44.944530Z","shell.execute_reply":"2024-09-09T16:04:20.751710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import zipfile\nzip_path = '/kaggle/input/quora-insincere-questions-classification/embeddings.zip'\nextract_to = '/kaggle/working/embeddings'\n\n# Create extraction directory if it doesn't exist\nif not os.path.exists(extract_to):\n    os.makedirs(extract_to)\n# Extract the zip file\nwith zipfile.ZipFile(zip_path, 'r') as zip_ref:\n    zip_ref.extractall(extract_to)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T16:04:20.754173Z","iopub.execute_input":"2024-09-09T16:04:20.754606Z","iopub.status.idle":"2024-09-09T16:07:01.826035Z","shell.execute_reply.started":"2024-09-09T16:04:20.754564Z","shell.execute_reply":"2024-09-09T16:07:01.824798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Loading embedding Model**","metadata":{}},{"cell_type":"code","source":"from tqdm import tqdm\nembedding_vector = {}\nf = open('/kaggle/working/embeddings/glove.840B.300d/glove.840B.300d.txt')\nfor line in tqdm(f):\n    value = line.split(' ')\n    word = value[0]\n    coef = np.array(value[1:],dtype = 'float32')\n    embedding_vector[word] = coef","metadata":{"execution":{"iopub.status.busy":"2024-09-09T16:07:01.827765Z","iopub.execute_input":"2024-09-09T16:07:01.828170Z","iopub.status.idle":"2024-09-09T16:10:32.764450Z","shell.execute_reply.started":"2024-09-09T16:07:01.828127Z","shell.execute_reply":"2024-09-09T16:10:32.763176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_vector['hello'].shape","metadata":{"execution":{"iopub.status.busy":"2024-09-09T16:10:32.766204Z","iopub.execute_input":"2024-09-09T16:10:32.766729Z","iopub.status.idle":"2024-09-09T16:10:32.774705Z","shell.execute_reply.started":"2024-09-09T16:10:32.766675Z","shell.execute_reply":"2024-09-09T16:10:32.773662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t = Tokenizer()\nt.fit_on_texts(train['clean_text'])\nencodings_tr = t.texts_to_sequences(train['clean_text'])\nencodings_tr = pad_sequences(encodings_tr,maxlen = 50,padding='post')\nencodings_tr = pad_sequences(encodings_tr,maxlen = 50,padding='post')\n\nencodings_ts = t.texts_to_sequences(test['clean_text'])\nencodings_ts = pad_sequences(encodings_ts,maxlen = 50,padding='post')\nencodings_ts = pad_sequences(encodings_ts,maxlen = 50,padding='post')","metadata":{"execution":{"iopub.status.busy":"2024-09-09T16:10:32.778468Z","iopub.execute_input":"2024-09-09T16:10:32.778888Z","iopub.status.idle":"2024-09-09T16:11:51.877563Z","shell.execute_reply.started":"2024-09-09T16:10:32.778833Z","shell.execute_reply":"2024-09-09T16:11:51.876532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_size = len(t.word_index) + 1\nembed_M = np.zeros((vocab_size,300))\nfor word,i in tqdm(t.word_index.items()):\n    if word in embedding_vector:\n        embed_M[i] = embedding_vector[word]","metadata":{"execution":{"iopub.status.busy":"2024-09-09T16:11:51.878852Z","iopub.execute_input":"2024-09-09T16:11:51.879208Z","iopub.status.idle":"2024-09-09T16:11:52.652819Z","shell.execute_reply.started":"2024-09-09T16:11:51.879171Z","shell.execute_reply":"2024-09-09T16:11:52.651663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embed_M.shape","metadata":{"execution":{"iopub.status.busy":"2024-09-09T15:54:15.811399Z","iopub.execute_input":"2024-09-09T15:54:15.811876Z","iopub.status.idle":"2024-09-09T15:54:15.820248Z","shell.execute_reply.started":"2024-09-09T15:54:15.811828Z","shell.execute_reply":"2024-09-09T15:54:15.818913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train_ohe = tf.keras.utils.to_categorical(train['target'], num_classes=2)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T16:11:52.654426Z","iopub.execute_input":"2024-09-09T16:11:52.655375Z","iopub.status.idle":"2024-09-09T16:11:52.689711Z","shell.execute_reply.started":"2024-09-09T16:11:52.655319Z","shell.execute_reply":"2024-09-09T16:11:52.688743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Modelling**","metadata":{}},{"cell_type":"code","source":"model = Sequential()\ninput = Input(shape=(50,))\nmodel.add(input)\nmodel.add(Embedding(vocab_size,300,input_length = 50,weights = [embed_M],trainable = False))\nmodel.add(Bidirectional(LSTM(64)))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(128,activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(Dense(2,activation='softmax'))\nmodel.compile(loss='categorical_crossentropy', optimizer = Adam(learning_rate=1e-4), metrics =[ F1Score(threshold=0.5)])","metadata":{"execution":{"iopub.status.busy":"2024-09-09T16:11:52.690926Z","iopub.execute_input":"2024-09-09T16:11:52.691315Z","iopub.status.idle":"2024-09-09T16:11:55.152556Z","shell.execute_reply.started":"2024-09-09T16:11:52.691256Z","shell.execute_reply":"2024-09-09T16:11:55.151446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-09-09T16:11:55.154206Z","iopub.execute_input":"2024-09-09T16:11:55.154798Z","iopub.status.idle":"2024-09-09T16:11:55.182424Z","shell.execute_reply.started":"2024-09-09T16:11:55.154731Z","shell.execute_reply":"2024-09-09T16:11:55.181351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callbacks = [\n    tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=1, verbose=1),\n]\nhistory = model.fit(\n    encodings_tr, y_train_ohe,\n    # validation_data = (X_test,y_test),\n    validation_split = 0.2,\n    epochs = 10,\n    callbacks = callbacks,\n    batch_size = 512,\n)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T16:11:55.183852Z","iopub.execute_input":"2024-09-09T16:11:55.184222Z","iopub.status.idle":"2024-09-09T16:14:57.988250Z","shell.execute_reply.started":"2024-09-09T16:11:55.184183Z","shell.execute_reply":"2024-09-09T16:14:57.987272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_,val_X,__,val_y = train_test_split(encodings_tr,train['target'],test_size = 0.2)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T16:15:49.242052Z","iopub.execute_input":"2024-09-09T16:15:49.242839Z","iopub.status.idle":"2024-09-09T16:15:49.472878Z","shell.execute_reply.started":"2024-09-09T16:15:49.242788Z","shell.execute_reply":"2024-09-09T16:15:49.471712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_2 = model.predict(val_X)\nx,y,th = precision_recall_curve(y_true = val_y,probas_pred = y_pred_2[:,1])\nopt = 2*x*y/(x+y)\nbest_th = th[np.argmax(opt)]\nbest_th","metadata":{"execution":{"iopub.status.busy":"2024-09-09T16:20:05.179683Z","iopub.execute_input":"2024-09-09T16:20:05.181017Z","iopub.status.idle":"2024-09-09T16:20:05.272356Z","shell.execute_reply.started":"2024-09-09T16:20:05.180951Z","shell.execute_reply":"2024-09-09T16:20:05.271079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\npred = model.predict(encodings_ts)\npred_logit = pred >= best_th\nout = pd.DataFrame()\nout['qid'] = test['qid']\nout['prediction'] = pred_logit[:,1]\nout['prediction'] = out['prediction'].astype(int)\nout.to_csv('submission.csv',index = False)","metadata":{"execution":{"iopub.status.busy":"2024-09-09T16:25:07.707455Z","iopub.execute_input":"2024-09-09T16:25:07.707937Z","iopub.status.idle":"2024-09-09T16:25:08.464996Z","shell.execute_reply.started":"2024-09-09T16:25:07.707895Z","shell.execute_reply":"2024-09-09T16:25:08.463827Z"},"trusted":true},"execution_count":null,"outputs":[]}]}