{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-13T19:16:57.575629Z","iopub.execute_input":"2023-07-13T19:16:57.576355Z","iopub.status.idle":"2023-07-13T19:16:57.587028Z","shell.execute_reply.started":"2023-07-13T19:16:57.576316Z","shell.execute_reply":"2023-07-13T19:16:57.586013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install tensorflow\n","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:16:57.590381Z","iopub.execute_input":"2023-07-13T19:16:57.592624Z","iopub.status.idle":"2023-07-13T19:17:10.848413Z","shell.execute_reply.started":"2023-07-13T19:16:57.592590Z","shell.execute_reply":"2023-07-13T19:17:10.847254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n%matplotlib inline\nimport re\n\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split as split\n\nimport nltk\nnltk.download('stopwords')\nfrom nltk.corpus import stopwords\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set_style('darkgrid')\nplt.rcParams[\"figure.figsize\"] = (9, 6)","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:17:10.852287Z","iopub.execute_input":"2023-07-13T19:17:10.852667Z","iopub.status.idle":"2023-07-13T19:17:28.200146Z","shell.execute_reply.started":"2023-07-13T19:17:10.852636Z","shell.execute_reply":"2023-07-13T19:17:28.199227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read the dataset","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\")\ndata.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:17:28.201465Z","iopub.execute_input":"2023-07-13T19:17:28.202194Z","iopub.status.idle":"2023-07-13T19:17:32.212847Z","shell.execute_reply.started":"2023-07-13T19:17:28.202159Z","shell.execute_reply":"2023-07-13T19:17:32.211778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# - lets work with sample of the entire dataset. Dataset is huge in size already !!\nsample_data = data.sample(frac=.20)\nsample_data.pop('qid')\nsample_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:17:32.214526Z","iopub.execute_input":"2023-07-13T19:17:32.215146Z","iopub.status.idle":"2023-07-13T19:17:32.327880Z","shell.execute_reply.started":"2023-07-13T19:17:32.215108Z","shell.execute_reply":"2023-07-13T19:17:32.326741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_data.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:17:32.331623Z","iopub.execute_input":"2023-07-13T19:17:32.332008Z","iopub.status.idle":"2023-07-13T19:17:32.340283Z","shell.execute_reply.started":"2023-07-13T19:17:32.331973Z","shell.execute_reply":"2023-07-13T19:17:32.339167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# - No null values in the dataset\nsample_data.info()","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:17:32.341869Z","iopub.execute_input":"2023-07-13T19:17:32.344787Z","iopub.status.idle":"2023-07-13T19:17:32.462545Z","shell.execute_reply.started":"2023-07-13T19:17:32.344750Z","shell.execute_reply":"2023-07-13T19:17:32.461458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA 📉","metadata":{}},{"cell_type":"code","source":"sincere_df = data[data['target'] == 0]\nsincere_df['question_text'].values[:10]","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:17:32.464055Z","iopub.execute_input":"2023-07-13T19:17:32.464415Z","iopub.status.idle":"2023-07-13T19:17:32.545433Z","shell.execute_reply.started":"2023-07-13T19:17:32.464382Z","shell.execute_reply":"2023-07-13T19:17:32.544436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"insincere_df = data[data['target'] == 1]\ninsincere_df['question_text'].values[:10]","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:17:32.546967Z","iopub.execute_input":"2023-07-13T19:17:32.547332Z","iopub.status.idle":"2023-07-13T19:17:32.572860Z","shell.execute_reply.started":"2023-07-13T19:17:32.547300Z","shell.execute_reply":"2023-07-13T19:17:32.571868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Class Imabalance: Target variable","metadata":{}},{"cell_type":"code","source":"ax = sns.countplot(\n    data = data, \n    x = 'target', \n    order = data['target'].value_counts().index, \n    palette=\"PuBu_r\", \n    width = .6\n)\n\nfor p, label in zip(ax.patches, data['target'].value_counts()):\n    ax.annotate(label, (p.get_x() + .20, p.get_height() + .8))\n\nax.set_xlabel('Target')\nax.set_ylabel('Count')\nax.set_title('Quroa Target Column Categories')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:17:32.574340Z","iopub.execute_input":"2023-07-13T19:17:32.574709Z","iopub.status.idle":"2023-07-13T19:17:32.996534Z","shell.execute_reply.started":"2023-07-13T19:17:32.574676Z","shell.execute_reply":"2023-07-13T19:17:32.995438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"note:\n- clearly target variable is imbalanced.\n- treat target imbalance.","metadata":{}},{"cell_type":"markdown","source":"# Data Cleaning & Preparation 🧮","metadata":{}},{"cell_type":"code","source":"def custome_preprocessing(text: str) -> str:\n    \"\"\" function to clean redundant string values \n        from array of strings returns a cleaned \n        string values removing unwanted symbols and \n        numbers.\n        ----\n        text: string values\n    \"\"\"\n    for sent in text.split('-'):\n        t = re.sub(r'http://\\S+|https://\\S+', '', sent.lower())\n        t = re.sub(r'[a-z]+.com\\S+', '', t)\n        t = re.sub(r'[0-9]', '', t)\n        t = re.sub(r'[^\\w\\s]', '', t)\n        t = re.sub(r'<.*?>', '', t)\n        t = re.sub(r'[#£^.*<>?!+=/)(%]', '', t)\n    return t","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:17:32.998162Z","iopub.execute_input":"2023-07-13T19:17:32.998570Z","iopub.status.idle":"2023-07-13T19:17:33.006866Z","shell.execute_reply.started":"2023-07-13T19:17:32.998530Z","shell.execute_reply":"2023-07-13T19:17:33.005557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stoppies = stopwords.words('english')","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:17:33.008262Z","iopub.execute_input":"2023-07-13T19:17:33.009386Z","iopub.status.idle":"2023-07-13T19:17:33.024567Z","shell.execute_reply.started":"2023-07-13T19:17:33.009347Z","shell.execute_reply":"2023-07-13T19:17:33.023357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_data['clean'] = sample_data['question_text'].apply(lambda x: custome_preprocessing(x))\nsample_data['clean'] = sample_data['clean'].apply(lambda x: ' '.join(x for x in x.split() if len(x) > 1))\nsample_data['clean'] = sample_data['clean'].apply(lambda x: ' '.join(x for x in x.split() if x not in set(stoppies)))\n\nfor sent in sample_data['clean'].str.split()[:20]:\n    print(*sent)","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:17:33.026569Z","iopub.execute_input":"2023-07-13T19:17:33.027361Z","iopub.status.idle":"2023-07-13T19:17:55.095312Z","shell.execute_reply.started":"2023-07-13T19:17:33.027324Z","shell.execute_reply":"2023-07-13T19:17:55.093428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for sent in sample_data['question_text'].str.split()[:20]:\n    print(*sent)","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:17:55.096670Z","iopub.execute_input":"2023-07-13T19:17:55.097110Z","iopub.status.idle":"2023-07-13T19:17:56.340862Z","shell.execute_reply.started":"2023-07-13T19:17:55.097076Z","shell.execute_reply":"2023-07-13T19:17:56.339897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_data[sample_data['question_text'].str.contains(\"Why do Europeans\")][:10]","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:17:56.344948Z","iopub.execute_input":"2023-07-13T19:17:56.345695Z","iopub.status.idle":"2023-07-13T19:17:56.598335Z","shell.execute_reply.started":"2023-07-13T19:17:56.345667Z","shell.execute_reply":"2023-07-13T19:17:56.597259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split the dataset for the training 📘","metadata":{}},{"cell_type":"code","source":"questions = np.array(sample_data['clean'].values)\nlabels = np.array(sample_data['target'].values)\n\n(train_inputs, val_inputs, train_target, val_target) = split(\n    questions, \n    labels, \n    test_size = .30, \n    random_state = 42, \n    stratify = labels\n) \n\nprint(f\"training size: {len(train_inputs)}\", f\"validation size: {len(val_inputs)}\", sep = \"\\n\")","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:17:56.599888Z","iopub.execute_input":"2023-07-13T19:17:56.600277Z","iopub.status.idle":"2023-07-13T19:17:56.715559Z","shell.execute_reply.started":"2023-07-13T19:17:56.600242Z","shell.execute_reply":"2023-07-13T19:17:56.714557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generate Tokens and Padded Sequencess 🧱","metadata":{}},{"cell_type":"code","source":"# - Parameters\nvocab_size = 10_000\nmax_length = 120\nembedding_dim = 16\ntruncation_type = 'post'\noov_token = '<OOV>'","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:17:56.717026Z","iopub.execute_input":"2023-07-13T19:17:56.717609Z","iopub.status.idle":"2023-07-13T19:17:56.722656Z","shell.execute_reply.started":"2023-07-13T19:17:56.717573Z","shell.execute_reply":"2023-07-13T19:17:56.721744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# - Initialize tensorflow tokenizer\ntokenizer = tf.keras.preprocessing.text.Tokenizer(num_words = vocab_size, oov_token = oov_token)\n\n# - Fit & generate word index dictionary on training set\ntokenizer.fit_on_texts(train_inputs)\nword_index = tokenizer.word_index\n\ntraining_seqs  = tokenizer.texts_to_sequences(train_inputs)\ntraining_paded = tf.keras.utils.pad_sequences(training_seqs, maxlen = max_length, truncating = truncation_type)\n\n# - Apply tokenizer on validation dataset\nval_seqs  = tokenizer.texts_to_sequences(val_inputs)\nval_paded = tf.keras.utils.pad_sequences(val_seqs, maxlen = max_length, truncating = truncation_type)","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:17:56.724177Z","iopub.execute_input":"2023-07-13T19:17:56.724910Z","iopub.status.idle":"2023-07-13T19:18:04.154501Z","shell.execute_reply.started":"2023-07-13T19:17:56.724871Z","shell.execute_reply":"2023-07-13T19:18:04.153538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build NLP Model: Using Tensorflow Sequential Model API 🚀","metadata":{}},{"cell_type":"code","source":"base_model = tf.keras.Sequential([\n    tf.keras.layers.Embedding(\n        vocab_size, \n        embedding_dim, \n        input_length = max_length, \n        name='embeddings',\n    ),\n    tf.keras.layers.GlobalAveragePooling1D(),\n    tf.keras.layers.Dense(16, activation = 'relu'),\n    tf.keras.layers.Dense(8,  activation = 'relu'),\n    tf.keras.layers.Dense(1,  activation = 'sigmoid')\n])\n\nbase_model.compile(loss = 'binary_crossentropy', optimizer = 'adam', metrics = ['accuracy'])\n\nhistory = base_model.fit(\n    training_paded, \n    train_target, \n    validation_data = (val_paded, val_target), \n    epochs = 10\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:18:04.155815Z","iopub.execute_input":"2023-07-13T19:18:04.156262Z","iopub.status.idle":"2023-07-13T19:24:33.014930Z","shell.execute_reply.started":"2023-07-13T19:18:04.156227Z","shell.execute_reply":"2023-07-13T19:24:33.013914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_learning_graph(tf_object: tf.keras.callbacks.History) -> None:\n    plt.subplot(211)\n    plt.title(\"Cross-Entropy loss\", pad=10)\n    plt.plot(tf_object.history['loss'], label='training', color='skyblue')\n    plt.plot(tf_object.history['val_loss'], label='validation', color='teal')\n    plt.ylabel(\"Loss\")\n    plt.xlabel(\"Epochs\")\n    plt.legend()\n    plt.show()\n\n    # plot accuracy learning curves\n    plt.subplot(212)\n    plt.title('Accuracy', pad=10)\n    plt.plot(tf_object.history['accuracy'], label='training', color='skyblue')\n    plt.plot(tf_object.history['val_accuracy'], label='validation', color='teal')\n    plt.ylabel(\"Accuracy\")\n    plt.xlabel(\"Epochs\")\n    plt.legend()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:24:33.016612Z","iopub.execute_input":"2023-07-13T19:24:33.017219Z","iopub.status.idle":"2023-07-13T19:24:33.026072Z","shell.execute_reply.started":"2023-07-13T19:24:33.017182Z","shell.execute_reply":"2023-07-13T19:24:33.024897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_learning_graph(history)","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:24:33.027538Z","iopub.execute_input":"2023-07-13T19:24:33.027871Z","iopub.status.idle":"2023-07-13T19:24:33.793113Z","shell.execute_reply.started":"2023-07-13T19:24:33.027837Z","shell.execute_reply":"2023-07-13T19:24:33.791543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"__Notes__\n- Accuracy is not stable.\n- Validation loss can be lowered. It seems increasing at the end of epochs\n- Model fine tunning is required.","metadata":{}},{"cell_type":"code","source":"%%time\n%%capture\n\nbase_model2 = tf.keras.Sequential([\n    tf.keras.layers.Embedding(\n        vocab_size, \n        embedding_dim, \n        input_length = max_length\n    ),\n    tf.keras.layers.Flatten(),\n    tf.keras.layers.Dense(64, activation = 'relu'),\n    tf.keras.layers.Dense(16, activation = 'relu'),\n    tf.keras.layers.Dense(1, activation  = 'sigmoid'),\n])\n\nbase_model2.compile(loss = 'binary_crossentropy', optimizer = 'adam', metrics = ['accuracy'])\n\ntf_obj = base_model2.fit(\n    training_paded, \n    train_target, \n    validation_data = (val_paded, val_target), \n    epochs = 10, \n    batch_size = 64\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:24:33.794892Z","iopub.execute_input":"2023-07-13T19:24:33.795648Z","iopub.status.idle":"2023-07-13T19:27:44.780422Z","shell.execute_reply.started":"2023-07-13T19:24:33.795609Z","shell.execute_reply":"2023-07-13T19:27:44.779246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_learning_graph(tf_obj)","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:27:44.781961Z","iopub.execute_input":"2023-07-13T19:27:44.783748Z","iopub.status.idle":"2023-07-13T19:27:45.825740Z","shell.execute_reply.started":"2023-07-13T19:27:44.783713Z","shell.execute_reply":"2023-07-13T19:27:45.823545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n%%capture\n\nbase_model3 = tf.keras.Sequential([\n    tf.keras.layers.Embedding(\n        vocab_size, \n        embedding_dim, \n        input_length = max_length\n    ),\n    tf.keras.layers.LSTM(64),\n    tf.keras.layers.Dense(16, activation = 'relu'),\n    tf.keras.layers.Dense(1, activation  = 'sigmoid'),\n])\n\nbase_model3.compile(loss = 'binary_crossentropy', optimizer = 'adam', metrics = ['accuracy'])\n\ntf_obj = base_model3.fit(\n    training_paded, \n    train_target, \n    validation_data = (val_paded, val_target), \n    epochs = 10, \n    batch_size = 64\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:27:45.827290Z","iopub.execute_input":"2023-07-13T19:27:45.827664Z","iopub.status.idle":"2023-07-13T19:32:53.963228Z","shell.execute_reply.started":"2023-07-13T19:27:45.827630Z","shell.execute_reply":"2023-07-13T19:32:53.962159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_learning_graph(tf_obj)","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:34:22.656713Z","iopub.execute_input":"2023-07-13T19:34:22.657653Z","iopub.status.idle":"2023-07-13T19:34:23.330920Z","shell.execute_reply.started":"2023-07-13T19:34:22.657617Z","shell.execute_reply":"2023-07-13T19:34:23.330028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n%%capture\n\nbase_model4 = tf.keras.Sequential([\n    tf.keras.layers.Embedding(\n        vocab_size, \n        embedding_dim, \n        input_length = max_length, \n        name = 'embedding'\n    ),\n    tf.keras.layers.LSTM(128),\n    tf.keras.layers.Dropout(0.5),\n    tf.keras.layers.Dense(64),\n    tf.keras.layers.Dropout(0.5),\n    tf.keras.layers.Dense(16, activation = 'relu'),\n    tf.keras.layers.Dense(1,  activation = 'sigmoid'),\n])\n\nbase_model4.compile(loss = 'binary_crossentropy', optimizer = 'adam', metrics = ['accuracy'])\n\ntf_obj = base_model4.fit(\n    training_paded, \n    train_target,\n    validation_data = (val_paded, val_target),\n    epochs = 10,\n    batch_size = 64,\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:45:37.407787Z","iopub.execute_input":"2023-07-13T19:45:37.408170Z","iopub.status.idle":"2023-07-13T19:52:01.762959Z","shell.execute_reply.started":"2023-07-13T19:45:37.408138Z","shell.execute_reply":"2023-07-13T19:52:01.761768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_learning_graph(tf_obj)","metadata":{"execution":{"iopub.status.busy":"2023-07-13T20:02:24.113951Z","iopub.execute_input":"2023-07-13T20:02:24.114337Z","iopub.status.idle":"2023-07-13T20:02:24.845886Z","shell.execute_reply.started":"2023-07-13T20:02:24.114304Z","shell.execute_reply":"2023-07-13T20:02:24.844971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ** CONTINUING FURTHER **","metadata":{"execution":{"iopub.status.busy":"2023-07-13T19:32:53.964615Z","iopub.execute_input":"2023-07-13T19:32:53.965444Z","iopub.status.idle":"2023-07-13T19:32:53.969984Z","shell.execute_reply.started":"2023-07-13T19:32:53.965408Z","shell.execute_reply":"2023-07-13T19:32:53.968930Z"},"trusted":true},"execution_count":null,"outputs":[]}]}