{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-17T19:31:31.795178Z","iopub.execute_input":"2023-11-17T19:31:31.795570Z","iopub.status.idle":"2023-11-17T19:31:31.805282Z","shell.execute_reply.started":"2023-11-17T19:31:31.795539Z","shell.execute_reply":"2023-11-17T19:31:31.804148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Detail design of Features","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport re","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:31.809061Z","iopub.execute_input":"2023-11-17T19:31:31.810191Z","iopub.status.idle":"2023-11-17T19:31:31.819497Z","shell.execute_reply.started":"2023-11-17T19:31:31.810147Z","shell.execute_reply":"2023-11-17T19:31:31.818151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/quora-insincere-questions-classification/train.csv\").sample(1000)\nX_train = train_df[\"question_text\"]\n\nfor row in X_train[:10]:\n    print(row)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:51:02.061119Z","iopub.execute_input":"2023-11-17T19:51:02.061527Z","iopub.status.idle":"2023-11-17T19:51:05.291815Z","shell.execute_reply.started":"2023-11-17T19:51:02.061498Z","shell.execute_reply":"2023-11-17T19:51:05.290558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text = train_df['question_text']","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:34.973762Z","iopub.execute_input":"2023-11-17T19:31:34.974471Z","iopub.status.idle":"2023-11-17T19:31:34.979832Z","shell.execute_reply.started":"2023-11-17T19:31:34.974425Z","shell.execute_reply":"2023-11-17T19:31:34.978647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(\"../input/quora-insincere-questions-classification/test.csv\")\nX_test = test_df[\"question_text\"].fillna(\"dieter\").values\ny = train_df[\"target\"]","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:34.983263Z","iopub.execute_input":"2023-11-17T19:31:34.983969Z","iopub.status.idle":"2023-11-17T19:31:35.870484Z","shell.execute_reply.started":"2023-11-17T19:31:34.983933Z","shell.execute_reply":"2023-11-17T19:31:35.869492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. Remove Numbers ","metadata":{}},{"cell_type":"code","source":"def removeNumbers(text):\n    text = ''.join([i for i in text if not i.isdigit()])         \n    return text\n","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:35.871720Z","iopub.execute_input":"2023-11-17T19:31:35.872340Z","iopub.status.idle":"2023-11-17T19:31:35.878498Z","shell.execute_reply.started":"2023-11-17T19:31:35.872304Z","shell.execute_reply":"2023-11-17T19:31:35.877168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_removeNumbers = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_removeNumbers['TextBefore'] = text.copy()","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:35.880222Z","iopub.execute_input":"2023-11-17T19:31:35.880689Z","iopub.status.idle":"2023-11-17T19:31:35.894886Z","shell.execute_reply.started":"2023-11-17T19:31:35.880644Z","shell.execute_reply":"2023-11-17T19:31:35.893588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_removeNumbers['TextAfter'] = text_removeNumbers['TextBefore'].apply(removeNumbers)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:35.896939Z","iopub.execute_input":"2023-11-17T19:31:35.898272Z","iopub.status.idle":"2023-11-17T19:31:35.910607Z","shell.execute_reply.started":"2023-11-17T19:31:35.898194Z","shell.execute_reply":"2023-11-17T19:31:35.909299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_removeNumbers['Changed'] = np.where(text_removeNumbers['TextBefore']==text_removeNumbers['TextAfter'], 'no', 'yes')\n\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_removeNumbers[text_removeNumbers['Changed']=='yes']), len(text_removeNumbers), 100*len(text_removeNumbers[text_removeNumbers['Changed']=='yes'])/len(text_removeNumbers)))\n","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:35.912398Z","iopub.execute_input":"2023-11-17T19:31:35.912804Z","iopub.status.idle":"2023-11-17T19:31:35.926922Z","shell.execute_reply.started":"2023-11-17T19:31:35.912742Z","shell.execute_reply":"2023-11-17T19:31:35.925649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_removeNumbers[\n    text_removeNumbers['Changed'] == 'yes'\n] # shows all the text where numbers are removed","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:35.932128Z","iopub.execute_input":"2023-11-17T19:31:35.932506Z","iopub.status.idle":"2023-11-17T19:31:35.950940Z","shell.execute_reply.started":"2023-11-17T19:31:35.932477Z","shell.execute_reply":"2023-11-17T19:31:35.949565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"2. Replace Repetitions of Punctuation\n    * exlamation marks with the tag \"multiExclamation\"\n    * question marks with the tag \"multiQuestion\"\n    * stop marks with the tag \"multiStop\"\n      \n    ","metadata":{}},{"cell_type":"code","source":"def replaceMultiExclamationMark(text):\n    \n    text = re.sub(r\"(\\!)\\1+\", ' multiExclamation ', text)\n    return text\n\ndef replaceMultiQuestionMark(text):\n\n    text = re.sub(r\"(\\?)\\1+\", ' multiQuestion ', text)\n    return text\n\ndef replaceMultiStopMark(text):\n\n    text = re.sub(r\"(\\.)\\1+\", ' multiStop ', text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:35.952461Z","iopub.execute_input":"2023-11-17T19:31:35.952795Z","iopub.status.idle":"2023-11-17T19:31:35.960385Z","shell.execute_reply.started":"2023-11-17T19:31:35.952745Z","shell.execute_reply":"2023-11-17T19:31:35.959133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_replaceRepOfPunct = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_replaceRepOfPunct['TextBefore'] = text_removeNumbers['TextAfter']","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:35.961919Z","iopub.execute_input":"2023-11-17T19:31:35.962261Z","iopub.status.idle":"2023-11-17T19:31:35.974911Z","shell.execute_reply.started":"2023-11-17T19:31:35.962232Z","shell.execute_reply":"2023-11-17T19:31:35.973905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntext_replaceRepOfPunct['TextAfter'] = text_replaceRepOfPunct['TextBefore'].apply(replaceMultiExclamationMark)\ntext_replaceRepOfPunct['TextAfter'] = text_replaceRepOfPunct['TextBefore'].apply(replaceMultiQuestionMark)\ntext_replaceRepOfPunct['TextAfter'] = text_replaceRepOfPunct['TextBefore'].apply(replaceMultiStopMark)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:35.977210Z","iopub.execute_input":"2023-11-17T19:31:35.977563Z","iopub.status.idle":"2023-11-17T19:31:35.989570Z","shell.execute_reply.started":"2023-11-17T19:31:35.977533Z","shell.execute_reply":"2023-11-17T19:31:35.988230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_replaceRepOfPunct['Changed'] = np.where(text_replaceRepOfPunct['TextBefore']==text_replaceRepOfPunct['TextAfter'], 'no', 'yes')","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:35.991680Z","iopub.execute_input":"2023-11-17T19:31:35.992184Z","iopub.status.idle":"2023-11-17T19:31:36.003367Z","shell.execute_reply.started":"2023-11-17T19:31:35.992138Z","shell.execute_reply":"2023-11-17T19:31:36.001270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_replaceRepOfPunct[text_replaceRepOfPunct['Changed']=='yes']), len(text_replaceRepOfPunct), 100*len(text_replaceRepOfPunct[text_replaceRepOfPunct['Changed']=='yes'])/len(text_replaceRepOfPunct)))","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.005513Z","iopub.execute_input":"2023-11-17T19:31:36.005971Z","iopub.status.idle":"2023-11-17T19:31:36.017727Z","shell.execute_reply.started":"2023-11-17T19:31:36.005937Z","shell.execute_reply":"2023-11-17T19:31:36.016276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_removeNumbers[\n    text_removeNumbers['Changed'] == 'yes'\n]","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.019584Z","iopub.execute_input":"2023-11-17T19:31:36.019995Z","iopub.status.idle":"2023-11-17T19:31:36.035708Z","shell.execute_reply.started":"2023-11-17T19:31:36.019961Z","shell.execute_reply":"2023-11-17T19:31:36.034516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"3. Remove Punctuation","metadata":{}},{"cell_type":"code","source":"import string\ndef translator(text):\n    table = str.maketrans('', '', string.punctuation)\n    return text.translate(table)\n\ntext_removePunctuation = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_removePunctuation['TextBefore'] = text_replaceRepOfPunct['TextAfter']","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.036868Z","iopub.execute_input":"2023-11-17T19:31:36.037229Z","iopub.status.idle":"2023-11-17T19:31:36.050921Z","shell.execute_reply.started":"2023-11-17T19:31:36.037191Z","shell.execute_reply":"2023-11-17T19:31:36.049415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_removePunctuation['TextAfter'] = text_removePunctuation['TextBefore'].apply(translator) ","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.053083Z","iopub.execute_input":"2023-11-17T19:31:36.053647Z","iopub.status.idle":"2023-11-17T19:31:36.068647Z","shell.execute_reply.started":"2023-11-17T19:31:36.053603Z","shell.execute_reply":"2023-11-17T19:31:36.067712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_removePunctuation['Changed'] = np.where(text_removePunctuation['TextBefore']==text_removePunctuation['TextAfter'], 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_removePunctuation[text_removePunctuation['Changed']=='yes']), len(text_removePunctuation), 100*len(text_removePunctuation[text_removePunctuation['Changed']=='yes'])/len(text_removePunctuation)))\n","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.070460Z","iopub.execute_input":"2023-11-17T19:31:36.070941Z","iopub.status.idle":"2023-11-17T19:31:36.084964Z","shell.execute_reply.started":"2023-11-17T19:31:36.070899Z","shell.execute_reply":"2023-11-17T19:31:36.083771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_removePunctuation[\n    text_removePunctuation['Changed'] == 'yes'\n].iloc[:3]","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.087686Z","iopub.execute_input":"2023-11-17T19:31:36.088094Z","iopub.status.idle":"2023-11-17T19:31:36.102347Z","shell.execute_reply.started":"2023-11-17T19:31:36.088055Z","shell.execute_reply":"2023-11-17T19:31:36.101283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"4. Replace Contractions","metadata":{}},{"cell_type":"code","source":"contraction_patterns = [ (r'won\\'t', 'will not'), (r'can\\'t', 'cannot'), (r'i\\'m', 'i am'), (r'ain\\'t', 'is not'),\n                        (r'(\\w+)\\'ll', '\\g<1> will'), (r'(\\w+)n\\'t', '\\g<1> not'), (r'(\\w+)\\'ve', '\\g<1> have'),\n                        (r'(\\w+)\\'s', '\\g<1> is'), (r'(\\w+)\\'re', '\\g<1> are'), (r'(\\w+)\\'d', '\\g<1> would'),\n                        (r'&', 'and'), (r'dammit', 'damn it'), (r'dont', 'do not'), (r'wont', 'will not') ]","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.103698Z","iopub.execute_input":"2023-11-17T19:31:36.104089Z","iopub.status.idle":"2023-11-17T19:31:36.114505Z","shell.execute_reply.started":"2023-11-17T19:31:36.104057Z","shell.execute_reply":"2023-11-17T19:31:36.113563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def replaceContraction(text):\n    patterns = [(re.compile(regex), repl) for (regex, repl) in contraction_patterns]\n    for (pattern, repl) in patterns:\n        (text, count) = re.subn(pattern, repl, text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.115918Z","iopub.execute_input":"2023-11-17T19:31:36.116366Z","iopub.status.idle":"2023-11-17T19:31:36.126685Z","shell.execute_reply.started":"2023-11-17T19:31:36.116321Z","shell.execute_reply":"2023-11-17T19:31:36.125577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_replaceContractions = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_replaceContractions['TextBefore'] = text_removePunctuation['TextAfter']","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.127990Z","iopub.execute_input":"2023-11-17T19:31:36.128421Z","iopub.status.idle":"2023-11-17T19:31:36.141368Z","shell.execute_reply.started":"2023-11-17T19:31:36.128381Z","shell.execute_reply":"2023-11-17T19:31:36.139995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_replaceContractions['TextAfter'] = text_replaceContractions['TextBefore'].apply(replaceContraction)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.143094Z","iopub.execute_input":"2023-11-17T19:31:36.143489Z","iopub.status.idle":"2023-11-17T19:31:36.162300Z","shell.execute_reply.started":"2023-11-17T19:31:36.143456Z","shell.execute_reply":"2023-11-17T19:31:36.161141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_replaceContractions['Changed'] = np.where(text_replaceContractions['TextBefore']==text_replaceContractions['TextAfter'], 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_replaceContractions[text_replaceContractions['Changed']=='yes']), len(text_replaceContractions), 100*len(text_replaceContractions[text_replaceContractions['Changed']=='yes'])/len(text_replaceContractions)))","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.168136Z","iopub.execute_input":"2023-11-17T19:31:36.168514Z","iopub.status.idle":"2023-11-17T19:31:36.177293Z","shell.execute_reply.started":"2023-11-17T19:31:36.168483Z","shell.execute_reply":"2023-11-17T19:31:36.175998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_replaceContractions[\n    text_replaceContractions['Changed'] == 'yes'\n].iloc[:3]","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.178586Z","iopub.execute_input":"2023-11-17T19:31:36.179077Z","iopub.status.idle":"2023-11-17T19:31:36.198291Z","shell.execute_reply.started":"2023-11-17T19:31:36.179031Z","shell.execute_reply":"2023-11-17T19:31:36.196981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"5. Lowercase","metadata":{}},{"cell_type":"code","source":"text_lowercase = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_lowercase['TextBefore'] = text_replaceContractions['TextAfter']\n\ntext_lowercase['TextAfter'] = text_lowercase['TextBefore'].str.lower()\n\n\ntext_lowercase['Changed'] = np.where(text_lowercase['TextBefore']==text_lowercase['TextAfter'], 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_lowercase[text_lowercase['Changed']=='yes']), len(text_lowercase), 100*len(text_lowercase[text_lowercase['Changed']=='yes'])/len(text_lowercase)))\n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.201365Z","iopub.execute_input":"2023-11-17T19:31:36.201884Z","iopub.status.idle":"2023-11-17T19:31:36.218094Z","shell.execute_reply.started":"2023-11-17T19:31:36.201838Z","shell.execute_reply":"2023-11-17T19:31:36.216436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"6. Replace Negations with Antonyms","metadata":{}},{"cell_type":"code","source":"import nltk\nfrom nltk.corpus import wordnet\n\ndef replace(word, pos=None):\n\n    antonyms = set()\n    for syn in wordnet.synsets(word, pos=pos):\n        for lemma in syn.lemmas():\n            for antonym in lemma.antonyms():\n                antonyms.add(antonym.name())\n    if len(antonyms) == 1:\n        return antonyms.pop()\n    else:\n        return None\n\ndef replaceNegations(text):\n\n    i, l = 0, len(text)\n    words = []\n    while i < l:\n        word = text[i]\n        if word == 'not' and i+1 < l:\n            ant = replace(text[i+1])\n            if ant:\n                words.append(ant)\n                i += 2\n                continue\n        words.append(word)\n        i += 1\n    return words\n\ndef tokenize1(text):\n    tokens = nltk.word_tokenize(text)\n    tokens = replaceNegations(tokens)\n    text = \" \".join(tokens)\n    return text\n","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.219389Z","iopub.execute_input":"2023-11-17T19:31:36.219757Z","iopub.status.idle":"2023-11-17T19:31:36.230989Z","shell.execute_reply.started":"2023-11-17T19:31:36.219726Z","shell.execute_reply":"2023-11-17T19:31:36.230077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!unzip /usr/share/nltk_data/corpora/wordnet.zip -d /usr/share/nltk_data/corpora/","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.232017Z","iopub.execute_input":"2023-11-17T19:31:36.232525Z","iopub.status.idle":"2023-11-17T19:31:36.242265Z","shell.execute_reply.started":"2023-11-17T19:31:36.232495Z","shell.execute_reply":"2023-11-17T19:31:36.241147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_replaceNegations = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_replaceNegations['TextBefore'] = text_lowercase['TextAfter']\n\ntext_replaceNegations['TextAfter'] = text_replaceNegations['TextBefore'].apply(tokenize1)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.244405Z","iopub.execute_input":"2023-11-17T19:31:36.245372Z","iopub.status.idle":"2023-11-17T19:31:36.272908Z","shell.execute_reply.started":"2023-11-17T19:31:36.245323Z","shell.execute_reply":"2023-11-17T19:31:36.271592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_replaceNegations['Changed'] = np.where(text_replaceNegations['TextBefore'].str.replace(\" \",\"\")==text_replaceNegations['TextAfter'].str.replace(\" \",\"\").str.replace(\"``\",'\"').str.replace(\"''\",'\"'), 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_replaceNegations[text_replaceNegations['Changed']=='yes']), len(text_replaceNegations), 100*len(text_replaceNegations[text_replaceNegations['Changed']=='yes'])/len(text_replaceNegations)))","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.274174Z","iopub.execute_input":"2023-11-17T19:31:36.275069Z","iopub.status.idle":"2023-11-17T19:31:36.294106Z","shell.execute_reply.started":"2023-11-17T19:31:36.275032Z","shell.execute_reply":"2023-11-17T19:31:36.292603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"7. Handle Capitalized Words","metadata":{}},{"cell_type":"code","source":"def addCapTag(word):\n    \"\"\" Finds a word with at least 3 characters capitalized and adds the tag ALL_CAPS_ \"\"\"\n    if(len(re.findall(\"[A-Z]{3,}\", word))):\n        word = word.replace('\\\\', '' )\n        transformed = re.sub(\"[A-Z]{3,}\", \"ALL_CAPS_\"+word, word)\n        return transformed\n    else:\n        return word\n\ndef tokenize2(text):\n    finalTokens = []\n    tokens = nltk.word_tokenize(text)\n    for w in tokens:\n        finalTokens.append(addCapTag(w))\n    text = \" \".join(finalTokens)\n    return text","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.295610Z","iopub.execute_input":"2023-11-17T19:31:36.296438Z","iopub.status.idle":"2023-11-17T19:31:36.305667Z","shell.execute_reply.started":"2023-11-17T19:31:36.296405Z","shell.execute_reply":"2023-11-17T19:31:36.304514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_handleCapWords = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_handleCapWords['TextBefore'] = text_replaceNegations['TextAfter']","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.307927Z","iopub.execute_input":"2023-11-17T19:31:36.308959Z","iopub.status.idle":"2023-11-17T19:31:36.318954Z","shell.execute_reply.started":"2023-11-17T19:31:36.308909Z","shell.execute_reply":"2023-11-17T19:31:36.317896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_handleCapWords['TextAfter'] = text_handleCapWords['TextBefore'].apply(addCapTag)\ntext_handleCapWords['TextAfter'] = text_handleCapWords['TextBefore'].apply(tokenize2)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.320430Z","iopub.execute_input":"2023-11-17T19:31:36.321812Z","iopub.status.idle":"2023-11-17T19:31:36.348969Z","shell.execute_reply.started":"2023-11-17T19:31:36.321739Z","shell.execute_reply":"2023-11-17T19:31:36.347547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_handleCapWords['Changed'] = np.where(text_handleCapWords['TextBefore'].str.replace(\" \",\"\")==text_handleCapWords['TextAfter'].str.replace(\" \",\"\").str.replace(\"``\",'\"').str.replace(\"''\",'\"'), 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_handleCapWords[text_handleCapWords['Changed']=='yes']), len(text_handleCapWords), 100*len(text_handleCapWords[text_handleCapWords['Changed']=='yes'])/len(text_handleCapWords)))\n","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.350740Z","iopub.execute_input":"2023-11-17T19:31:36.351164Z","iopub.status.idle":"2023-11-17T19:31:36.371365Z","shell.execute_reply.started":"2023-11-17T19:31:36.351131Z","shell.execute_reply":"2023-11-17T19:31:36.369456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"8. Remove Stopwords","metadata":{}},{"cell_type":"code","source":"from nltk.corpus import stopwords\nstoplist = stopwords.words('english')\n\ndef tokenize(text):\n    finalTokens = []\n    tokens = nltk.word_tokenize(text)\n    for w in tokens:\n        if (w not in stoplist):\n            finalTokens.append(w)\n    text = \" \".join(finalTokens)\n    return text\n\ntext_removeStopwords = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_removeStopwords['TextBefore'] = text_handleCapWords['TextAfter']\n\ntext_removeStopwords['TextAfter'] = text_removeStopwords['TextBefore'].apply(tokenize)\n\ntext_removeStopwords['Changed'] = np.where(text_removeStopwords['TextBefore'].str.replace(\" \",\"\")==text_removeStopwords['TextAfter'].str.replace(\" \",\"\").str.replace(\"``\",'\"').str.replace(\"''\",'\"'), 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_removeStopwords[text_removeStopwords['Changed']=='yes']), len(text_removeStopwords), 100*len(text_removeStopwords[text_removeStopwords['Changed']=='yes'])/len(text_removeStopwords)))\n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.373255Z","iopub.execute_input":"2023-11-17T19:31:36.373676Z","iopub.status.idle":"2023-11-17T19:31:36.411464Z","shell.execute_reply.started":"2023-11-17T19:31:36.373644Z","shell.execute_reply":"2023-11-17T19:31:36.410534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"9. Replace Elongated Words","metadata":{}},{"cell_type":"code","source":"def replaceElongated(word):\n    \"\"\" Replaces an elongated word with its basic form, unless the word exists in the lexicon \"\"\"\n\n    repeat_regexp = re.compile(r'(\\w*)(\\w)\\2(\\w*)')\n    repl = r'\\1\\2\\3'\n    if wordnet.synsets(word):\n        return word\n    repl_word = repeat_regexp.sub(repl, word)\n    if repl_word != word:      \n        return replaceElongated(repl_word)\n    else:       \n        return repl_word\n    \ndef tokenize(text):\n    finalTokens = []\n    tokens = nltk.word_tokenize(text)\n    for w in tokens:\n        finalTokens.append(replaceElongated(w))\n    text = \" \".join(finalTokens)\n    return text\n\ntext_removeElWords = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_removeElWords['TextBefore'] = text_removeStopwords['TextAfter']\n\ntext_removeElWords['TextAfter'] = text_removeElWords['TextBefore'].apply(tokenize)\n\ntext_removeElWords['Changed'] = np.where(text_removeElWords['TextBefore'].str.replace(\" \",\"\")==text_removeElWords['TextAfter'].str.replace(\" \",\"\").str.replace(\"``\",'\"').str.replace(\"''\",'\"'), 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_removeElWords[text_removeElWords['Changed']=='yes']), len(text_removeElWords), 100*len(text_removeElWords[text_removeElWords['Changed']=='yes'])/len(text_removeElWords)))\n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:31:36.412531Z","iopub.execute_input":"2023-11-17T19:31:36.414049Z","iopub.status.idle":"2023-11-17T19:31:36.463537Z","shell.execute_reply.started":"2023-11-17T19:31:36.413998Z","shell.execute_reply":"2023-11-17T19:31:36.462583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"10. Stemming/Lemmatizing","metadata":{}},{"cell_type":"code","source":"from nltk.stem.porter import PorterStemmer\nstemmer = PorterStemmer() #set stemmer\nfrom nltk.stem import WordNetLemmatizer\nlemmatizer = WordNetLemmatizer() # set lemmatizer\n\ndef tokenize(text):\n    finalTokens = []\n    tokens = nltk.word_tokenize(text)\n    for w in tokens:\n        finalTokens.append(stemmer.stem(w)) # change this to lemmatizer.lemmatize(w) for Lemmatizing\n    text = \" \".join(finalTokens)\n    return text\n\ntext_stemming = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_stemming['TextBefore'] = text_removeElWords['TextAfter']\n\ntext_stemming['TextAfter'] = text_stemming['TextBefore'].apply(tokenize)\n    \n    \ntext_stemming['Changed'] = np.where(text_stemming['TextBefore'].str.replace(\" \",\"\")==text_stemming['TextAfter'].str.replace(\" \",\"\").str.replace(\"``\",'\"').str.replace(\"''\",'\"'), 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_stemming[text_stemming['Changed']=='yes']), len(text_stemming), 100*len(text_stemming[text_stemming['Changed']=='yes'])/len(text_stemming)))","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:35:34.924727Z","iopub.execute_input":"2023-11-17T19:35:34.925167Z","iopub.status.idle":"2023-11-17T19:35:34.976587Z","shell.execute_reply.started":"2023-11-17T19:35:34.925137Z","shell.execute_reply":"2023-11-17T19:35:34.975382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sincere_questions = train_df[train_df['target'] == 0]\ninsincere_questions = train_df[train_df['target'] == 1]","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:51:09.116693Z","iopub.execute_input":"2023-11-17T19:51:09.117108Z","iopub.status.idle":"2023-11-17T19:51:09.125150Z","shell.execute_reply.started":"2023-11-17T19:51:09.117070Z","shell.execute_reply":"2023-11-17T19:51:09.123752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sincere_questions.shape, insincere_questions.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:51:09.334723Z","iopub.execute_input":"2023-11-17T19:51:09.335170Z","iopub.status.idle":"2023-11-17T19:51:09.345219Z","shell.execute_reply.started":"2023-11-17T19:51:09.335136Z","shell.execute_reply":"2023-11-17T19:51:09.343571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sincere_questions['length'] = sincere_questions['question_text'].apply(lambda x : len(x))\ninsincere_questions['length'] = insincere_questions['question_text'].apply(lambda x : len(x))\ntrain_df['length'] = train_df['question_text'].apply(lambda x : len(x))","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:51:19.698766Z","iopub.execute_input":"2023-11-17T19:51:19.699177Z","iopub.status.idle":"2023-11-17T19:51:19.712172Z","shell.execute_reply.started":"2023-11-17T19:51:19.699146Z","shell.execute_reply":"2023-11-17T19:51:19.710459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport json\nimport string\nimport numpy as np\nimport pandas as pd\nimport plotly.offline as py\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.graph_objs as go\n\n# from pandas.io.json import json_normalize\nfrom pandas import json_normalize \n\nfrom plotly import tools\npy.init_notebook_mode(connected=True)\npd.options.mode.chained_assignment = None\npd.options.display.max_columns = 999\ncolor = sns.color_palette()\nnp.random.seed(13)\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2023-11-17T19:58:00.439384Z","iopub.execute_input":"2023-11-17T19:58:00.439826Z","iopub.status.idle":"2023-11-17T19:58:00.450222Z","shell.execute_reply.started":"2023-11-17T19:58:00.439775Z","shell.execute_reply":"2023-11-17T19:58:00.448856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The Meta Features Based On Word/Character","metadata":{}},{"cell_type":"code","source":"from tqdm import tqdm \nprint(\">> Generating Count Based And Demographical Features\")\nfor df in ([train_df]):\n    df['length'] = df['question_text'].apply(lambda x : len(x))\n    df['capitals'] = df['question_text'].apply(lambda comment: sum(1 for c in comment if c.isupper()))\n    df['caps_vs_length'] = df.apply(lambda row: float(row['capitals'])/float(row['length']),axis=1)\n    df['num_exclamation_marks'] = df['question_text'].apply(lambda comment: comment.count('!'))\n    df['num_question_marks'] = df['question_text'].apply(lambda comment: comment.count('?'))\n    df['num_punctuation'] = df['question_text'].apply(lambda comment: sum(comment.count(w) for w in '.,;:'))\n    df['num_symbols'] = df['question_text'].apply(lambda comment: sum(comment.count(w) for w in '*&$%'))\n    df['num_words'] = df['question_text'].apply(lambda comment: len(comment.split()))\n    df['num_unique_words'] = df['question_text'].apply(lambda comment: len(set(w for w in comment.split())))\n    df['words_vs_unique'] = df['num_unique_words'] / df['num_words']\n    df['num_smilies'] = df['question_text'].apply(lambda comment: sum(comment.count(w) for w in (':-)', ':)', ';-)', ';)')))\n    df['num_sad'] = df['question_text'].apply(lambda comment: sum(comment.count(w) for w in (':-<', ':()', ';-()', ';(')))","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:22:06.351314Z","iopub.execute_input":"2023-11-17T20:22:06.351805Z","iopub.status.idle":"2023-11-17T20:22:06.414851Z","shell.execute_reply.started":"2023-11-17T20:22:06.351754Z","shell.execute_reply":"2023-11-17T20:22:06.413191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[train_df.columns[2:]].head(8)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:22:11.268329Z","iopub.execute_input":"2023-11-17T20:22:11.268847Z","iopub.status.idle":"2023-11-17T20:22:11.291220Z","shell.execute_reply.started":"2023-11-17T20:22:11.268800Z","shell.execute_reply":"2023-11-17T20:22:11.289758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List Of Bad Words by Google-Profanity Words \nbad_words = ['cockknocker', 'n1gger', 'ing', 'fukker', 'nympho', 'fcuking', 'gook', 'freex', 'arschloch', 'fistfucked', 'chinc', 'raunch', 'fellatio', 'splooge', 'nutsack', 'lmfao', 'wigger', 'bastard', 'asses', 'fistfuckings', 'blue', 'waffle', 'beeyotch', 'pissin', 'dominatrix', 'fisting', 'vullva', 'paki', 'cyberfucker', 'chuj', 'penuus', 'masturbate', 'b00b*', 'fuks', 'sucked', 'fuckingshitmotherfucker', 'feces', 'panty', 'coital', 'wh00r.', 'whore', 'condom', 'hells', 'foreskin', 'wanker', 'hoer', 'sh1tz', 'shittings', 'wtf', 'recktum', 'dick*', 'pr0n', 'pasty', 'spik', 'phukked', 'assfuck', 'xxx', 'nigger*', 'ugly', 's_h_i_t', 'mamhoon', 'pornos', 'masterbates', 'mothafucks', 'Mother', 'Fukkah', 'chink', 'pussy', 'palace', 'azazel', 'fistfucking', 'ass-fucker', 'shag', 'chincs', 'duche', 'orgies', 'vag1na', 'molest', 'bollock', 'a-hole', 'seduce', 'Cock*', 'dog-fucker', 'shitz', 'Mother', 'Fucker', 'penial', 'biatch', 'junky', 'orifice', '5hit', 'kunilingus', 'cuntbag', 'hump', 'butt', 'fuck', 'titwank', 'schaffer', 'cracker', 'f.u.c.k', 'breasts', 'd1ld0', 'polac', 'boobs', 'ritard', 'fuckup', 'rape', 'hard', 'on', 'skanks', 'coksucka', 'cl1t', 'herpy', 's.o.b.', 'Motha', 'Fucker', 'penus', 'Fukker', 'p.u.s.s.y.', 'faggitt', 'b!tch', 'doosh', 'titty', 'pr1k', 'r-tard', 'gigolo', 'perse', 'lezzies', 'bollock*', 'pedophiliac', 'Ass', 'Monkey', 'mothafucker', 'amcik', 'b*tch', 'beaner', 'masterbat*', 'fucka', 'phuk', 'menses', 'pedophile', 'climax', 'cocksucking', 'fingerfucked', 'asswhole', 'basterdz', 'cahone', 'ahole', 'dickflipper', 'diligaf', 'Lesbian', 'sperm', 'pisser', 'dykes', 'Skanky', 'puuker', 'gtfo', 'orgasim', 'd0ng', 'testicle*', 'pen1s', 'piss-off', '@$$', 'fuck', 'trophy', 'arse*', 'fag', 'organ', 'potty', 'queerz', 'fannybandit', 'muthafuckaz', 'booger', 'pussypounder', 'titt', 'fuckoff', 'bootee', 'schlong', 'spunk', 'rumprammer', 'weed', 'bi7ch', 'pusse', 'blow', 'job', 'kusi*', 'assbanged', 'dumbass', 'kunts', 'chraa', 'cock', 'sucker', 'l3i+ch', 'cabron', 'arrse', 'cnut', 'how', 'to', 'murdep', 'fcuk', 'phuked', 'gang-bang', 'kuksuger', 'mothafuckers', 'ghey', 'clit', 'licker', 'feg', 'ma5terbate', 'd0uche', 'pcp', 'ejaculate', 'nigur', 'clits', 'd0uch3', 'b00bs', 'fucked', 'assbang', 'mutha', 'goddamned', 'cazzo', 'lmao', 'godamn', 'kill', 'coon', 'penis-breath', 'kyke', 'heshe', 'homo', 'tawdry', 'pissing', 'cumshot', 'motherfucker', 'menstruation', 'n1gr', 'rectus', 'oral', 'twats', 'scrot', 'God', 'damn', 'jerk', 'nigga', 'motherfuckin', 'kawk', 'homey', 'hooters', 'rump', 'dickheads', 'scrud', 'fist', 'fuck', 'carpet', 'muncher', 'cipa', 'cocaine', 'fanyy', 'frigga', 'massa', '5h1t', 'brassiere', 'inbred', 'spooge', 'shitface', 'tush', 'Fuken', 'boiolas', 'fuckass', 'wop*', 'cuntlick', 'fucker', 'bodily', 'bullshits', 'hom0', 'sumofabiatch', 'jackass', 'dilld0', 'puuke', 'cums', 'pakie', 'cock-sucker', 'pubic', 'pron', 'puta', 'penas', 'weiner', 'vaj1na', 'mthrfucker', 'souse', 'loin', 'clitoris', 'f.ck', 'dickface', 'rectal', 'whored', 'bookie', 'chota', 'bags', 'sh!t', 'pornography', 'spick', 'seamen', 'Phukker', 'beef', 'curtain', 'eat', 'hair', 'pie', 'mother', 'fucker', 'faigt', 'yeasty', 'Clit', 'kraut', 'CockSucker', 'Ekrem*', 'screwing', 'scrote', 'fubar', 'knob', 'end', 'sleazy', 'dickwhipper', 'ass', 'fuck', 'fellate', 'lesbos', 'nobjokey', 'dogging', 'fuck', 'hole', 'hymen', 'damn', 'dego', 'sphencter', 'queef*', 'gaylord', 'va1jina', 'a55', 'fuck', 'douchebag', 'blowjob', 'mibun', 'fucking', 'dago', 'heroin', 'tw4t', 'raper', 'muff', 'fitt*', 'wetback*', 'mo-fo', 'fuk*', 'klootzak', 'sux', 'damnit', 'pimmel', 'assh0lez', 'cntz', 'fux', 'gonads', 'bullshit', 'nigg3r', 'fack', 'weewee', 'shi+', 'shithead', 'pecker', 'Shytty', 'wh0re', 'a2m', 'kkk', 'penetration', 'kike', 'naked', 'kooch', 'ejaculation', 'bang', 'hoare', 'jap', 'foad', 'queef', 'buttwipe', 'Shity', 'dildo', 'dickripper', 'crackwhore', 'beaver', 'kum', 'sh!+', 'qweers', 'cocksuka', 'sexy', 'masterbating', 'peeenus', 'gays', 'cocksucks', 'b17ch', 'nad', 'j3rk0ff', 'fannyflaps', 'God-damned', 'masterbate', 'erotic', 'sadism', 'turd', 'flipping', 'the', 'bird', 'schizo', 'whiz', 'fagg1t', 'cop', 'some', 'wood', 'banger', 'Shyty', 'f', 'you', 'scag', 'soused', 'scank', 'clitorus', 'kumming', 'quim', 'penis', 'bestial', 'bimbo', 'gfy', 'spiks', 'shitings', 'phuking', 'paddy', 'mulkku', 'anal', 'leakage', 'bestiality', 'smegma', 'bull', 'shit', 'pillu*', 'schmuck', 'cuntsicle', 'fistfucker', 'shitdick', 'dirsa', 'm0f0']\nprint(\">> Words in bad_word list:\", len(bad_words))","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:22:23.869704Z","iopub.execute_input":"2023-11-17T20:22:23.870114Z","iopub.status.idle":"2023-11-17T20:22:23.889147Z","shell.execute_reply.started":"2023-11-17T20:22:23.870081Z","shell.execute_reply":"2023-11-17T20:22:23.887932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\">> Generating Features on Bad Words\")\nfor df in ([train_df]):\n    df[\"badwordcount\"] = df['question_text'].apply(lambda comment: sum(comment.count(w) for w in bad_words))\n    df['num_chars'] =    df['question_text'].apply(len)\n    df[\"normchar_badwords\"] = df[\"badwordcount\"]/df['num_chars']\n    df[\"normword_badwords\"] = df[\"badwordcount\"]/df['num_words']","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:22:47.996321Z","iopub.execute_input":"2023-11-17T20:22:47.996820Z","iopub.status.idle":"2023-11-17T20:22:48.150870Z","shell.execute_reply.started":"2023-11-17T20:22:47.996767Z","shell.execute_reply":"2023-11-17T20:22:48.149652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[['badwordcount','num_chars','normchar_badwords','normword_badwords']].head(8)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:22:52.323681Z","iopub.execute_input":"2023-11-17T20:22:52.324109Z","iopub.status.idle":"2023-11-17T20:22:52.340347Z","shell.execute_reply.started":"2023-11-17T20:22:52.324074Z","shell.execute_reply":"2023-11-17T20:22:52.339002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Tagging Parts Of Speech And More Feature Engineering..\nI suspect that the insincere questions have significant adverbs/adjective that makes them toxic. I am hopeful that these features might model understand various POS structures in the question_text","metadata":{}},{"cell_type":"code","source":"import string\nfrom nltk import pos_tag\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import TweetTokenizer\n\ndef tag_part_of_speech(text):\n    text_splited = text.split(' ')\n    text_splited = [''.join(c for c in s if c not in string.punctuation) for s in text_splited]\n    text_splited = [s for s in text_splited if s]\n    pos_list = pos_tag(text_splited)\n    noun_count = len([w for w in pos_list if w[1] in ('NN','NNP','NNPS','NNS')])\n    adjective_count = len([w for w in pos_list if w[1] in ('JJ','JJR','JJS')])\n    verb_count = len([w for w in pos_list if w[1] in ('VB','VBD','VBG','VBN','VBP','VBZ')])\n    return[noun_count, adjective_count, verb_count]","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:23:35.647151Z","iopub.execute_input":"2023-11-17T20:23:35.647578Z","iopub.status.idle":"2023-11-17T20:23:35.656384Z","shell.execute_reply.started":"2023-11-17T20:23:35.647541Z","shell.execute_reply":"2023-11-17T20:23:35.655070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\">> Generating POS Features\")\nfor df in ([train_df]):\n    df['nouns'], df['adjectives'], df['verbs'] = zip(*df['question_text'].apply(\n        lambda comment: tag_part_of_speech(comment)))\n    df['nouns_vs_length'] = df['nouns'] / df['length']\n    df['adjectives_vs_length'] = df['adjectives'] / df['length']\n    df['verbs_vs_length'] = df['verbs'] /df['length']\n    df['nouns_vs_words'] = df['nouns'] / df['num_words']\n    df['adjectives_vs_words'] = df['adjectives'] / df['num_words']\n    df['verbs_vs_words'] = df['verbs'] / df['num_words']\n    # More Handy Features\n    df[\"count_words_title\"] = df[\"question_text\"].apply(lambda x: len([w for w in str(x).split() if w.istitle()]))\n    df[\"mean_word_len\"] = df[\"question_text\"].apply(lambda x: np.mean([len(w) for w in str(x).split()]))\n    df['punct_percent']= df['num_punctuation']*100/df['num_words']","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:23:44.291867Z","iopub.execute_input":"2023-11-17T20:23:44.292425Z","iopub.status.idle":"2023-11-17T20:23:45.280726Z","shell.execute_reply.started":"2023-11-17T20:23:44.292380Z","shell.execute_reply":"2023-11-17T20:23:45.279836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[['nouns','nouns_vs_length','adjectives_vs_length','verbs_vs_length','nouns_vs_words','adjectives_vs_words','verbs_vs_words']].head(8)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:24:02.703013Z","iopub.execute_input":"2023-11-17T20:24:02.703435Z","iopub.status.idle":"2023-11-17T20:24:02.721286Z","shell.execute_reply.started":"2023-11-17T20:24:02.703403Z","shell.execute_reply":"2023-11-17T20:24:02.720156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(figsize= [20,15])\nsns.heatmap(train_df.drop(['qid','question_text'], axis=1).corr(), annot=True, fmt=\".2f\", ax=ax, \n            cbar_kws={'label': 'Correlation Coefficient'}, cmap='viridis')\nax.set_title(\"Correlation Matrix for Insincerity and New Features\", fontsize=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:24:19.393054Z","iopub.execute_input":"2023-11-17T20:24:19.393613Z","iopub.status.idle":"2023-11-17T20:24:27.990722Z","shell.execute_reply.started":"2023-11-17T20:24:19.393574Z","shell.execute_reply":"2023-11-17T20:24:27.989644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Statistics","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport string\nimport random\nimport operator\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.decomposition import NMF, LatentDirichletAllocation, TruncatedSVD\nfrom statistics import *\nfrom sklearn.feature_extraction.text import CountVectorizer\nimport concurrent.futures\nimport time\nimport pyLDAvis.sklearn\nfrom pylab import bone, pcolor, colorbar, plot, show, rcParams, savefig\n!pip install textstat\nimport textstat\nimport warnings\nimport nltk\nwarnings.filterwarnings('ignore')\n\n%matplotlib inline\nimport os\nprint(os.listdir(\"../input\"))\n\n# Plotly based imports for visualization\nfrom plotly import tools\n!pip install chart-studio\nimport chart_studio as py\nfrom plotly.offline import init_notebook_mode, iplot\ninit_notebook_mode(connected=True)\nimport plotly.graph_objs as go\nimport plotly.figure_factory as ff\n\n# spaCy based imports\nimport spacy\nfrom spacy.lang.en.stop_words import STOP_WORDS\nfrom spacy.lang.en import English\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:38:34.264018Z","iopub.execute_input":"2023-11-17T20:38:34.264473Z","iopub.status.idle":"2023-11-17T20:39:02.839826Z","shell.execute_reply.started":"2023-11-17T20:38:34.264440Z","shell.execute_reply":"2023-11-17T20:39:02.838257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SpaCy Parser for questions\npunctuations = string.punctuation\nstopwords = list(STOP_WORDS)\n\nparser = English()\ndef spacy_tokenizer(sentence):\n    mytokens = parser(sentence)\n    mytokens = [ word.lemma_.lower().strip() if word.lemma_ != \"-PRON-\" else word.lower_ for word in mytokens ]\n    mytokens = [ word for word in mytokens if word not in stopwords and word not in punctuations ]\n    mytokens = \" \".join([i for i in mytokens])\n    return mytokens","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:39:28.997386Z","iopub.execute_input":"2023-11-17T20:39:28.998437Z","iopub.status.idle":"2023-11-17T20:39:29.251193Z","shell.execute_reply.started":"2023-11-17T20:39:28.998379Z","shell.execute_reply":"2023-11-17T20:39:29.249856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"quora_train = train_df","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:40:11.664732Z","iopub.execute_input":"2023-11-17T20:40:11.665200Z","iopub.status.idle":"2023-11-17T20:40:11.670212Z","shell.execute_reply.started":"2023-11-17T20:40:11.665168Z","shell.execute_reply":"2023-11-17T20:40:11.669141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tqdm.pandas()\nsincere_questions = quora_train[\"question_text\"][quora_train[\"target\"] == 0].progress_apply(spacy_tokenizer)\ninsincere_questions = quora_train[\"question_text\"][quora_train[\"target\"] == 1].progress_apply(spacy_tokenizer)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:40:12.478527Z","iopub.execute_input":"2023-11-17T20:40:12.479073Z","iopub.status.idle":"2023-11-17T20:40:12.862333Z","shell.execute_reply.started":"2023-11-17T20:40:12.479027Z","shell.execute_reply":"2023-11-17T20:40:12.861076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef plot_readability(a,b,title,bins=0.1,colors=['#3A4750', '#F64E8B']):\n    trace1 = ff.create_distplot([a,b], [\"Sincere questions\",\"Insincere questions\"], bin_size=bins, colors=colors, show_rug=False)\n    trace1['layout'].update(title=title)\n    iplot(trace1, filename='Distplot')\n    table_data= [[\"Statistical Measures\",\"Sincere questions\",\"Insincere questions\"],\n                [\"Mean\",mean(a),mean(b)],\n                [\"Standard Deviation\",pstdev(a),pstdev(b)],\n                [\"Variance\",pvariance(a),pvariance(b)],\n                [\"Median\",median(a),median(b)],\n                [\"Maximum value\",max(a),max(b)],\n                [\"Minimum value\",min(a),min(b)]]\n    trace2 = ff.create_table(table_data)\n    iplot(trace2, filename='Table')","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:40:27.437593Z","iopub.execute_input":"2023-11-17T20:40:27.438053Z","iopub.status.idle":"2023-11-17T20:40:27.446719Z","shell.execute_reply.started":"2023-11-17T20:40:27.438015Z","shell.execute_reply":"2023-11-17T20:40:27.445121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"syllable_sincere = np.array(quora_train[\"question_text\"][quora_train[\"target\"] == 0].progress_apply(textstat.syllable_count))\nsyllable_insincere = np.array(quora_train[\"question_text\"][quora_train[\"target\"] == 1].progress_apply(textstat.syllable_count))\nplot_readability(syllable_sincere,syllable_insincere,\"Syllable Analysis\",5)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:40:35.276690Z","iopub.execute_input":"2023-11-17T20:40:35.277331Z","iopub.status.idle":"2023-11-17T20:40:36.999369Z","shell.execute_reply.started":"2023-11-17T20:40:35.277280Z","shell.execute_reply":"2023-11-17T20:40:36.998094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"length_sincere = np.array(quora_train[\"question_text\"][quora_train[\"target\"] == 0].progress_apply(len))\nlength_insincere = np.array(quora_train[\"question_text\"][quora_train[\"target\"] == 1].progress_apply(len))\nplot_readability(length_sincere,length_insincere,\"Question Length\",40,['#C65D17','#DDB967'])\n","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:40:56.309400Z","iopub.execute_input":"2023-11-17T20:40:56.309983Z","iopub.status.idle":"2023-11-17T20:40:56.506235Z","shell.execute_reply.started":"2023-11-17T20:40:56.309941Z","shell.execute_reply":"2023-11-17T20:40:56.504115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spw_sincere = np.array(quora_train[\"question_text\"][quora_train[\"target\"] == 0].progress_apply(textstat.avg_syllables_per_word))\nspw_insincere = np.array(quora_train[\"question_text\"][quora_train[\"target\"] == 1].progress_apply(textstat.avg_syllables_per_word))\nplot_readability(spw_sincere,spw_insincere,\"Average syllables per word\",0.2,['#8D99AE','#EF233C'])","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:41:13.628245Z","iopub.execute_input":"2023-11-17T20:41:13.628746Z","iopub.status.idle":"2023-11-17T20:41:13.843993Z","shell.execute_reply.started":"2023-11-17T20:41:13.628698Z","shell.execute_reply":"2023-11-17T20:41:13.842971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lpw_sincere = np.array(quora_train[\"question_text\"][quora_train[\"target\"] == 0].progress_apply(textstat.avg_letter_per_word))\nlpw_insincere = np.array(quora_train[\"question_text\"][quora_train[\"target\"] == 1].progress_apply(textstat.avg_letter_per_word))\nplot_readability(lpw_sincere,lpw_insincere,\"Average letters per word\",2,['#8491A3','#2B2D42'])","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:41:27.604875Z","iopub.execute_input":"2023-11-17T20:41:27.605320Z","iopub.status.idle":"2023-11-17T20:41:27.802421Z","shell.execute_reply.started":"2023-11-17T20:41:27.605275Z","shell.execute_reply":"2023-11-17T20:41:27.800992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Readability features\n# This basically returns the readability statistics for given text.\n\n# 6.1 The Flesch Reading Ease formula\n# The following table can be helpful to assess the ease of readability in a document.\n\n# Score - Difficulty\n# 90-100 - Very Easy\n# 80-89 - Easy\n# 70-79 - Fairly Easy\n# 60-69 - Standard\n# 50-59 - Fairly Difficult\n# 30-49 - Difficult\n# 0-29 - Very Confusing","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:41:48.301250Z","iopub.execute_input":"2023-11-17T20:41:48.301664Z","iopub.status.idle":"2023-11-17T20:41:48.307250Z","shell.execute_reply.started":"2023-11-17T20:41:48.301634Z","shell.execute_reply":"2023-11-17T20:41:48.305899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fre_sincere = np.array(quora_train[\"question_text\"][quora_train[\"target\"] == 0].progress_apply(textstat.flesch_reading_ease))\nfre_insincere = np.array(quora_train[\"question_text\"][quora_train[\"target\"] == 1].progress_apply(textstat.flesch_reading_ease))\nplot_readability(fre_sincere,fre_insincere,\"Flesch Reading Ease\",20)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:41:57.731843Z","iopub.execute_input":"2023-11-17T20:41:57.732247Z","iopub.status.idle":"2023-11-17T20:41:57.965759Z","shell.execute_reply.started":"2023-11-17T20:41:57.732216Z","shell.execute_reply":"2023-11-17T20:41:57.964367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fkg_sincere = np.array(quora_train[\"question_text\"][quora_train[\"target\"] == 0].progress_apply(textstat.flesch_kincaid_grade))\nfkg_insincere = np.array(quora_train[\"question_text\"][quora_train[\"target\"] == 1].progress_apply(textstat.flesch_kincaid_grade))\nplot_readability(fkg_sincere,fkg_insincere,\"Flesch Kincaid Grade\",4,['#C1D37F','#491F21'])","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:42:16.926069Z","iopub.execute_input":"2023-11-17T20:42:16.926495Z","iopub.status.idle":"2023-11-17T20:42:17.152910Z","shell.execute_reply.started":"2023-11-17T20:42:16.926461Z","shell.execute_reply":"2023-11-17T20:42:17.151684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Automated Readability Index\n# Returns the ARI (Automated Readability Index) which outputs a number that approximates the grade level needed to comprehend the text.For example if the ARI is 6.5, then the grade level to comprehend the text is 6th to 7th grade.\n\nari_sincere = np.array(quora_train[\"question_text\"][quora_train[\"target\"] == 0].progress_apply(textstat.automated_readability_index))\nari_insincere = np.array(quora_train[\"question_text\"][quora_train[\"target\"] == 1].progress_apply(textstat.automated_readability_index))\nplot_readability(ari_sincere,ari_insincere,\"Automated Readability Index\",10,['#488286','#FF934F'])\n","metadata":{"execution":{"iopub.status.busy":"2023-11-17T20:42:53.807031Z","iopub.execute_input":"2023-11-17T20:42:53.807457Z","iopub.status.idle":"2023-11-17T20:42:54.016275Z","shell.execute_reply.started":"2023-11-17T20:42:53.807423Z","shell.execute_reply":"2023-11-17T20:42:54.014988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom plotly import tools\nimport plotly.offline as py\nimport plotly.graph_objs as go\nfrom nltk.corpus import stopwords \nfrom nltk import word_tokenize, sent_tokenize, pos_tag, ne_chunk, FreqDist\nfrom textblob import TextBlob\nimport collections\nfrom sklearn.feature_extraction.text import TfidfVectorizer, CountVectorizer\nfrom sklearn.preprocessing import normalize\nfrom wordcloud import WordCloud\n%matplotlib inline\npy.init_notebook_mode(connected=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T21:03:22.331952Z","iopub.execute_input":"2023-11-17T21:03:22.333453Z","iopub.status.idle":"2023-11-17T21:03:22.475580Z","shell.execute_reply.started":"2023-11-17T21:03:22.333391Z","shell.execute_reply":"2023-11-17T21:03:22.474278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"insicere_quiz=train_df[train_df['target']==1]\nsincere_quiz=train_df[train_df['target']==0]","metadata":{"execution":{"iopub.status.busy":"2023-11-17T21:03:57.798621Z","iopub.execute_input":"2023-11-17T21:03:57.799808Z","iopub.status.idle":"2023-11-17T21:03:57.810398Z","shell.execute_reply.started":"2023-11-17T21:03:57.799745Z","shell.execute_reply":"2023-11-17T21:03:57.809399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#most common words\ntokens=[]\nfor i in train_df[0:10]['question_text']:\n    for j in word_tokenize(i):\n        tokens.append(j)\n\n\nfrequency_distribution=FreqDist(tokens).most_common()\nprint(frequency_distribution)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T21:04:32.820822Z","iopub.execute_input":"2023-11-17T21:04:32.821260Z","iopub.status.idle":"2023-11-17T21:04:32.831565Z","shell.execute_reply.started":"2023-11-17T21:04:32.821228Z","shell.execute_reply":"2023-11-17T21:04:32.830170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculating Sentment Analysis with TextBlob\nfor i in train_df[0:5]['question_text']:\n    print(i,\" => \",TextBlob(i).sentiment)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T21:05:21.809298Z","iopub.execute_input":"2023-11-17T21:05:21.809715Z","iopub.status.idle":"2023-11-17T21:05:21.870900Z","shell.execute_reply.started":"2023-11-17T21:05:21.809682Z","shell.execute_reply":"2023-11-17T21:05:21.869135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extracting the sentiment polarity of a text\nfor i in train_df[0:5]['question_text']:\n    print(TextBlob(i).sentiment.polarity)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T21:05:33.903720Z","iopub.execute_input":"2023-11-17T21:05:33.904144Z","iopub.status.idle":"2023-11-17T21:05:33.912870Z","shell.execute_reply.started":"2023-11-17T21:05:33.904112Z","shell.execute_reply":"2023-11-17T21:05:33.911521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extracting the sentiment subjectivity of a text\nfor i in train_df[0:5]['question_text']:\n    print(TextBlob(i).sentiment.subjectivity)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T21:05:43.700486Z","iopub.execute_input":"2023-11-17T21:05:43.700905Z","iopub.status.idle":"2023-11-17T21:05:43.709902Z","shell.execute_reply.started":"2023-11-17T21:05:43.700871Z","shell.execute_reply":"2023-11-17T21:05:43.708310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in train_df[0:2]['question_text']:\n    print(TextBlob(i).ngrams(n=3))","metadata":{"execution":{"iopub.status.busy":"2023-11-17T21:05:52.177718Z","iopub.execute_input":"2023-11-17T21:05:52.178201Z","iopub.status.idle":"2023-11-17T21:05:52.186767Z","shell.execute_reply.started":"2023-11-17T21:05:52.178166Z","shell.execute_reply":"2023-11-17T21:05:52.185060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corpus=[]\nfor i in train_df[0:5]['question_text']:\n    corpus.append(i)\n\ncvect = CountVectorizer(ngram_range=(1,1))\ncounts = cvect.fit_transform(corpus)\nnormalized_counts = normalize(counts, norm='l1', axis=1)\n\ntfidf = TfidfVectorizer(ngram_range=(1,1), smooth_idf=False)\ntfs = tfidf.fit_transform(corpus)\nnew_tfs = normalized_counts.multiply(tfidf.idf_)\n\nfeature_names = tfidf.get_feature_names_out()\ncorpus_index = [n for n in corpus]\ndf = pd.DataFrame(new_tfs.T.todense(), index=feature_names, columns=corpus_index)\n\nprint(df)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T21:06:39.264648Z","iopub.execute_input":"2023-11-17T21:06:39.265299Z","iopub.status.idle":"2023-11-17T21:06:39.300215Z","shell.execute_reply.started":"2023-11-17T21:06:39.265240Z","shell.execute_reply":"2023-11-17T21:06:39.298607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Bow with collection\ntoken=[]\nfor i in train_df[0:5]['question_text']:\n    token.append(i)\n\nbow = [collections.Counter(words.split(\" \")) for words in token]\ntotal_bow=sum(bow,collections.Counter())\nprint(total_bow)","metadata":{"execution":{"iopub.status.busy":"2023-11-17T21:07:03.878854Z","iopub.execute_input":"2023-11-17T21:07:03.879313Z","iopub.status.idle":"2023-11-17T21:07:03.888848Z","shell.execute_reply.started":"2023-11-17T21:07:03.879276Z","shell.execute_reply":"2023-11-17T21:07:03.887147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}