{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom nltk.corpus import brown\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\nimport numpy as np\nimport pandas as pd\nimport re\nimport spacy\nfrom spacy.lang.en.stop_words import STOP_WORDS\nfrom tqdm import tqdm\nimport seaborn as sns\nsns.set_style('darkgrid')\nimport matplotlib.pyplot as plt\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"df = pd.read_csv('../input/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54c44c482b930382f35f01e0b43f2a5c1383004b"},"cell_type":"code","source":"brown_vocab = brown.words(categories=brown.categories())\nlen(brown_vocab)\nbrown_vocab_lower = list(map(lambda x:x.lower(),brown_vocab))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"abab9a21b82768cb3983219def9106323cfe2924"},"cell_type":"code","source":"brown_vocab_lower = set(brown_vocab_lower)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"965947eac28473d40bd5cf155861abcfe0543449"},"cell_type":"markdown","source":"# Get Regex"},{"metadata":{"trusted":true,"_uuid":"1ef0c3ed86c021faf6150608a024f469d2e1299f"},"cell_type":"code","source":"# pronouns\nfp = ['I','we','me', 'us', 'my', 'our', 'mine', 'ours'] # singular & plural\nsp = ['you' ,'your' ,'yours'] # singular & plural\ntps=['he', 'she','it','him','her','his', 'hers','its'] # singular\ntpp = ['they','them','their','theirs'] # plural\nintensive_pr = ['myself','yourself','herself','himself','itself','ourselves','yourselves','themselves'] # singular & plural\ninterrogative = ['what','whatever','which','whichever','who','whoever','whom','whomever','whose']\nneg = [\"cant\",\"couldn't\",\"shan't\",\"shouldn't\",\"wouldn't\",\"haven't\",\"didn't\",\"not\",\"never\",\"won't\"]\n# ('|').join(fp)\n# ('|').join(sp)\n# ('|').join(tps)\n# ('|').join(tpp)\n# ('|').join(interrogative)\n# ('|').join(intensive_pr) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f60646106c84d2a8c05c80a3630e8480c12d14e6"},"cell_type":"code","source":"doc = \"hello-hello you shouldn't gove,98 a fuck dam about those bastards.\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"30d6af7af7a88683ed429a5892a20ee7e1da7758"},"cell_type":"code","source":"(\" \").join(re.findall(r\"[a-zA-Z0-9\\']+\", doc))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e3e1cb3f30e48b660fa7d96ca4a864fdd3975535"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54859ae67ec42be87ae7f3bfcdcba405b149ec33"},"cell_type":"code","source":"def get_features(doc):\n    \n    i1 = len(re.findall(r'(?=\\bI\\b|\\bwe\\b|\\bme\\b|\\bus\\b|\\bmy\\b|\\bour\\b|\\bmine\\b|\\bours\\b)', doc)) # FP\n    i2 = len(re.findall(r'(?=\\byou\\b|\\byour\\b|\\byours\\b)', doc)) # sP\n    i3 = len(re.findall(r'(?=\\bhe\\b|\\bshe\\b|\\bit\\b|\\bhim\\b|\\bher\\b|\\bhis\\b|\\bhers\\b|\\bits\\b)', doc)) # tps\n    i4 = len(re.findall(r'(?=\\bthey\\b|\\bthem\\b|\\btheir\\b|\\btheirs\\b)', doc)) # tpp\n    i5 = len(re.findall(r'(?=\\bwhat\\b|\\bwhatever\\b|\\bwhich\\b|\\bwhichever\\b|\\bwho\\b|\\bwhoever\\b|\\bwhom\\b|\\bwhomever\\b|\\bwhoses\\b)', doc)) # interrogative\n    i6 = len(re.findall(r'(?=\\bmyself\\b|\\byourself\\b|\\bherself\\b|\\bhimself\\b|\\bitself\\b|\\bourselves\\b|\\byourselves\\b|\\bthemselves\\b)', doc)) # intensive\n    i7 = len(re.findall(r'(?=\\bfor\\b|\\band\\b|\\bnor\\b|\\bbut\\b|\\byet\\b|\\bso\\b|\\bbefore\\b|\\bonce\\b|\\bsince\\b|\\bthough\\b|\\bwhile\\b|\\bas\\b|\\bbecause\\b|\\bafter\\b)', doc)) # conjunctions\n    \n    i8 = len(doc) - len( re.findall('[a-zA-Z]', doc)) - doc.count(' ') - len(re.findall('[0-9]', doc))\n    i9 = doc.count(',')\n    i10 = doc.count('?')\n    i11 = len(doc.split())\n    # negs\n    i12 = len(re.findall(r'(?=\\bcan\\'t\\b|\\bcouldn\\'t\\b|\\bshan\\'t\\b|\\bshouldn\\'t\\b|\\bwouldn\\'t\\b|\\bhaven\\'t\\b|\\bdidn\\'t\\b|\\bnot\\b|\\bnever\\b|\\bwon\\'t\\b|\\bdon\\'t\\b|\\bhadn\\'t\\b|\\bcant\\b|\\bcouldnt\\b|\\bshant\\b|\\bshouldnt\\b|\\bwouldnt\\b|\\bhavent\\b|\\bdidnt\\b|\\bwont\\b|\\bdont\\b|\\bhadnt\\b)', doc))\n    \n#     spelling mistakes\n    temp_doc = (\" \").join(re.findall(r\"[a-zA-Z0-9\\']+\", doc))\n    i13 = len(set(temp_doc.split())-brown_vocab_lower)\n    \n    return (i1,i2,i3,i4,i5,i6,i7,i8,i9,i10,i11,i12,i13)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"da9c50b401d5d6cfbc4aeeac6fc71e048f55c971"},"cell_type":"code","source":"tqdm.pandas()\ndf['feat_vect'] = df['question_text'].progress_apply(get_features)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3ea8257e13813fb7380903c3d9b7c2495c17923a"},"cell_type":"code","source":"df['fp'] = df['feat_vect'].apply(lambda x: x[0])\ndf['sp'] = df['feat_vect'].apply(lambda x: x[1])\ndf['tps'] = df['feat_vect'].apply(lambda x: x[2])\ndf['tpp'] = df['feat_vect'].apply(lambda x: x[3])\ndf['interrogative'] = df['feat_vect'].apply(lambda x: x[4])\ndf['intensive'] = df['feat_vect'].apply(lambda x: x[5])\ndf['conjunction'] = df['feat_vect'].apply(lambda x: x[6])\ndf['special_chars'] = df['feat_vect'].apply(lambda x: x[7])\ndf['commas'] = df['feat_vect'].apply(lambda x: x[8])\ndf['qm'] = df['feat_vect'].apply(lambda x: x[9])\ndf['len'] = df['feat_vect'].apply(lambda x: x[10])\ndf['negs'] = df['feat_vect'].apply(lambda x: x[11])\ndf['sm'] = df['feat_vect'].apply(lambda x: x[12])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1a03536423aefc47742869fa0fecc8436a531322"},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"74b0c7dc8e0909ed55f0fa23d37db6f1becfd603"},"cell_type":"code","source":"sincere = df[df['target']==0]\ninsincere = df[df['target']==1]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9b66df7bb4aecf0b8ffead46a681088cb6c0d63d"},"cell_type":"markdown","source":"# Get Stats for both classes"},{"metadata":{"trusted":true,"_uuid":"50d447ee1033b25e918fa3cd7883cd58b0b6d39e"},"cell_type":"code","source":"def get_stats_df(class_df):\n    columns = ['fp', 'sp', 'tps', 'tpp', 'interrogative', 'intensive', 'conjunction', 'commas', 'qm',\n               'special_chars', 'len','negs','sm']\n    stats_dict = dict()\n    for each_col in columns:\n        col_name = each_col+'_prob'\n        temp = class_df.groupby(each_col).count()\n        temp[col_name] = temp['qid']/temp['qid'].sum()\n        temp.sort_values(by=col_name, ascending=False, inplace=True)\n        stats_dict[each_col] = temp[['qid',col_name]]\n        \n    return stats_dict","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f0f94b6c661fda510d3c4ee503db97149b71416f"},"cell_type":"code","source":"sincere_stats = get_stats_df(sincere)\ninsincere_stats = get_stats_df(insincere)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d86cb374eaee3b86f80068393d733df40eaeff02"},"cell_type":"markdown","source":"# Function for visualizing features"},{"metadata":{"trusted":true,"_uuid":"615a4267c33469bd38c1c7ce29204c1645a229bd"},"cell_type":"code","source":"def get_plot(feat_name):\n    \n    col_name = feat_name+'_prob'\n    fig = plt.figure(figsize=(5,4), dpi=100)\n    ax1 = plt.axes()\n    ax1.set_title(feat_name)\n    ax1.scatter(x=sincere_stats[feat_name].index, y =sincere_stats[feat_name][col_name], s=10, c='b', marker=\"x\", label='sincere')\n    ax1.scatter(x=insincere_stats[feat_name].index, y =insincere_stats[feat_name][col_name], s=10, c='r', marker=\"o\", label='insincere')\n    ax1.set_xlabel('count_' + feat_name, fontsize=10)\n    ax1.set_ylabel('prob', fontsize=10)\n    plt.legend(loc='upper right')\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1c3adbeac76815dd38a467f03c77074543df99b7"},"cell_type":"code","source":"    for feat in sincere_stats.keys():\n        get_plot(feat)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}