{"cells":[{"metadata":{"trusted":true,"_uuid":"283971bd63ade0fbae074185054f59fc17cc1f7c"},"cell_type":"code","source":"from gensim.utils import simple_preprocess\nimport matplotlib.pyplot as plt\nimport nltk\nimport pandas as pd\nimport spacy\n\ntrain_df = pd.read_csv('../input/train.csv')\ntest_df = pd.read_csv('../input/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7952c978da1a0c33162a106a794bba2fcdf12bf3"},"cell_type":"code","source":"train_df.target.hist(bins=3)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fcb96b4640fe7dc054d28ec0229e1df625e15a44"},"cell_type":"code","source":"train_df[['target']].boxplot()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b233c77fa4b4f59cf5d6aa2b5c6c5a666ac808b0"},"cell_type":"code","source":"train_df.groupby('target').count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"377ff7c3d7466819818a89a0321e8179741965e8"},"cell_type":"code","source":"nlp = spacy.load('en', disable=['ner'])\n\nnegative_questions = train_df.loc[train_df.target == 1].question_text\nall_tokens = [token\n              for question in negative_questions\n              for token in nlp(question)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"f410cab1ba8fb92c4180f08ee1ffa194307ed622"},"cell_type":"code","source":"clean_words = [token.lemma_\n               for token in all_tokens\n               if not token.is_punct\n               and token.lemma_ != '-PRON-'\n               and not nlp.vocab[token.lemma_].is_stop]\nclean_word_dist = nltk.FreqDist(clean_words)\nclean_word_dist.most_common(30)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"710aeec36f14724bd8d570a1ec15c5ffded36147"},"cell_type":"code","source":"puncts = [token.text for token in all_tokens if token.is_punct]\npunct_dist = nltk.FreqDist(puncts)\npunct_dist.most_common(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"371d14033d5af70efbd273dd16e08a8fa91a3e1c"},"cell_type":"code","source":"subjects = [token.lemma_\n            for token in all_tokens\n            if not token.is_stop\n            and 'subj' in token.dep_ and 'subj' in token.dep_\n            and token.pos_ in {'PROPN', 'NOUN'}]\nsubject_dist = nltk.FreqDist(subjects)\nsubject_dist.most_common(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"fcecf67205b5e46c0b655c8c518b703f74d188bc"},"cell_type":"code","source":"train_lengths = [len(simple_preprocess(doc))\n                 for doc in train_df.question_text]\ntest_lengths = [len(simple_preprocess(doc))\n                for doc in test_df.question_text]\ncombined_lengths = train_lengths + test_lengths\npd.Series(combined_lengths).describe(percentiles=[.25, .5, .75, .9, .95, .99])","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"7a65016500416c105bbd087798cea7eba268c3aa"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}