{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")\nprint(\"Train shape : \",train.shape)\nprint(\"Test shape : \",test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ccefa98d2f1e87d5435be6d3d16dff744860a1d6"},"cell_type":"code","source":"train.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"829e67ac3ae5740d493dfe6d9a51b35134926cb3"},"cell_type":"code","source":"test.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"647e6acb90cbc7a24857926148e2e72ca7ff9169"},"cell_type":"code","source":"from collections import defaultdict\nimport operator\nfrom nltk.tokenize import LineTokenizer, RegexpTokenizer\n\n# define a tokenizer\npattern = r'''(?x)            # set flag to allow verbose regexps\n        (?:\\w+\\d*\\.\\d+)+      # locations that have period within token \n      | [\\w\\d]+(?:-[\\w\\d]+)*  # ids with optional internal hyphens\n      | \\=+                   # separator such as =====\n      | [][.,;\"'?():_`]       # these are separate tokens; includes ], [\n    '''\ntokenizer = RegexpTokenizer(pattern)\n\n# statistics\ntrain_questions = train['question_text'].tolist()\nvocab = defaultdict(int)\nnbr_line_dist = []\nnbr_token_dist = []\nfor question in train_questions:\n    # number of lines in question\n    lines = list(LineTokenizer().tokenize(question))\n    nbr_line_dist.append(len(lines))\n    # number of tokens in questions \n    tokens = tokenizer.tokenize(question)\n    nbr_token_dist.append(len(tokens))\n    # token vocabulary\n    for token in tokens:\n        vocab[token] += 1\nprint('training data vocabulary size is: {}'.format(len(vocab)))\nsorted_vocab = sorted(vocab.items(), key=operator.itemgetter(1))[::-1]\nprint('top 10 most common words are: {}'.format(sorted_vocab[:10]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"89ad655d1b97a8c7628200462a5448d65a96db6a"},"cell_type":"code","source":"import plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\nimport plotly.tools as tls","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7cdb66a86960d043de46f440730b41cd6e58a6e6"},"cell_type":"code","source":"data = [go.Histogram(x=nbr_line_dist)]\npy.iplot(data, filename='histogram of number of lines')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d86e4de6d90162bda60f571dab9543a0c9aee6ac"},"cell_type":"code","source":"data = [go.Histogram(x=nbr_token_dist)]\npy.iplot(data, filename='histogram of number of lines')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0da81af5ec15abc086e7a68a4bd4f513211aa675"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}