{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"from keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Input, Embedding, Dense\nfrom keras.layers import GlobalMaxPool1D\nfrom keras.models import Model\n\nfrom tqdm import tqdm_notebook as tqdm\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"22aaa3636eda5ec7a1df199aecee881820e6b973"},"cell_type":"markdown","source":"# Data\nLoad and explore the data first."},{"metadata":{"trusted":true,"_uuid":"fb7c6dd132b9e2c8e7ab25f82422692ec1136699"},"cell_type":"code","source":"df_train = pd.read_csv(\"../input/train.csv\")\ndf_test = pd.read_csv(\"../input/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"38b23d015cbe3a6d498b1aefc25ff367bff32b5a"},"cell_type":"markdown","source":"## Shapes and examples\nTake a look at the shape of the input, as well as some examples"},{"metadata":{"trusted":true,"_uuid":"7a1a7714a2dfaf6b5b4ac1083fd80fa64b82336b"},"cell_type":"code","source":"print(f\"train: {df_train.shape}\")\nprint(f\"test: {df_test.shape}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"70ce96ba2edd0488fefa5b7a9baabe4eea5bc894"},"cell_type":"code","source":"sincere = df_train.loc[df_train['target'] == 0]\nsincere.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e3b277ffef4737f73f56a52c93ea3c9bfa72c769"},"cell_type":"code","source":"insincere = df_train.loc[df_train['target'] == 1]\ninsincere.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"28b1c59a5fb222f975ccfed095f6639174699cdf"},"cell_type":"markdown","source":"# Questions insights\n\n## NLP-fy the questions first"},{"metadata":{"trusted":true,"_uuid":"07f4326a5caf1c7af628a21e74e8be269a09082c"},"cell_type":"code","source":"q_fraction = 0.2\n\nimport spacy\n\nnlp = spacy.load(\"en_core_web_sm\")\nquestions = df_train[\"question_text\"].sample(frac=q_fraction).values\nprint(np.shape(questions))\n\ndocs = [nlp(q) for q in tqdm(questions)]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2a73a51d28e40d68dd38e31dc2caf61920f2cac8"},"cell_type":"markdown","source":"## Word count distribution"},{"metadata":{"trusted":true,"_uuid":"5594360a58ceb6f5701ee663e9b17fe229e2acb6"},"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nwords_in_question = [len(doc) for doc in tqdm(docs)]\n\nsns.set(style=\"white\", palette=\"muted\", color_codes=True)\nsns.distplot(words_in_question, color=\"b\")\n\nprint(np.bincount(words_in_question))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"240cd367d222866ec91a02948725ee3bf59433af"},"cell_type":"markdown","source":"Maybe questions with more than 60 words can be ignored - they are a very low percent."},{"metadata":{"trusted":true,"_uuid":"a33e4b93171c84e8c3651005c097625f471ddfe9"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}