{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"06e73b54188dfd9e2b5e2e0178367c4eebd4b06b"},"cell_type":"code","source":"data = pd.read_csv('../input/train.csv').sample(10000)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"# remove commonly used words and apply stemming\n\nimport nltk\nstopwords = nltk.corpus.stopwords.words('english')\ncustom_stopwords = ['will']\nstopwords.extend(custom_stopwords)\n\ndef clean_sentences(text):\n    words = text.split(' ')\n    clean_words = [word for word in words if word not in stopwords]\n    return ' '.join(clean_words)\n\ndocs = data['question_text'].str.lower()\ndocs = docs.str.replace('[^a-z ]','')\ndocs_clean = docs.apply(clean_sentences)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a7b10d7d5957ff257965ee9a60b387f378075b5f"},"cell_type":"code","source":"path = '../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin'\nimport gensim","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fffd888d169e4c545d3b6480c2e4653a42d0d5f2"},"cell_type":"code","source":"embeddings = gensim.models.KeyedVectors.load_word2vec_format(path,binary=True)\nembeddings","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6d33c5893ba83d05e03aaac33768affa9b9665eb"},"cell_type":"code","source":"#embeddings['google']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b3d11bc961d77ddbcd1df58798b9b0afd4038015"},"cell_type":"code","source":"docs.index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"9c9d5265c9a80c4393af3eac2c9694b6ec73b232"},"cell_type":"code","source":"# create vector representation for each document using word vector representation\n\ndocs_vectors = pd.DataFrame()\nfor doc in docs_clean:\n    temp = pd.DataFrame()\n    words = doc.split(' ')\n    for word in words:\n        try:\n            word2vec = embeddings[word]\n            temp = temp.append(pd.Series(word2vec),ignore_index = True)\n        except:\n            pass\n    doc_vector = temp.mean()\n    docs_vectors = docs_vectors.append(doc_vector,ignore_index = True)\ndocs_vectors","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c8f934e60f92836a57b68a765357cb22474ee2a3"},"cell_type":"code","source":"docs_vectors.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0fe86d86c7a1fbee9716aa980ed329ab202360b8"},"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score , f1_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"889e90caa6f5f11e47d668d12db077411fbfcffd"},"cell_type":"code","source":"docs_vectors_imputed = docs_vectors.fillna(docs_vectors.mean().mean())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cabbee2b12382785d15441f62a2eb877e8e27692"},"cell_type":"code","source":"docs_vectors_imputed.index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc0ba9405839b69656b2cfc99eadef9ff5347abc"},"cell_type":"code","source":"docs_vectors_imputed.index = data.index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9f92f90aadbd911d2feaa963b74053600c4bcd85"},"cell_type":"code","source":"docs_vectors_imputed.index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8f6c4f83ccc8b6aa1097a2e3a07a212417a3327e"},"cell_type":"code","source":"train , validate = train_test_split(docs_vectors_imputed,test_size = 0.3 , random_state = 300)\n\ntrain_x = train\ntrain_y = data.loc[train.index]['target']\n\nvalidate_x = validate\nvalidate_y = data.loc[validate.index]['target']\n\nmodel_dt = DecisionTreeClassifier(max_depth = 20)\nmodel_dt.fit(train_x,train_y)\nvalidate_pred = model_dt.predict(validate_x)\nprint(f1_score(validate_y,validate_pred))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8bdb8bc546ea21d5d5f4f5af1a518233bd635b0f"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f8ef86f1d474a4dcab1a02870083ef43a542dbac"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"835cd9789dc9bbc9d0f3067408e3ec8639ef9eee"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}