{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport nltk\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize \nimport string\n\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom collections import Counter\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_extraction.text import TfidfTransformer\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n\nfrom sklearn.naive_bayes import MultinomialNB, BernoulliNB\nfrom sklearn.metrics import accuracy_score # for evaluating results\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d4e813adeadd9b92bebc8d3ed982238f536fa210"},"cell_type":"markdown","source":"**Load data** "},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5f79f8d3421a7c214121df7bd20d56505590f6d9"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e7aa1a179f19d279c1c741a42ad00bf9b0a9b90f"},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0368fb725279c5cfb84d1041be6380153916fcad"},"cell_type":"markdown","source":"**Processing data**"},{"metadata":{"trusted":true,"_uuid":"d820c9a19d35c11848bbb9ba414265a24c1f67b4"},"cell_type":"code","source":"# Load stop word\neng_stopwords = set(stopwords.words(\"english\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"19b1740e7e8a061162408764ef39a5f17985481b"},"cell_type":"code","source":"def processingData(train, test):\n    # Tokenize\n    train['question_text'] = train[\"question_text\"].apply(lambda x: \" \".join(word_tokenize(str(x))))\n    test['question_text'] = test[\"question_text\"].apply(lambda x: \" \".join(word_tokenize(str(x))))\n\n    # Remove punctuation\n    train['question_text'] = train[\"question_text\"].apply(lambda x: x.translate(str.maketrans('','',string.punctuation)))\n    test['question_text'] = test[\"question_text\"].apply(lambda x: x.translate(str.maketrans('','',string.punctuation)))\n\n    ## Remove stopwords in the text ##\n    train[\"question_text\"] = train[\"question_text\"].apply(lambda x: \" \".join([w for w in str(x).lower().split() if not w in eng_stopwords]))\n    test[\"question_text\"] = test[\"question_text\"].apply(lambda x: \" \".join([w for w in str(x).lower().split() if not w in eng_stopwords]))\n    \n    return train, test","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4a7548a87295b8920c8dae6927b6915ee6178a0d"},"cell_type":"code","source":"train, test = processingData(train, test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"626d61871f7e4ba646b531365a15d6c401954112"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fe9a99bbd2dc7d80b70707ac81460bdd3ef9f0df"},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f06332bb99528b1a770f3c55bb0d8a370b8e4896"},"cell_type":"markdown","source":"**Extracting features from text**"},{"metadata":{"trusted":true,"_uuid":"5f3453724a755c6cc1812518364a459bdd4efb93"},"cell_type":"code","source":"train_question_list = train['question_text']\ntest_question_list = test['question_text']\n\nvectorizer  = CountVectorizer()\n\nx_train =  vectorizer.fit_transform(train_question_list)\nx_test =  vectorizer.transform(test_question_list)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ee43c3e3d11ff44df842de70dd63b554e40a0ddb"},"cell_type":"code","source":"y_train_tfidf = np.array(train[\"target\"].tolist())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8a41ff304b26ec628a9b18c17042c10a0e8a322b"},"cell_type":"code","source":"train_x, validate_x, train_y, validate_y = train_test_split(x_train, y_train_tfidf, test_size=0.3)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"03d08f1cb732960bf1ae1876c9148a5a5a55b28d"},"cell_type":"markdown","source":"**Training**"},{"metadata":{"trusted":true,"_uuid":"ff11774312d3ab2725159d0c8901c604b404f8f9"},"cell_type":"code","source":"clf = MultinomialNB()\nclf.fit(train_x, train_y)\ny_vad = clf.predict(validate_x)\nprint('accuracy = %.2f%%' % \\\n      (accuracy_score(validate_y, y_vad)*100))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"44957c0974863e7a232d310d75903932745620e7"},"cell_type":"markdown","source":"**Prediction**"},{"metadata":{"trusted":true,"_uuid":"efc079b6eb7420cef3ed6188a9e9b8c24b2a5f12"},"cell_type":"code","source":"y_predict = clf.predict(x_test)\npredict = pd.DataFrame(data = y_predict, columns=['prediction'])\npredict = predict.astype(int)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a2b995b1df3957776ac4f4a667822d1c9da89dc1"},"cell_type":"markdown","source":"**Extracting result**"},{"metadata":{"trusted":true,"_uuid":"cd7c3963af1d7a7d09437dae2b6802b1308d3545"},"cell_type":"code","source":"id = test['qid']\nid_df = pd.DataFrame(id)\n# Join predicted into result dataframe and write result as a CSV file\nresult = id_df.join(predict)\nresult.to_csv(\"submission.csv\", index = False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}