{"cells":[{"metadata":{"_uuid":"facfe0a56cc06e51266f4548589d26ab66bb3e76"},"cell_type":"markdown","source":"# Quora notebook (Naive Bayes attempt)\nThis is a kernel using Naive Bayes classification.\n\nGeneral guidance and help from Jannes Klaas' ML workshop series and the provided materials. <br>\nUsed example code from KFold documentation: https://scikit-learn.org/stable/modules/generated/sklearn.model_selection.KFold.html <br>\nThis kernel by Dust was also very helpful: https://www.kaggle.com/stardust0/naive-bayes-and-logistic-regression-baseline <br>\nF1 scoring thanks to SRK: https://www.kaggle.com/sudalairajkumar/a-look-at-different-embeddings"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\npath = \"../input\" # Change to \"../input\" on Kaggle, or \"../../LOCAL FILES/all\" on local\nprint(os.listdir(path))\n\n# Any results you write to the current directory are saved as output.\n\nfrom sklearn.model_selection import KFold\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.naive_bayes import GaussianNB, MultinomialNB, ComplementNB, BernoulliNB\nfrom sklearn import metrics\nfrom sklearn.metrics import f1_score","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"# Preprocess with count vectors"},{"metadata":{"trusted":true,"_uuid":"9f9b5f62e37033345712a40a6268dbf3979357e7"},"cell_type":"code","source":"train = pd.read_csv(path + \"/train.csv\")\ntest = pd.read_csv(path + \"/test.csv\")\nsample = pd.read_csv(path + \"/sample_submission.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0dbdeb972ba1f1822de85ed62054ea7c690c56af"},"cell_type":"code","source":"count_vectorizer = CountVectorizer()\ncount_vectorizer.fit(train['question_text'].append(test['question_text']))\ntrain_data = count_vectorizer.transform(train['question_text'])\ntest_data = count_vectorizer.transform(test['question_text'])\n\ntrain_targets = train['target']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8c6d230894628fb5a45dfbcf0b8b50cf49b8b77d"},"cell_type":"markdown","source":"# Building the model"},{"metadata":{"trusted":true,"_uuid":"cfd2229aa754fd1e76b2cb6fac6263730da6d891"},"cell_type":"code","source":"kf = KFold(n_splits=4, shuffle = False, random_state = 42)\nprint(kf)\npredictions = np.zeros((train.shape[0], ))\nfinal_predictions = 0\n\nfor train_index, test_index in kf.split(train):\n    print(\"TRAIN:\", train_index, \"TEST:\", test_index)\n    X_train, X_test = train_data[train_index], train_data[test_index]\n    y_train, y_test = train_targets[train_index], train_targets[test_index]\n    model = MultinomialNB()\n    model.fit(X_train,y_train)\n    predictions[test_index] = model.predict_proba(X_test)[:, 1]\n    final_predictions = final_predictions + model.predict_proba(test_data)[:, 1]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0791866afb1fb430de4207baee3c9fe820d312a9"},"cell_type":"markdown","source":"# Calculating the best threshold and scoring the model"},{"metadata":{"trusted":true,"_uuid":"28457f4acabb8eb38019509254e25be053f30f5f"},"cell_type":"code","source":"for thresh in np.arange(0.1, 0.901, 0.02):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(train_targets, (predictions>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"6976277795cf9fe68787371acf1b36dbc8034964"},"cell_type":"code","source":"predictions = (predictions > 0.68).astype(np.int)\nfinal_predictions = (final_predictions > 0.68).astype(np.int)\nf1_score(train_targets, predictions)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0107d0002a8ba04a050938815c52ed5b63a995ab"},"cell_type":"markdown","source":"# Output the results"},{"metadata":{"trusted":true,"_uuid":"010b074241d0f6c0446499a2316c3ff3e56bdea1"},"cell_type":"code","source":"output = pd.DataFrame({\"qid\":test[\"qid\"].values})\noutput['prediction'] = final_predictions\noutput.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}