{"cells":[{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import pandas as pd\nfrom pandas import DataFrame\nimport nltk\nfrom tqdm import tqdm\nfrom contextlib import contextmanager\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.naive_bayes import MultinomialNB\nfrom nltk.tokenize import word_tokenize\nimport time\nimport numpy as np\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import classification_report\nfrom nltk.stem.snowball import SnowballStemmer\nfrom sklearn.model_selection import train_test_split","execution_count":1,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0dc5cd16bb4b63698a461b5d77cf093d9a15c5f4"},"cell_type":"code","source":"def tokenize(raw):\n    return [w.lower() for w in word_tokenize(raw) if w.isalpha()]\n\nclass StemmedTfidfVectorizer(TfidfVectorizer):\n    en_stemmer = SnowballStemmer('english')\n    \n    def build_analyzer(self):\n        analyzer = super(StemmedTfidfVectorizer, self).build_analyzer()\n        return lambda doc: (StemmedTfidfVectorizer.en_stemmer.stem(w) for w in analyzer(doc))","execution_count":2,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2c92ecb53ed8dfe43ca0471b806aa18b758ac5c2"},"cell_type":"code","source":"@contextmanager\ndef timer(task_name=\"timer\"):\n    # a timer cm from https://www.kaggle.com/lopuhin/mercari-golf-0-3875-cv-in-75-loc-1900-s\n    print(\"----{} started\".format(task_name))\n    t0 = time.time()\n    yield\n    print(\"----{} done in {:.0f} seconds\".format(task_name, time.time() - t0))","execution_count":3,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4513c0a6d3e2725a8f5badcba7672c1ff4f1f68b"},"cell_type":"code","source":"with timer(\"reading_data\"):\n    train = pd.read_csv(\"../input/train.csv\")\n    test_df = pd.read_csv('../input/test.csv')","execution_count":4,"outputs":[{"output_type":"stream","text":"----reading_data started\n----reading_data done in 4 seconds\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"## split to train and val\ntrain_df, val_df = train_test_split(train, test_size=0.1, random_state=2018)","execution_count":5,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e4290a89c21d02960fc17f45b6509b493e5eaa86"},"cell_type":"code","source":"tfidf = StemmedTfidfVectorizer(\n    tokenizer=tokenize, \n    analyzer=\"word\", \n    stop_words='english', \n    ngram_range=(1,1), \n    min_df=3    # limit of minimum number of counts: 3\n)\n\nwith timer('tfidf train'):\n    txt_all = pd.concat([train.question_text, test_df.question_text])\n    tfidf.fit(txt_all)\n    \nwith timer('construct training and validation dataset'):\n    train_X = tfidf.transform(train_df.question_text)\n    val_X = tfidf.transform(val_df.question_text)\n    \nwith timer('transforming the test set'):\n    test_X = tfidf.transform(test_df['question_text'])\n    \n## Get the target values\ntrain_y = train_df['target'].values\nval_y = val_df['target'].values","execution_count":6,"outputs":[{"output_type":"stream","text":"----tfidf train started\n----tfidf train done in 438 seconds\n----construct training and validation dataset started\n----construct training and validation dataset done in 341 seconds\n----transforming the test set started\n----transforming the test set done in 97 seconds\n","name":"stdout"}]},{"metadata":{"trusted":true,"_uuid":"afbc5dc278a3d8d291267ca6376bfc1e3e7e7555"},"cell_type":"code","source":"clf = MultinomialNB().fit(train_X, train_y)\ny_val_pred = clf.predict_proba(val_X)","execution_count":8,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# threshold search\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, f1_score(val_y, (y_val_pred[:,1] > thresh).astype(int))))","execution_count":9,"outputs":[{"output_type":"stream","text":"F1 score at threshold 0.1 is 0.44418621029649974\nF1 score at threshold 0.11 is 0.45708724781974025\nF1 score at threshold 0.12 is 0.47062549485352334\nF1 score at threshold 0.13 is 0.48076682642217133\nF1 score at threshold 0.14 is 0.4880437158469945\nF1 score at threshold 0.15 is 0.4961531416010258\nF1 score at threshold 0.16 is 0.5007895114598785\nF1 score at threshold 0.17 is 0.5039777247414479\nF1 score at threshold 0.18 is 0.506275720164609\nF1 score at threshold 0.19 is 0.5082684305985277\nF1 score at threshold 0.2 is 0.5086259163313674\nF1 score at threshold 0.21 is 0.5097727272727272\nF1 score at threshold 0.22 is 0.5101861093172078\nF1 score at threshold 0.23 is 0.5090337784760408\nF1 score at threshold 0.24 is 0.5083023543990087\nF1 score at threshold 0.25 is 0.5039099752050353\nF1 score at threshold 0.26 is 0.49882598486824936\nF1 score at threshold 0.27 is 0.49295115921694393\nF1 score at threshold 0.28 is 0.48753159800505574\nF1 score at threshold 0.29 is 0.4817150063051703\nF1 score at threshold 0.3 is 0.4742341826510163\nF1 score at threshold 0.31 is 0.4652488489366367\nF1 score at threshold 0.32 is 0.45917530385504435\nF1 score at threshold 0.33 is 0.4513334852974698\nF1 score at threshold 0.34 is 0.44558139534883723\nF1 score at threshold 0.35 is 0.4353730754046585\nF1 score at threshold 0.36 is 0.4265402843601896\nF1 score at threshold 0.37 is 0.41928379150012235\nF1 score at threshold 0.38 is 0.4112288837363365\nF1 score at threshold 0.39 is 0.39908991320468523\nF1 score at threshold 0.4 is 0.38925025676138303\nF1 score at threshold 0.41 is 0.3809275664408546\nF1 score at threshold 0.42 is 0.37039647577092516\nF1 score at threshold 0.43 is 0.3607408070144046\nF1 score at threshold 0.44 is 0.3504033354482009\nF1 score at threshold 0.45 is 0.3409571048038946\nF1 score at threshold 0.46 is 0.3309432908092001\nF1 score at threshold 0.47 is 0.32105511069241643\nF1 score at threshold 0.48 is 0.3092350300830866\nF1 score at threshold 0.49 is 0.29852188194377355\nF1 score at threshold 0.5 is 0.287781036168133\n","name":"stdout"}]},{"metadata":{"trusted":true,"_uuid":"37190c66ec00b6894109eb55dd47f8c60ee6c514"},"cell_type":"code","source":"y_pred = clf.predict_proba(test_X)\npred_test_y = (y_pred[:, 1] > 0.22).astype(int)\nout_df = pd.DataFrame({\"qid\":test_df[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","execution_count":20,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}