{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport re\n\nimport matplotlib.pyplot as plt\nimport sklearn.metrics as metrics\n\nfrom vowpalwabbit import pyvw\n\nfrom gensim.parsing import preprocessing as prep\nfrom collections import defaultdict\nfrom sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv')\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e1f9bb9d2ab6fd0c98344e580c2345c901183f9a"},"cell_type":"code","source":"plt.hist(train['target']);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3a94f17351b9298562fa07efff99527da7193e84"},"cell_type":"code","source":"class Tokenizer(object):\n    def __call__(self, doc): \n        striped = prep.strip_punctuation(doc)\n        striped = prep.strip_tags(striped)\n        striped = prep.strip_multiple_whitespaces(striped).lower()\n        return striped","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aa510764e47305f956288cf6d966eea78b915596"},"cell_type":"code","source":"class FilterRareWords(object):\n    def __init__(self):\n        self.cv = defaultdict(int)\n    def fit(self, texts):\n        for text in texts:\n            for word in text.split():\n                self.cv[word] += 1\n    def __call__(self, text):\n        return ' '.join([self.filter_word(word) for word in text.split()])\n    def filter_word(self, word):\n        return '' if self.cv[word] < 2 else word","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5e703e40fa08110912eb3865911191d40e1ea6ff"},"cell_type":"code","source":"tokenizer = Tokenizer()\nfilter_words = FilterRareWords()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"96446a7881376de44195d0a5a4d3b95a5d7d72cb"},"cell_type":"code","source":"train[train['target'] == 1].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"84c30177986d7f54db374b28a46f2b79df003748"},"cell_type":"code","source":"train['question_text'] = train['question_text'].apply(tokenizer)\n\nfilter_words.fit(train['question_text'])\ntrain['question_text'] = train['question_text'].apply(filter_words)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e58b3da7affb7d7d40b111a01e8c21d3243ccf91"},"cell_type":"code","source":"pos_weight = train['target'].sum() / train.shape[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3c326c84fdc095c5d4dbab76cd7d47e5217f7e4d"},"cell_type":"code","source":"def make_vw_feature_line(label, importance, text):\n    return '{} {} |text {}'.format(label, importance, text)\n\ndef make_vw_corpus(texts, labels):\n    for text, label in zip(texts, labels):\n        if label == 1.0:\n            cur_feautre = make_vw_feature_line('1', 1 - pos_weight, text)\n        else:\n            cur_feautre = make_vw_feature_line('-1', pos_weight, text)\n        yield cur_feautre","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6ccc69b95ada9a7f3509968a7e6365a7c3f0c8b8"},"cell_type":"code","source":"X_train, X_test = train_test_split(train, test_size=0.1, shuffle=True, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c74c6e1a4f412cd240c395c3682b2378e4e930d3"},"cell_type":"code","source":"vw = pyvw.vw(\n    quiet=True,\n    loss_function='logistic',\n    link='logistic',\n    b=29,\n    ngram=2,\n    skips=1,\n    random_seed=42,\n    l1=3.4742122764e-09,\n    l2=1.24232077629e-11,\n    learning_rate=0.751849318433,\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"36d294c0ded9fec866d9309c3049a99fcc380a72"},"cell_type":"code","source":"def get_pred(feature):\n    ex = vw.example(feature)\n    pred = vw.predict(ex)\n    ex.finish()\n    return pred","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"55aab80992a28e182cae0574c71d2576a4b72745"},"cell_type":"code","source":"%%time\nfor fit_iter in range(3):\n    for num, feature in enumerate(make_vw_corpus(X_train['question_text'], X_train['target'])):\n        ex = vw.example(feature)\n        vw.learn(ex)\n        ex.finish()\n        \n    print('pass num {} done'.format(fit_iter))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"baadb9b232f7746d704270496643139469d53396"},"cell_type":"code","source":"pred = np.array([get_pred(x) for x in make_vw_corpus(X_test['question_text'], X_test['target'])])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5070de9e7486323d8a45e5de63efdbb5bde83565"},"cell_type":"code","source":"thresholds = np.linspace(0, 1, 100)\nf1_scores = [metrics.f1_score(X_test['target'], pred > threshold) for threshold in thresholds]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ff2ca82b08f2093aaef4fa59f6cf7fb735c47906"},"cell_type":"code","source":"plt.plot(thresholds, f1_scores)\nplt.grid(True)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4c7e31cc2d5c9b1f6b114529c54b0e0c1d9fe962"},"cell_type":"code","source":"print('best f1 score is {} with threshold {}'.format(np.max(f1_scores), thresholds[np.argmax(f1_scores)]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f166371f1edade8549b05d95f87ecc1d91733b34"},"cell_type":"code","source":"test = pd.read_csv('../input/test.csv')\n\ntest['question_text'] = test['question_text'].apply(tokenizer)\ntest['question_text'] = test['question_text'].apply(filter_words)\n\npred = np.array([get_pred(x) for x in make_vw_corpus(test['question_text'], [1] * len(test))])\n\nexample = pd.read_csv('../input/sample_submission.csv')\nexample['prediction'] = (pred > thresholds[np.argmax(f1_scores)]).astype(int)\nexample.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"59f511651bd936ae1347de3fc34e8e8445833189"},"cell_type":"code","source":"plt.hist(example['prediction']);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d5d0f4568d9ecd3b453d6a44256b2850e020070f"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}