{"cells":[{"metadata":{"trusted":true,"_uuid":"4bc1719df6f836da114033b076493295bf393c3f"},"cell_type":"code","source":"import warnings\nimport traceback\nimport sys\nfrom datetime import datetime\nimport json\n\nimport numpy as np\nimport pandas as pd\nfrom timeit import default_timer as timer\n\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import roc_auc_score, log_loss, average_precision_score\nfrom sklearn.model_selection import ParameterSampler\nfrom scipy.stats import randint as randint\nfrom scipy.stats import uniform as uniform\nfrom sklearn.utils import check_random_state\n\nimport fastText","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"19db119d86ce31c7c5dbbf4c9bd44969fad876ba"},"cell_type":"code","source":"PUNCTS_FASTTEXT = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', '•',  '~', '@', '£', \n '·', '{', '}', '©', '^', '®',  '<', '→', '°', '€', '™', '›',  '♥', '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', \n '“', '★', '”', '–', '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '═', '¦', '║', '―', '¥', '▓', '—', '‹', '─', \n '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲', 'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', \n '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø', '¹', '≤', '‡', '√', ]\n\n\ndef clean_fasttext(x):\n    x = str(x)\n    for punct in PUNCTS_FASTTEXT:\n        x = x.replace(punct, f' {punct} ')\n    return x\n\n\ndef predict_fasttext_single(model, x):\n    labels, probs = model.predict(x, 2)\n    if labels[0] == '__label__1':\n        return probs[0]\n    else:\n        return probs[1]\n\n\ndef predict_fasttext(model, df):\n    return df.cleaned_text.apply(lambda x: predict_fasttext_single(model, x))\n\n\n# from https://www.kaggle.com/c/quora-insincere-questions-classification/discussion/76391\ndef scoring(y_true, y_proba, verbose=True):\n    from sklearn.metrics import roc_curve, precision_recall_curve, f1_score\n    from sklearn.model_selection import RepeatedStratifiedKFold\n\n    def threshold_search(y_true, y_proba):\n        precision , recall, thresholds = precision_recall_curve(y_true, y_proba)\n        thresholds = np.append(thresholds, 1.001) \n        F = 2 / (1/precision + 1/recall)\n        best_score = np.max(F)\n        best_th = thresholds[np.argmax(F)]\n        return best_th \n\n\n    rkf = RepeatedStratifiedKFold(n_splits=5, n_repeats=10)\n\n    scores = []\n    ths = []\n    for train_index, test_index in rkf.split(y_true, y_true):\n        y_prob_train, y_prob_test = y_proba[train_index], y_proba[test_index]\n        y_true_train, y_true_test = y_true[train_index], y_true[test_index]\n\n        # determine best threshold on 'train' part \n        best_threshold = threshold_search(y_true_train, y_prob_train)\n\n        # use this threshold on 'test' part for score \n        sc = f1_score(y_true_test, (y_prob_test >= best_threshold).astype(int))\n        scores.append(sc)\n        ths.append(best_threshold)\n\n    best_th = np.mean(ths)\n    score = np.mean(scores)\n\n    if verbose: print(f'Best threshold: {np.round(best_th, 4)}, Score: {np.round(score,5)}')\n\n    return best_th, score\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7c481ccdd26f39cf2c4632d19888a12fda370ae7"},"cell_type":"code","source":"!mkdir -p ../tmp","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c85141bbf51a965774e6a1586810a8a1f9d8bea8"},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\").sample(frac=1, random_state=3465).reset_index(drop=True)\n\ntrain['cleaned_text'] = train[\"question_text\"].apply(lambda x: clean_fasttext(x)).str.replace('\\n', ' ')\nfasttext_labeled = '__label__' + train.target.astype(str) + ' ' + train.cleaned_text\n\nnp.savetxt('../tmp/train.txt', fasttext_labeled.values, fmt='%s')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2e6f3feef9c0793e29d9827a4405db683cc0cb09"},"cell_type":"code","source":"test = pd.read_csv(\"../input/test.csv\", index_col='qid')\ntest['cleaned_text'] = test[\"question_text\"].apply(lambda x: clean_fasttext(x)).str.replace('\\n', ' ')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ebfaf7d27b74876a50f6aef1dbdd9a3a034885d6"},"cell_type":"code","source":"parameters = {\n        'lr': 0.161195,\n        'dim': 300,\n        'ws': 5,\n        'epoch': 4,\n        'minCount': 80,\n        'minCountLabel': 0,\n        'minn': 4,\n        'maxn': 5,\n        'neg': 5,\n        'wordNgrams': 3,\n        'loss': \"hs\",\n        'bucket': 2000000,\n        'thread': 4,\n        'lrUpdateRate': 100,\n        't': 1e-4,\n        'pretrainedVectors': '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec',\n        'verbose': 0\n    }","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b724aa952dd41ba0c23f3f065e6d39b29140a30f"},"cell_type":"code","source":"model = fastText.train_supervised(input='../tmp/train.txt', **parameters)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"47e38e33ddda716e1d7bc4ca21e34d3b8f8b566b"},"cell_type":"code","source":"train_pred = predict_fasttext(model, train)\ntest_pred = predict_fasttext(model, test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"164992df31a79db1c271c9d896b43948b3f83925"},"cell_type":"code","source":"best_th, f1 = scoring(train.target, train_pred, verbose=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6443272813e02127611cea41de5c2a4d570b3032"},"cell_type":"code","source":"best_th, f1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"804a01e2a979b0b07be445ce9c154d158d9d97c5"},"cell_type":"code","source":"pred = (test_pred > best_th).astype('int').rename('prediction')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7a07c43f18bbb9bb45cbb7ecb8723560d4897fd5"},"cell_type":"code","source":"pd.DataFrame(pred).to_csv('submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8d0ebf64ce4e2a9b9ee82eeebcbe8e480bdb1005"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}