{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom pathlib import Path \nfrom tqdm import tqdm_notebook\nfrom fastai import *\nfrom fastai.text import *","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dc0c513075393d027bad14b0d3fcfa7b4ebfea8e"},"cell_type":"code","source":"puncts = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', '•',  '~', '@', '£', \n '·', '_', '{', '}', '©', '^', '®', '`',  '<', '→', '°', '€', '™', '›',  '♥', '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', \n '“', '★', '”', '–', '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', '—', '‹', '─', \n '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲', 'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', \n '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø', '¹', '≤', '‡', '√', ]\n\ndef clean_text(x):\n    x = str(x)\n    for punct in puncts:\n        x = x.replace(punct, f' {punct} ')\n    return x\n\ndef clean_numbers(x):\n    x = re.sub('[0-9]{5,}', '#####', x)\n    x = re.sub('[0-9]{4}', '####', x)\n    x = re.sub('[0-9]{3}', '###', x)\n    x = re.sub('[0-9]{2}', '##', x)\n    return x\n\nmispell_dict = {\"aren't\" : \"are not\",\n\"can't\" : \"cannot\",\n\"couldn't\" : \"could not\",\n\"didn't\" : \"did not\",\n\"doesn't\" : \"does not\",\n\"don't\" : \"do not\",\n\"hadn't\" : \"had not\",\n\"hasn't\" : \"has not\",\n\"haven't\" : \"have not\",\n\"he'd\" : \"he would\",\n\"he'll\" : \"he will\",\n\"he's\" : \"he is\",\n\"i'd\" : \"I would\",\n\"i'd\" : \"I had\",\n\"i'll\" : \"I will\",\n\"i'm\" : \"I am\",\n\"isn't\" : \"is not\",\n\"it's\" : \"it is\",\n\"it'll\":\"it will\",\n\"i've\" : \"I have\",\n\"let's\" : \"let us\",\n\"mightn't\" : \"might not\",\n\"mustn't\" : \"must not\",\n\"shan't\" : \"shall not\",\n\"she'd\" : \"she would\",\n\"she'll\" : \"she will\",\n\"she's\" : \"she is\",\n\"shouldn't\" : \"should not\",\n\"that's\" : \"that is\",\n\"there's\" : \"there is\",\n\"they'd\" : \"they would\",\n\"they'll\" : \"they will\",\n\"they're\" : \"they are\",\n\"they've\" : \"they have\",\n\"we'd\" : \"we would\",\n\"we're\" : \"we are\",\n\"weren't\" : \"were not\",\n\"we've\" : \"we have\",\n\"what'll\" : \"what will\",\n\"what're\" : \"what are\",\n\"what's\" : \"what is\",\n\"what've\" : \"what have\",\n\"where's\" : \"where is\",\n\"who'd\" : \"who would\",\n\"who'll\" : \"who will\",\n\"who're\" : \"who are\",\n\"who's\" : \"who is\",\n\"who've\" : \"who have\",\n\"won't\" : \"will not\",\n\"wouldn't\" : \"would not\",\n\"you'd\" : \"you would\",\n\"you'll\" : \"you will\",\n\"you're\" : \"you are\",\n\"you've\" : \"you have\",\n\"'re\": \" are\",\n\"wasn't\": \"was not\",\n\"we'll\":\" will\",\n\"didn't\": \"did not\",\n\"tryin'\":\"trying\"}\n\ndef _get_mispell(mispell_dict):\n    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))\n    return mispell_dict, mispell_re\n\nmispellings, mispellings_re = _get_mispell(mispell_dict)\ndef replace_typical_misspell(text):\n    def replace(match):\n        return mispellings[match.group(0)]\n    return mispellings_re.sub(replace, text)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"59f6ad6b0237e15585287b42f9cda364c3bfdf07"},"cell_type":"code","source":"lm = True","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f16f676a01eaab2ff9c0a76feb94626be0ffbf94"},"cell_type":"markdown","source":"# Language Model"},{"metadata":{"trusted":true,"_uuid":"43ce3c09b14e74bc39f5176db6061c651280f88d"},"cell_type":"code","source":"if lm: path = Path('../input'); list(path.iterdir())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"091feb4a29e9dee4a5d7504f410df3fd91f4904a"},"cell_type":"code","source":"train_df = pd.read_csv(path/'train.csv')\ntrain_df[\"question_text\"] = train_df[\"question_text\"].apply(lambda x: x.lower())\ntrain_df[\"question_text\"] = train_df[\"question_text\"].apply(lambda x: clean_text(x))\ntrain_df[\"question_text\"] = train_df[\"question_text\"].apply(lambda x: clean_numbers(x))\ntrain_df[\"question_text\"] = train_df[\"question_text\"].apply(lambda x: replace_typical_misspell(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0edef300616d3ef855113bd93c5a561a10a56a6a"},"cell_type":"code","source":"if lm:\n    bs = 48\n    data_lm = (TextList.from_df(train_df, path, cols='question_text')\n                .random_split_by_pct(0.1)\n                .label_for_lm()           \n                .databunch(path='.', bs=bs))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d302385a978d5bf660f3556e3d539b4a1a2f250c"},"cell_type":"code","source":"if lm:\n    data_lm.save('tmp_lm')\n    data_lm = TextLMDataBunch.load('.', 'tmp_lm', bs=bs)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1ad8e87b519aafd967bc491abf918f77919da596"},"cell_type":"raw","source":"data_lm.show_batch()"},{"metadata":{"trusted":true,"_uuid":"4835252db30dc641d407550ca7a7c45b98dd123c"},"cell_type":"code","source":"if lm:\n    learn = language_model_learner(data_lm, drop_mult=0.3, emb_sz=300)\n    learn.unfreeze()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5a2221a979a58caee8448ba1d1512130edcc9a4a"},"cell_type":"raw","source":"learn.lr_find()\nlearn.recorder.plot(skip_end=15)"},{"metadata":{"trusted":true,"_uuid":"aa881c6247cbddec0abacc39a1dede7454bc95fe"},"cell_type":"code","source":"if lm: learn.fit_one_cycle(1, 1e-2, moms=(0.8,0.7))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b5747c05961179cd1675c80865ddb67116283042"},"cell_type":"code","source":"if lm: learn.save('language_model')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d1a5fad2ffc24a1ff28ba9380d6e90dfa83aa1df"},"cell_type":"code","source":"if lm: learn.save_encoder('lm_encoder')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b6ceb4d5d65f537071fcdc4ea65b2c1f7b38d6bf"},"cell_type":"markdown","source":"# Classifier"},{"metadata":{"trusted":true,"_uuid":"81713f6e9b655613508501e6842a332ac2e31dbc"},"cell_type":"code","source":"class TextClasDataBunch(TextDataBunch):\n    \"Create a `TextDataBunch` suitable for training an RNN classifier.\"\n    @classmethod\n    def create(cls, train_ds, valid_ds, test_ds=None, path:PathOrStr='.', bs=64, pad_idx=1, pad_first=True,\n               no_check:bool=False, shuffle=[True, True, False], **kwargs) -> DataBunch:\n        \"Function that transform the `datasets` in a `DataBunch` for classification.\"\n        datasets = [train_ds, valid_ds, test_ds]\n        collate_fn = partial(pad_collate, pad_idx=pad_idx, pad_first=pad_first)\n        train_sampler = SortishSampler(datasets[0].x, key=lambda t: len(datasets[0][t][0].data), bs=bs//2)\n        train_dl = DataLoader(datasets[0], batch_size=bs//2, sampler=train_sampler, drop_last=True, **kwargs)\n        dataloaders = [train_dl]\n        dataloaders.append(DataLoader(datasets[1], batch_size=bs, **kwargs))\n        dataloaders.append(DataLoader(datasets[2], batch_size=bs, **kwargs))\n        return cls(*dataloaders, path=path, collate_fn=collate_fn)\n    \nTextList._bunch = TextClasDataBunch","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"path = Path('../input'); list(path.iterdir())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8557e86b503d381273575ea5bbead1ec478bb1c0"},"cell_type":"code","source":"if lm:\n    train_df = pd.read_csv(path/'train.csv').sample(frac=0.3, random_state=42)\nelse:\n    train_df = pd.read_csv(path/'train.csv').sample(frac=0.9, random_state=42)\n\ntrain_df[\"question_text\"] = train_df[\"question_text\"].apply(lambda x: x.lower())\ntrain_df[\"question_text\"] = train_df[\"question_text\"].apply(lambda x: clean_text(x))\ntrain_df[\"question_text\"] = train_df[\"question_text\"].apply(lambda x: clean_numbers(x))\ntrain_df[\"question_text\"] = train_df[\"question_text\"].apply(lambda x: replace_typical_misspell(x))\n#train0 = train_df[train_df.target==0].sample(n=100000, random_state=42)\n#train1 = train_df[train_df.target==1].sample(n=100000, random_state=42, replace=True)\n#train = pd.concat((train0, train1)); len(train)\n#train.reset_index(inplace=True, drop=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aee9a731749532d1c9590139a5c64af39f85b034"},"cell_type":"code","source":"test_df = pd.read_csv(path/'test.csv')\ntest_df[\"question_text\"] = test_df[\"question_text\"].apply(lambda x: x.lower())\ntest_df[\"question_text\"] = test_df[\"question_text\"].apply(lambda x: clean_text(x))\ntest_df[\"question_text\"] = test_df[\"question_text\"].apply(lambda x: clean_numbers(x))\ntest_df[\"question_text\"] = test_df[\"question_text\"].apply(lambda x: replace_typical_misspell(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aee9a731749532d1c9590139a5c64af39f85b034"},"cell_type":"code","source":"test = TextList.from_df(test_df, path, cols='question_text')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2e74705319e2f92636a7c16ae8c4b65e18d2913d"},"cell_type":"raw","source":"train['weights'] = 0\ntrain.loc[train.target==1, 'weights'] = 0.06187\ntrain.loc[train.target==0, 'weights'] = 1-0.06187"},{"metadata":{"trusted":true,"_uuid":"fcc786d532650ed37e86660f0fcc6b84e9b20d81"},"cell_type":"raw","source":"valid = train.sample(frac=0.1, weights=train.weights.values, random_state=42)"},{"metadata":{"trusted":true,"_uuid":"7e5a7e9110d2edd24e5c350b9ca4ccc077171073"},"cell_type":"code","source":"if lm:\n    data = (TextList.from_df(train_df, path, cols='question_text', vocab=data_lm.vocab)\n                    .random_split_by_pct(0.1)\n                    .label_from_df(cols=2)\n                    .add_test(test)\n                    .databunch(path='.')) \nelse:\n    data = (TextList.from_df(train_df, path, cols='question_text')\n                .random_split_by_pct(0.1)\n                .label_from_df(cols=2)\n                .add_test(test)\n                .databunch(path='.')) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7e5a7e9110d2edd24e5c350b9ca4ccc077171073"},"cell_type":"raw","source":"data.save()\ndata = TextDataBunch.load('.')"},{"metadata":{"trusted":true,"_uuid":"f59dfa90ed8d9439b1cb02586de0be18a61674cb"},"cell_type":"raw","source":"data.show_batch()"},{"metadata":{"trusted":true,"_uuid":"a77d92a1042dcc93f18104283481aae191573c38"},"cell_type":"code","source":"f_score = Fbeta_binary(beta2=1,clas = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4be9553ff662ba500dfac18d83991cfb70c9cf61"},"cell_type":"code","source":"learn = text_classifier_learner(data, drop_mult=0.5, metrics=[accuracy, f_score], emb_sz=300)\nif lm: learn.load_encoder('lm_encoder')\nlearn.freeze()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7a859e4e83424f44c0c82ff1d7c90d3cd694869b"},"cell_type":"code","source":"gc.collect();","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dec251866954f2934b5316070a81045eda33b44c"},"cell_type":"raw","source":"learn.lr_find()\nlearn.recorder.plot()"},{"metadata":{"trusted":true,"_uuid":"57a5a8ff96fe629d48153ce461d8a9f73b5f012a","scrolled":false},"cell_type":"code","source":"learn.fit_one_cycle(1, 1e-2, moms=(0.8,0.7))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9ad70ed0eadd99e4a1e7121000e561f3d44996e9"},"cell_type":"raw","source":"learn.save('first')"},{"metadata":{"_uuid":"9c68f2821a7cb6171d7768412eaa2093d3c8221e"},"cell_type":"raw","source":"learn.load('first');"},{"metadata":{"_uuid":"02ae7221fdfe1925422d5070970266a89a2f8e85"},"cell_type":"raw","source":"learn.freeze_to(-2)\nlearn.fit_one_cycle(1, slice(1e-2/(2.6**4),1e-2), moms=(0.8,0.7))"},{"metadata":{"_uuid":"f3cd8df4cc6ceff7ee2244b722654401b714c1c8"},"cell_type":"raw","source":"learn.save('second')"},{"metadata":{"_uuid":"992a108e5f441edb4a1eb77ed4c8155e3af5ff6d"},"cell_type":"raw","source":"learn.load('second');"},{"metadata":{"_uuid":"8bc9c6bba52b1cdaa6fedcec8b43fe871bb8f74b"},"cell_type":"raw","source":"learn.freeze_to(-3)\nlearn.fit_one_cycle(1, slice(5e-3/(2.6**4),5e-3), moms=(0.8,0.7))"},{"metadata":{"_uuid":"bc2d5b272e98753a98639c6464889280acc462c1"},"cell_type":"raw","source":"learn.save('third')"},{"metadata":{"_uuid":"cf4ff0d1ee907889bffc029849441d77b3bafc47"},"cell_type":"raw","source":"learn.load('third');"},{"metadata":{"trusted":true,"_uuid":"9b17b36aebe9e1be91389c30db94293b79206b51"},"cell_type":"code","source":"learn.unfreeze()\nlearn.fit_one_cycle(2, slice(1e-3/(2.6**4),1e-3), moms=(0.8,0.7))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e201769d95486a2aef85f553ef5cf648702043b8"},"cell_type":"raw","source":"learn.save('final')"},{"metadata":{"_uuid":"59e76cf4ad73c67b63afca627e3cd439c9fed287"},"cell_type":"raw","source":"learn.load('final');"},{"metadata":{"trusted":true,"_uuid":"ea467e08ba405c40b2941fafa6aa1f90057a3455"},"cell_type":"code","source":"preds = learn.get_preds(DatasetType.Valid)\nproba = to_np(preds[0][:,1])\nytrue = to_np(preds[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"04166b453254abc6083743f2f887f1c902b9f567"},"cell_type":"code","source":"from sklearn.metrics import roc_curve, precision_recall_curve\ndef threshold_search(y_true, y_proba, plot=False):\n    precision, recall, thresholds = precision_recall_curve(y_true, y_proba)\n    thresholds = np.append(thresholds, 1.001) \n    F = 2 / (1/precision + 1/recall)\n    best_score = np.max(F)\n    best_th = thresholds[np.argmax(F)]\n    if plot:\n        plt.plot(thresholds, F, '-b')\n        plt.plot([best_th], [best_score], '*r')\n        plt.show()\n    search_result = {'threshold': best_th , 'f1': best_score}\n    return search_result ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"67261e5c0b8cf51dd512b7dd4afd71d1373365cf"},"cell_type":"code","source":"thr = threshold_search(ytrue, proba, plot=True); thr","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f307bb6844d4384bc43215112385cfe3f31f3967"},"cell_type":"code","source":"preds = learn.get_preds(DatasetType.Test)\nproba = to_np(preds[0][:,1])\npredsC = (proba > thr['threshold']).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cd20f0c905d874a57a6aeabd9168a65dee7792ed"},"cell_type":"code","source":"sub = pd.read_csv('../input/sample_submission.csv')\nsub.prediction = predsC\nsub.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d88a986e7331419cb906ea3e77614c002d1db82b"},"cell_type":"code","source":"sub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b2ba3257242527988d478875e740d2760559a0ca"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}