{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom transformers import AutoTokenizer, AutoModelForSequenceClassification\n\nfrom fastai.text.all import *\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import f1_score\n\nfrom torch.utils.data import Dataset\n\nimport nltk\nfrom nltk.corpus import stopwords\nfrom nltk.stem import SnowballStemmer\nfrom nltk.tokenize import word_tokenize\n\nfrom nltk.corpus import wordnet\nfrom nltk.stem import WordNetLemmatizer","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:19:14.038014Z","iopub.execute_input":"2022-01-17T17:19:14.038258Z","iopub.status.idle":"2022-01-17T17:19:14.044584Z","shell.execute_reply.started":"2022-01-17T17:19:14.038230Z","shell.execute_reply":"2022-01-17T17:19:14.043915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data sets","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/nlp-getting-started/train.csv\")\ntest_df = pd.read_csv(\"../input/nlp-getting-started/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:19:14.046014Z","iopub.execute_input":"2022-01-17T17:19:14.046624Z","iopub.status.idle":"2022-01-17T17:19:14.166691Z","shell.execute_reply.started":"2022-01-17T17:19:14.046589Z","shell.execute_reply":"2022-01-17T17:19:14.165890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape, test_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:19:14.168481Z","iopub.execute_input":"2022-01-17T17:19:14.168950Z","iopub.status.idle":"2022-01-17T17:19:14.175885Z","shell.execute_reply.started":"2022-01-17T17:19:14.168913Z","shell.execute_reply":"2022-01-17T17:19:14.174900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:19:14.178193Z","iopub.execute_input":"2022-01-17T17:19:14.179084Z","iopub.status.idle":"2022-01-17T17:19:14.194314Z","shell.execute_reply.started":"2022-01-17T17:19:14.179053Z","shell.execute_reply":"2022-01-17T17:19:14.193515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"target\"].value_counts()/train_df.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:19:14.195976Z","iopub.execute_input":"2022-01-17T17:19:14.196532Z","iopub.status.idle":"2022-01-17T17:19:14.207932Z","shell.execute_reply.started":"2022-01-17T17:19:14.196488Z","shell.execute_reply":"2022-01-17T17:19:14.207040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{}},{"cell_type":"markdown","source":"## Data cleaning","metadata":{}},{"cell_type":"code","source":"def clean_text(x):\n    x = str(x)\n    for punct in \"/-'\":\n        x = x.replace(punct, ' ')\n    for punct in '&':\n        x = x.replace(punct, f' {punct} ')\n    for punct in '?!.,\"#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~' + '“”’':\n        x = x.replace(punct, '')\n    return x\n\ndef clean_numbers(x):\n    x = re.sub('[0-9]{5,}', '#####', x)\n    x = re.sub('[0-9]{4}', '####', x)\n    x = re.sub('[0-9]{3}', '###', x)\n    x = re.sub('[0-9]{2}', '##', x)\n    return x\n\n# Clean the text\ntrain_df[\"text\"] = train_df[\"text\"].apply(lambda x: clean_text(x))\ntest_df[\"text\"] = test_df[\"text\"].apply(lambda x: clean_text(x))\n\n# Clean numbers\ntrain_df[\"text\"] = train_df[\"text\"].apply(lambda x: clean_numbers(x))\ntest_df[\"text\"] = test_df[\"text\"].apply(lambda x: clean_numbers(x))\n\n## fill up the missing values\ntrain_df[\"text\"] = train_df[\"text\"].fillna(\" \").values\ntest_df[\"text\"] = test_df[\"text\"].fillna(\" \").values","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:17:45.940098Z","iopub.execute_input":"2022-01-17T17:17:45.940368Z","iopub.status.idle":"2022-01-17T17:17:46.229153Z","shell.execute_reply.started":"2022-01-17T17:17:45.940338Z","shell.execute_reply":"2022-01-17T17:17:46.228369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This is a helper function to map NTLK position tags\n# Full list is available here: https://www.ling.upenn.edu/courses/Fall_2003/ling001/penn_treebank_pos.html\nnltk.download('averaged_perceptron_tagger')\nnltk.download('wordnet')\n\ndef get_wordnet_pos(tag):\n    if tag.startswith(\"J\"):\n        return wordnet.ADJ\n    elif tag.startswith(\"V\"):\n        return wordnet.VERB\n    elif tag.startswith(\"N\"):\n        return wordnet.NOUN\n    elif tag.startswith(\"R\"):\n        return wordnet.ADV\n    else:\n        return wordnet.NOUN\n        \ndef preprocess_lemmatizer(texts):\n    # Initialize the lemmatizer\n    wl = WordNetLemmatizer()  \n\n    final_text_list = []\n    for sent in texts:\n        # Check if the sentence is a missing value\n        if isinstance(sent, str) == False:\n            sent = \"\"\n\n        lemmatized_sentence = []\n        # Tokenize the sentence\n        words = word_tokenize(sent)\n        # Get position tags\n        word_pos_tags = nltk.pos_tag(words)\n        # Map the position tag and lemmatize the word/token\n        for idx, tag in enumerate(word_pos_tags):\n            lemmatized_sentence.append(wl.lemmatize(tag[0], get_wordnet_pos(tag[1])))\n\n        lemmatized_text = \" \".join(lemmatized_sentence)\n\n        final_text_list.append(lemmatized_text)\n\n    return final_text_list","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:19:34.172292Z","iopub.execute_input":"2022-01-17T17:19:34.172995Z","iopub.status.idle":"2022-01-17T17:19:34.184780Z","shell.execute_reply.started":"2022-01-17T17:19:34.172956Z","shell.execute_reply":"2022-01-17T17:19:34.184080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"text\"] = preprocess_lemmatizer(train_df[\"text\"].tolist())\ntest_df[\"text\"] = preprocess_lemmatizer(test_df[\"text\"].tolist())","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:19:38.350104Z","iopub.execute_input":"2022-01-17T17:19:38.350645Z","iopub.status.idle":"2022-01-17T17:19:55.831138Z","shell.execute_reply.started":"2022-01-17T17:19:38.350604Z","shell.execute_reply":"2022-01-17T17:19:55.830269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_text = pd.concat([train_df[\"text\"], test_df[\"text\"]], axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:19:55.832754Z","iopub.execute_input":"2022-01-17T17:19:55.833204Z","iopub.status.idle":"2022-01-17T17:19:55.840068Z","shell.execute_reply.started":"2022-01-17T17:19:55.833159Z","shell.execute_reply":"2022-01-17T17:19:55.838959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sample_text(n=10):\n    sample = all_text.sample(n)\n    print(\" | \".join(sample))","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:19:55.841982Z","iopub.execute_input":"2022-01-17T17:19:55.842694Z","iopub.status.idle":"2022-01-17T17:19:55.849276Z","shell.execute_reply.started":"2022-01-17T17:19:55.842626Z","shell.execute_reply":"2022-01-17T17:19:55.848643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_text()","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:19:55.850853Z","iopub.execute_input":"2022-01-17T17:19:55.851557Z","iopub.status.idle":"2022-01-17T17:19:55.937082Z","shell.execute_reply.started":"2022-01-17T17:19:55.851519Z","shell.execute_reply":"2022-01-17T17:19:55.936105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"text\"].apply(lambda x: len(x.split())).plot(kind=\"hist\");","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:20:08.285585Z","iopub.execute_input":"2022-01-17T17:20:08.286110Z","iopub.status.idle":"2022-01-17T17:20:08.504898Z","shell.execute_reply.started":"2022-01-17T17:20:08.286073Z","shell.execute_reply":"2022-01-17T17:20:08.504220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = AutoTokenizer.from_pretrained('bert-base-cased')","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:20:14.627505Z","iopub.execute_input":"2022-01-17T17:20:14.628034Z","iopub.status.idle":"2022-01-17T17:20:17.729047Z","shell.execute_reply.started":"2022-01-17T17:20:14.627998Z","shell.execute_reply":"2022-01-17T17:20:17.728313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class QuestionDataset(Dataset):\n    def __init__(self, X, y, tokenizer):\n        self.text = X.reset_index(drop=True)\n        self.targets = y.reset_index(drop=True)\n        self.tok = tokenizer\n    \n    def __len__(self):\n        return len(self.text)\n    \n    def __getitem__(self, idx):\n        \n        text = self.text[idx]\n        targ = self.targets[idx]\n        \n        return self.tok(text, padding='max_length', \n                        truncation=True,\n                        max_length=30,\n                        return_tensors=\"pt\")[\"input_ids\"][0], tensor(targ)","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:20:17.730542Z","iopub.execute_input":"2022-01-17T17:20:17.730788Z","iopub.status.idle":"2022-01-17T17:20:17.737898Z","shell.execute_reply.started":"2022-01-17T17:20:17.730754Z","shell.execute_reply":"2022-01-17T17:20:17.737226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = train_df\nX_train, X_valid, y_train, y_valid = train_test_split(df[\"text\"], df[\"target\"], \n                                                      stratify=df[\"target\"],  test_size=0.01)\n\ntrain_ds = QuestionDataset(X_train, y_train, tokenizer)\nvalid_ds = QuestionDataset(X_valid, y_valid, tokenizer)\n\ntrain_dl = DataLoader(train_ds, bs=256)\nvalid_dl = DataLoader(valid_ds, bs=512)\ndls = DataLoaders(train_dl, valid_dl).to(\"cuda\")","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:03:17.271610Z","iopub.execute_input":"2022-01-17T17:03:17.271919Z","iopub.status.idle":"2022-01-17T17:03:17.297775Z","shell.execute_reply.started":"2022-01-17T17:03:17.271884Z","shell.execute_reply":"2022-01-17T17:03:17.297074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Building","metadata":{}},{"cell_type":"code","source":"bert = AutoModelForSequenceClassification.from_pretrained('bert-base-cased').train()\n\nclassifier = nn.Sequential(\n    nn.Linear(768, 1024),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(1024, 2)\n)\n\nbert.classifier = classifier\n\nclass BertClassifier(Module):\n    def __init__(self, bert):\n        self.bert = bert\n    def forward(self, x):\n        x = self.bert(x)\n        return x.logits\n\nmodel = BertClassifier(bert).to(\"cuda\")","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:20:22.349653Z","iopub.execute_input":"2022-01-17T17:20:22.350169Z","iopub.status.idle":"2022-01-17T17:20:24.553114Z","shell.execute_reply.started":"2022-01-17T17:20:22.350128Z","shell.execute_reply":"2022-01-17T17:20:24.552408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_0 = (train_df[\"target\"] == 0).sum()\nn_1 = (train_df[\"target\"] == 1).sum()\nn = n_0 + n_1","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:20:30.331272Z","iopub.execute_input":"2022-01-17T17:20:30.331735Z","iopub.status.idle":"2022-01-17T17:20:30.337039Z","shell.execute_reply.started":"2022-01-17T17:20:30.331698Z","shell.execute_reply":"2022-01-17T17:20:30.336394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_weights = tensor([n / (n+n_0), n / (n+n_1)]).to('cuda')\nlearn = Learner(dls, model, \n                loss_func=nn.CrossEntropyLoss(weight=class_weights), \n                metrics=[accuracy, F1Score()]).to_fp16()\nlearn.lr_find()","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:20:32.337494Z","iopub.execute_input":"2022-01-17T17:20:32.338046Z","iopub.status.idle":"2022-01-17T17:21:48.910021Z","shell.execute_reply.started":"2022-01-17T17:20:32.338001Z","shell.execute_reply":"2022-01-17T17:21:48.909323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fit_one_cycle(2, lr_max=5e-5)","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:21:48.911669Z","iopub.execute_input":"2022-01-17T17:21:48.912064Z","iopub.status.idle":"2022-01-17T17:22:34.401143Z","shell.execute_reply.started":"2022-01-17T17:21:48.912027Z","shell.execute_reply":"2022-01-17T17:22:34.400350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds, targs = learn.get_preds()","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:22:53.804487Z","iopub.execute_input":"2022-01-17T17:22:53.804930Z","iopub.status.idle":"2022-01-17T17:22:53.940465Z","shell.execute_reply.started":"2022-01-17T17:22:53.804892Z","shell.execute_reply":"2022-01-17T17:22:53.939784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"thresholds = np.linspace(0.3, 0.7, 50)\nfor threshold in thresholds:\n    f1 = f1_score(targs, F.softmax(preds, dim=1)[:, 1]>threshold)\n    print(f\"threshold:{threshold:.4f} - f1:{f1:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:22:34.532280Z","iopub.execute_input":"2022-01-17T17:22:34.532675Z","iopub.status.idle":"2022-01-17T17:22:34.601153Z","shell.execute_reply.started":"2022-01-17T17:22:34.532632Z","shell.execute_reply":"2022-01-17T17:22:34.600360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tensor = tokenizer(list(test_df[\"text\"]),\n                        padding=\"max_length\",\n                        truncation=True,\n                        max_length=30,\n                        return_tensors=\"pt\")[\"input_ids\"]","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:23:05.371691Z","iopub.execute_input":"2022-01-17T17:23:05.372413Z","iopub.status.idle":"2022-01-17T17:23:05.705411Z","shell.execute_reply.started":"2022-01-17T17:23:05.372362Z","shell.execute_reply":"2022-01-17T17:23:05.704691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TestDS:\n    def __init__(self, tensors):\n        self.tensors = tensors\n    \n    def __len__(self):\n        return len(self.tensors)\n    \n    def __getitem__(self, idx):\n        t = self.tensors[idx]\n        return t, tensor(0)\n\ntest_dl = DataLoader(TestDS(test_tensor), bs=128)","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:23:07.799741Z","iopub.execute_input":"2022-01-17T17:23:07.800318Z","iopub.status.idle":"2022-01-17T17:23:07.805319Z","shell.execute_reply.started":"2022-01-17T17:23:07.800271Z","shell.execute_reply":"2022-01-17T17:23:07.804625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds = learn.get_preds(dl=test_dl)","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:23:09.825461Z","iopub.execute_input":"2022-01-17T17:23:09.826104Z","iopub.status.idle":"2022-01-17T17:23:13.148865Z","shell.execute_reply.started":"2022-01-17T17:23:09.826067Z","shell.execute_reply":"2022-01-17T17:23:13.148219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = (F.softmax(test_preds[0], dim=1)[:, 1]>0.53).int()","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:25:23.087583Z","iopub.execute_input":"2022-01-17T17:25:23.087906Z","iopub.status.idle":"2022-01-17T17:25:23.095292Z","shell.execute_reply.started":"2022-01-17T17:25:23.087869Z","shell.execute_reply":"2022-01-17T17:25:23.094491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"result_df = pd.DataFrame()\nresult_df[\"id\"] = test_df[\"id\"]\nresult_df[\"target\"] = prediction\n \nresult_df.to_csv(\"submission.csv\", encoding='utf-8', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:25:25.711089Z","iopub.execute_input":"2022-01-17T17:25:25.711336Z","iopub.status.idle":"2022-01-17T17:25:25.727346Z","shell.execute_reply.started":"2022-01-17T17:25:25.711309Z","shell.execute_reply":"2022-01-17T17:25:25.726698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_df[\"target\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-01-17T17:07:32.370711Z","iopub.execute_input":"2022-01-17T17:07:32.370965Z","iopub.status.idle":"2022-01-17T17:07:32.377995Z","shell.execute_reply.started":"2022-01-17T17:07:32.370936Z","shell.execute_reply":"2022-01-17T17:07:32.377216Z"},"trusted":true},"execution_count":null,"outputs":[]}]}