{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom nltk import bigrams\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\npd.read_csv(\"/kaggle/input/nlp-getting-started/train.csv\")\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-05T19:04:38.488195Z","iopub.execute_input":"2022-07-05T19:04:38.488593Z","iopub.status.idle":"2022-07-05T19:04:38.534353Z","shell.execute_reply.started":"2022-07-05T19:04:38.488561Z","shell.execute_reply":"2022-07-05T19:04:38.533227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/nlp-getting-started/train.csv\")\ndf_test = pd.read_csv(\"/kaggle/input/nlp-getting-started/test.csv\")\ntokenizer = Tokenizer.from_pretrained(\"bert-base-uncased\")\nN = 2\nngramizer =  lambda sentence:  [sentence[i:i+N] for i in range(len(sentence)-N+1)]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T21:17:17.467434Z","iopub.execute_input":"2022-07-05T21:17:17.468671Z","iopub.status.idle":"2022-07-05T21:17:18.477065Z","shell.execute_reply.started":"2022-07-05T21:17:17.468618Z","shell.execute_reply":"2022-07-05T21:17:18.475815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We could separate hashtags and use them as another contextual info","metadata":{}},{"cell_type":"code","source":"df[\"text_embedded\"] = df[\"text\"].apply(lambda x: tokenizer.encode(x).tokens)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T21:17:20.762063Z","iopub.execute_input":"2022-07-05T21:17:20.762593Z","iopub.status.idle":"2022-07-05T21:17:21.528473Z","shell.execute_reply.started":"2022-07-05T21:17:20.762537Z","shell.execute_reply":"2022-07-05T21:17:21.527513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df[\"text_embedded\"][1])\nprint(df[\"target\"][1])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T21:17:25.747883Z","iopub.execute_input":"2022-07-05T21:17:25.748252Z","iopub.status.idle":"2022-07-05T21:17:25.754386Z","shell.execute_reply.started":"2022-07-05T21:17:25.748223Z","shell.execute_reply":"2022-07-05T21:17:25.753411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CONTEXT_SIZE = 2\ntrain_sentences = df[\"text_embedded\"]\nngrams = list()\nfor idx, s in enumerate(train_sentences):\n    ngrams.extend([((s[i], s[i+1]), df[\"target\"][idx]) for i in range(0, len(s)-1)])\nngrams[:100]\nngram_to_idx = {ngram: i for i, (ngram, _) in enumerate(ngrams)}\n\n# add testing ngrams\ntest_sentences = df_test[\"text\"].apply(lambda x: tokenizer.encode(x).tokens)\nngrams_test = list()\nfor idx, s in enumerate(test_sentence):\n    ngrams_test.extend([((s[i], s[i+1]), df[\"target\"][idx]) for i in range(0, len(s)-1)])\n    \n#ngram_to_idx.update({ngram: i for i, (ngram, _) in enumerate(ngrams_test, start=len(ngram_to_idx))})","metadata":{"execution":{"iopub.status.busy":"2022-07-05T21:26:37.591318Z","iopub.execute_input":"2022-07-05T21:26:37.591778Z","iopub.status.idle":"2022-07-05T21:26:40.849313Z","shell.execute_reply.started":"2022-07-05T21:26:37.591743Z","shell.execute_reply":"2022-07-05T21:26:40.848123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\n\nEMBEDDING_DIM = 4\nCONTEXT_SIZE = 1\nclass NGramDisasterClassifier(nn.Module):\n    def __init__(self, vocab_size, embedding_dim, context_size):\n        super(NGramDisasterClassifier, self).__init__()\n        self.embeddings = nn.Embedding(vocab_size, embedding_dim)\n        self.linear1 = nn.Linear(context_size * embedding_dim, 128)\n        self.linear2 = nn.Linear(128, 2)\n    def forward(self, inputs):\n        embeds = self.embeddings(inputs).view((1, -1))\n        out = F.relu(self.linear1(embeds))\n        out = self.linear2(out)\n        log_probs = F.log_softmax(out, dim=1)\n        return log_probs\n    \nmodel = NGramDisasterClassifier(len(ngrams), EMBEDDING_DIM, CONTEXT_SIZE)\nprint(model)\ntest_tensor = torch.tensor(ngram_to_idx[ngrams[0][0]], dtype=torch.long)\nprint(test_tensor)\nmodel.forward(test_tensor)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T21:26:44.021619Z","iopub.execute_input":"2022-07-05T21:26:44.021996Z","iopub.status.idle":"2022-07-05T21:26:44.045091Z","shell.execute_reply.started":"2022-07-05T21:26:44.021965Z","shell.execute_reply":"2022-07-05T21:26:44.044208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optimizer = optim.SGD(model.parameters(), lr=0.01)\nloss_function = nn.NLLLoss()\nlosses = list()\nfor epoch in range(1):\n    total_loss=0\n    for ngram, target in ngrams:\n        #print(ngram)\n        #print(target)\n        \n        ngram_idx = ngram_to_idx[ngram]\n        ngram_idx = torch.tensor(ngram_idx, dtype=torch.long)\n        \n        #print(ngram_idx)\n        model.zero_grad()\n        probs = model(ngram_idx)\n        #print(probs.shape)\n        target = [1, 0] if target==0 else [0, 1]\n        #print(torch.tensor(target).shape)\n        #loss = loss_function(probs, torch.tensor([1], dtype=torch.long))\n        loss = loss_function(probs[0], torch.tensor(target, dtype=torch.long))\n        \n        loss.backward()\n        optimizer.step()\n        \n        total_loss = loss.item()\n    losses.append(total_loss)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T21:26:50.544437Z","iopub.execute_input":"2022-07-05T21:26:50.544840Z","iopub.status.idle":"2022-07-05T21:27:24.507355Z","shell.execute_reply.started":"2022-07-05T21:26:50.544809Z","shell.execute_reply":"2022-07-05T21:27:24.506068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nsns.set_style(\"darkgrid\")\nsns.set_context(\"notebook\")\nsns.lineplot(data=losses)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T21:22:45.622988Z","iopub.status.idle":"2022-07-05T21:22:45.623769Z","shell.execute_reply.started":"2022-07-05T21:22:45.623460Z","shell.execute_reply":"2022-07-05T21:22:45.623488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p=model.forward(test_tensor)\nprint(p)\ndf = pd.read_csv(\"/kaggle/input/nlp-getting-started/test.csv\")\n\ndf[\"text\"] = df[\"text\"].apply(lambda x: tokenizer.encode(x).tokens)\nfor sentence in df[\"text\"][:10]:\n   print(sentence)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T21:09:54.375098Z","iopub.execute_input":"2022-07-05T21:09:54.375510Z","iopub.status.idle":"2022-07-05T21:09:54.709194Z","shell.execute_reply.started":"2022-07-05T21:09:54.375469Z","shell.execute_reply":"2022-07-05T21:09:54.708009Z"},"trusted":true},"execution_count":null,"outputs":[]}]}