{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nfrom pathlib import Path\n\nimport torch\nimport torch.nn.functional as F\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, Dataset\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom tqdm import tqdm, tqdm_notebook\nfrom sklearn.model_selection import train_test_split\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\nDATA_PATH = \"/kaggle/input/quora-insincere-questions-classification/train.csv\"\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-21T19:46:04.156502Z","iopub.execute_input":"2022-07-21T19:46:04.157076Z","iopub.status.idle":"2022-07-21T19:46:05.150132Z","shell.execute_reply.started":"2022-07-21T19:46:04.157040Z","shell.execute_reply":"2022-07-21T19:46:05.149099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Text preprocessing**","metadata":{}},{"cell_type":"code","source":"class Sequences(Dataset):\n    def __init__(self, path, is_train):\n        df = pd.read_csv(path)\n        train, test = train_test_split(df, test_size=0.2)\n        df = train if is_train else test\n        self.vectorizer = CountVectorizer(stop_words='english', max_df=0.99, min_df=0.005)\n        self.sequences = self.vectorizer.fit_transform(df.question_text.tolist())\n        self.labels = df.target.tolist()\n        self.token2idx = self.vectorizer.vocabulary_\n        self.idx2token = {idx: token for token, idx in self.token2idx.items()}\n        \n    def __getitem__(self, i):\n        return self.sequences[i, :].toarray(), self.labels[i]\n    \n    def __len__(self):\n        return self.sequences.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:08:27.742308Z","iopub.execute_input":"2022-07-21T18:08:27.742682Z","iopub.status.idle":"2022-07-21T18:08:27.759241Z","shell.execute_reply.started":"2022-07-21T18:08:27.742652Z","shell.execute_reply":"2022-07-21T18:08:27.758317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class PreprocessedSequences(Dataset): #???? big confuse\n    def __init__(self, path, max_seq_len, is_train):\n        self.max_seq_len = max_seq_len\n        df = pd.read_csv(path)\n        train, test = train_test_split(df, test_size=0.1)\n        df = train if is_train else test\n#         df = test if is_train else train\n        self.vectorizer = CountVectorizer(stop_words='english', min_df=0.015)\n        self.vectorizer.fit(df.question_text.tolist())\n        \n        self.token2idx = self.vectorizer.vocabulary_\n        self.token2idx['<PAD>'] = max(self.token2idx.values()) + 1\n\n        tokenizer = self.vectorizer.build_analyzer()\n        self.encode = lambda x: [self.token2idx[token] for token in tokenizer(x)\n                                 if token in self.token2idx]\n        self.pad = lambda x: x + (max_seq_len - len(x)) * [self.token2idx['<PAD>']] #makes sure each batch is the same size\n        \n        sequences = [self.encode(sequence)[:max_seq_len] for sequence in df.question_text.tolist()]\n        sequences, self.labels = zip(*[(sequence, label) for sequence, label\n                                    in zip(sequences, df.target.tolist()) if sequence])\n        self.sequences = [self.pad(sequence) for sequence in sequences]\n\n    def __getitem__(self, i):\n        assert len(self.sequences[i]) == self.max_seq_len\n        return self.sequences[i], self.labels[i]\n    \n    def __len__(self):\n        return len(self.sequences)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T20:04:33.079128Z","iopub.execute_input":"2022-07-21T20:04:33.079491Z","iopub.status.idle":"2022-07-21T20:04:33.094029Z","shell.execute_reply.started":"2022-07-21T20:04:33.079460Z","shell.execute_reply":"2022-07-21T20:04:33.093034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\ndevice","metadata":{"execution":{"iopub.status.busy":"2022-07-21T20:04:36.993998Z","iopub.execute_input":"2022-07-21T20:04:36.994348Z","iopub.status.idle":"2022-07-21T20:04:37.062415Z","shell.execute_reply.started":"2022-07-21T20:04:36.994319Z","shell.execute_reply":"2022-07-21T20:04:37.061337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Models**","metadata":{}},{"cell_type":"code","source":"class BagOfWordsClassifier(nn.Module):\n    def __init__(self, vocab_size, hidden1, hidden2):\n        super(BagOfWordsClassifier, self).__init__()\n        self.fc1 = nn.Linear(vocab_size, hidden1)\n        self.fc2 = nn.Linear(hidden1, hidden2)\n        self.fc3 = nn.Linear(hidden2, 1)\n    \n    def forward(self, inputs):\n        x = F.relu(self.fc1(inputs.squeeze(1).float())) # what does this squeeze do?\n        x = F.relu(self.fc2(x))\n        return self.fc3(x)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:27:31.803513Z","iopub.execute_input":"2022-07-21T17:27:31.804079Z","iopub.status.idle":"2022-07-21T17:27:31.810826Z","shell.execute_reply.started":"2022-07-21T17:27:31.804043Z","shell.execute_reply":"2022-07-21T17:27:31.809891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class RNN(nn.Module):\n    def __init__(\n        self,\n        vocab_size,\n        batch_size,\n        embedding_dimension=100,\n        hidden_size=128, \n        n_layers=1,\n        device='cpu',\n    ):\n        super(RNN, self).__init__()\n        self.n_layers = n_layers\n        self.hidden_size = hidden_size\n        self.device = device\n        self.batch_size = batch_size\n        \n        self.encoder = nn.Embedding(vocab_size, embedding_dimension)\n        self.rnn = nn.GRU(\n            embedding_dimension,\n            hidden_size,\n            num_layers=n_layers,\n            batch_first=True,\n        )\n        self.decoder = nn.Linear(hidden_size, 1)\n        \n    def init_hidden(self):\n        return torch.randn(self.n_layers, self.batch_size, self.hidden_size).to(self.device)\n    \n    def forward(self, inputs):\n        # Avoid breaking if the last batch has a different size\n        batch_size = inputs.size(0)\n        if batch_size != self.batch_size:\n            self.batch_size = batch_size\n            \n        encoded = self.encoder(inputs)\n        output, hidden = self.rnn(encoded, self.init_hidden())\n        output = self.decoder(output[:, :, -1]).squeeze()\n        return output","metadata":{"execution":{"iopub.status.busy":"2022-07-21T20:04:39.196022Z","iopub.execute_input":"2022-07-21T20:04:39.196384Z","iopub.status.idle":"2022-07-21T20:04:39.207553Z","shell.execute_reply.started":"2022-07-21T20:04:39.196341Z","shell.execute_reply":"2022-07-21T20:04:39.206426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Helper functions and train/test loops**","metadata":{}},{"cell_type":"code","source":"def get_correct(outputs, labels):\n    n_sincere = sum([x == 0 for x in labels]).item()\n    n_insincere = sum([x == 1 for x in labels]).item()\n    \n    n_correct_sincere = 0\n    n_correct_insincere = 0\n    \n    for o, l in zip(outputs, labels):\n#         print(round(torch.sigmoid(o).item()), l.item())\n        if round(torch.sigmoid(o).item()) == l.item():\n            if l.item() == 0:\n                n_correct_sincere += 1\n            else:\n                n_correct_insincere += 1\n    \n    return np.array([n_correct_sincere, n_sincere, n_correct_insincere, n_insincere])","metadata":{"execution":{"iopub.status.busy":"2022-07-21T20:10:55.872118Z","iopub.execute_input":"2022-07-21T20:10:55.872519Z","iopub.status.idle":"2022-07-21T20:10:55.880568Z","shell.execute_reply.started":"2022-07-21T20:10:55.872484Z","shell.execute_reply":"2022-07-21T20:10:55.879619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_loop(model, train_loader, test_loader, optimizer, criterion):\n    model.to(device)\n    train_losses = []\n    for epoch in range(10):\n        model.train()\n        progress_bar = tqdm_notebook(train_loader, leave=False)\n        losses = []\n\n        total = 0\n        for inputs, target in progress_bar:\n            model.zero_grad()\n            inputs, target = inputs.to(device), target.to(device)\n            output = model(inputs)\n            loss = criterion(output.squeeze(), target.float())\n\n            loss.backward()\n\n            nn.utils.clip_grad_norm_(model.parameters(), 3)\n\n            optimizer.step()\n\n            progress_bar.set_description(f'Loss: {loss.item():.3f}')\n            \n            losses.append(loss.item())\n            total += 1\n        \n        epoch_loss = sum(losses) / total\n        train_losses.append(epoch_loss)\n\n        tqdm.write(f'Epoch #{epoch + 1} \\t Train Loss: {epoch_loss:.3f}')\n        \n        test_loop(model, test_loader)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T20:35:09.936001Z","iopub.execute_input":"2022-07-21T20:35:09.936347Z","iopub.status.idle":"2022-07-21T20:35:09.948046Z","shell.execute_reply.started":"2022-07-21T20:35:09.936319Z","shell.execute_reply":"2022-07-21T20:35:09.947088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def test_loop(model, test_loader):\n    model.eval()\n    with torch.no_grad():\n        progress_bar = tqdm_notebook(test_loader, leave=False)\n        losses = []\n        stats = np.array([0,0,0,0])\n        \n        total = 0\n        for inputs, target in progress_bar:\n            inputs, target = inputs.to(device), target.to(device)\n            model.zero_grad()\n\n            output = model(inputs)\n            loss = criterion(output.squeeze(), target.float())\n\n            progress_bar.set_description(f'Loss: {loss.item():.3f}')\n\n            stats += get_correct(outputs, labels)\n            losses.append(loss.item())\n            total += 1\n\n        sincere_accuracy = stats[0] / stats[1] * 100\n        insincere_accuracy = stats[2] / stats[3] * 100\n        test_loss = sum(losses) / total\n        \n        tqdm.write(f'{stats[0]} / {stats[1]}, {stats[2]} / {stats[3]}')\n        tqdm.write(f'Test Loss: {test_loss:.3f} \\t Sincere Accuracy: {sincere_accuracy:.3f} \\t Insincere Accuracy: {insincere_accuracy:.3f}')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T20:36:48.450199Z","iopub.execute_input":"2022-07-21T20:36:48.450887Z","iopub.status.idle":"2022-07-21T20:36:48.460076Z","shell.execute_reply.started":"2022-07-21T20:36:48.450852Z","shell.execute_reply":"2022-07-21T20:36:48.459144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_sentiment(text):\n    model.eval()\n    with torch.no_grad():\n        test_vector = torch.LongTensor([train_data.pad(train_data.encode(text))]).to(device)\n        output = model(test_vector)\n        \n        prediction = torch.sigmoid(output)\n        print(prediction)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T20:04:18.385075Z","iopub.execute_input":"2022-07-21T20:04:18.385450Z","iopub.status.idle":"2022-07-21T20:04:18.391284Z","shell.execute_reply.started":"2022-07-21T20:04:18.385417Z","shell.execute_reply":"2022-07-21T20:04:18.390259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Runs**","metadata":{}},{"cell_type":"code","source":"train_data = Sequences(DATA_PATH, is_train=True)\ntrain_loader = DataLoader(train_data, batch_size=4096)\n\nprint(train_data[5][0].shape)\n\ntest_data = Sequences(DATA_PATH, is_train=False)\ntest_loader = DataLoader(test_data, batch_size=2048) #4096 gives shape (1,107) and idk why\n\nprint(test_data[5][0].shape)\n\nmodel = BagOfWordsClassifier(len(train_data.token2idx), hidden1=128, hidden2=64)\n\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = optim.Adam([p for p in model.parameters() if p.requires_grad], lr=0.001)\n\ntrain_loop(model, train_loader, optimizer, criterion)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:27:47.641027Z","iopub.execute_input":"2022-07-21T17:27:47.641983Z","iopub.status.idle":"2022-07-21T17:27:53.909331Z","shell.execute_reply.started":"2022-07-21T17:27:47.641947Z","shell.execute_reply":"2022-07-21T17:27:53.907654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = PreprocessedSequences(DATA_PATH, max_seq_len=128, is_train=True) #there's no way train accuracy is literally 92.3% every single time\n\ndef collate(batch):\n    inputs = torch.LongTensor([item[0] for item in batch])\n    target = torch.FloatTensor([item[1] for item in batch])\n    return inputs, target\n\nbatch_size = 256\ntrain_loader = DataLoader(train_data,  batch_size=batch_size, collate_fn=collate)\n\ntest_data = PreprocessedSequences(DATA_PATH, max_seq_len=128, is_train=False)\ntest_loader = DataLoader(test_data, batch_size=batch_size, collate_fn=collate) #max_seq_len might be a tweakable hyperparam","metadata":{"execution":{"iopub.status.busy":"2022-07-21T20:04:45.432339Z","iopub.execute_input":"2022-07-21T20:04:45.433222Z","iopub.status.idle":"2022-07-21T20:05:32.091422Z","shell.execute_reply.started":"2022-07-21T20:04:45.433172Z","shell.execute_reply":"2022-07-21T20:05:32.090419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = RNN(\n    hidden_size=128,\n    vocab_size=len(train_data.token2idx),\n    device=device,\n    batch_size=batch_size,\n)\nmodel = model.to(device)\n\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = optim.Adam([p for p in model.parameters() if p.requires_grad], lr=0.001)\n\ntrain_loop(model, train_loader, test_loader, optimizer, criterion)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T20:36:53.185328Z","iopub.execute_input":"2022-07-21T20:36:53.186024Z","iopub.status.idle":"2022-07-21T20:38:38.263700Z","shell.execute_reply.started":"2022-07-21T20:36:53.185983Z","shell.execute_reply":"2022-07-21T20:38:38.262076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:38:44.485682Z","iopub.execute_input":"2022-07-21T18:38:44.486035Z","iopub.status.idle":"2022-07-21T18:38:44.493312Z","shell.execute_reply.started":"2022-07-21T18:38:44.486004Z","shell.execute_reply":"2022-07-21T18:38:44.492171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(test_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:38:47.741368Z","iopub.execute_input":"2022-07-21T18:38:47.742636Z","iopub.status.idle":"2022-07-21T18:38:47.750154Z","shell.execute_reply.started":"2022-07-21T18:38:47.742591Z","shell.execute_reply":"2022-07-21T18:38:47.749076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_sentiment(\"Why are religious folks not considered clinically insane?\")","metadata":{"execution":{"iopub.status.busy":"2022-07-21T18:47:04.328068Z","iopub.execute_input":"2022-07-21T18:47:04.328436Z","iopub.status.idle":"2022-07-21T18:47:04.338999Z","shell.execute_reply.started":"2022-07-21T18:47:04.328405Z","shell.execute_reply":"2022-07-21T18:47:04.338002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"outputs = torch.tensor([-0.3,-0.2,0.6,0,1])\nlabels = torch.tensor([0,0,0,1,1])\nprint(get_correct(outputs, labels))","metadata":{"execution":{"iopub.status.busy":"2022-07-21T19:55:42.762905Z","iopub.execute_input":"2022-07-21T19:55:42.763248Z","iopub.status.idle":"2022-07-21T19:55:42.771434Z","shell.execute_reply.started":"2022-07-21T19:55:42.763219Z","shell.execute_reply":"2022-07-21T19:55:42.770421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.array([1,1,1,1]) + np.array([1,2,3,4])","metadata":{"execution":{"iopub.status.busy":"2022-07-21T20:00:59.585160Z","iopub.execute_input":"2022-07-21T20:00:59.585534Z","iopub.status.idle":"2022-07-21T20:00:59.592737Z","shell.execute_reply.started":"2022-07-21T20:00:59.585503Z","shell.execute_reply":"2022-07-21T20:00:59.591697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}