{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport torch\nimport time\nimport os\n\nfrom zipfile import ZipFile\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\nimport pandas as pd\nfrom sklearn.metrics import f1_score\n\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.model_selection import train_test_split\nimport os\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom tqdm.notebook import tqdm\nimport numpy as np\n#import Embeddings\nEPOCHS = 1\nBATCH_SIZE = 32\nN_EVAL = 100\nHIDDEN_DIM = 64\nSEED = 17\nclass LSTMNetwork(torch.nn.Module):\n    def __init__(self, input_dim, hidden_dim, layers, sequence_length):\n        super().__init__()\n        self.hidden_size = hidden_dim\n        self.sequence_length = sequence_length\n        self.recurrent_layer = nn.LSTM(input_size = input_dim, hidden_size = hidden_dim, num_layers = layers, batch_first = False)\n        self.classifier = nn.Linear(hidden_dim, 1)\n        self.sigmoid = nn.Sigmoid()\n\n    def forward(self, x):\n        output, (hn, cn) = self.recurrent_layer(x)\n        # REPLACE 25 WITH SEQUENCE LENGTH\n        answer = self.classifier(output[self.sequence_length-1])\n        answer = self.sigmoid(answer)\n        return answer\nclass RawWordsDataset(torch.utils.data.Dataset):\n    def __init__(self, data):\n        self.df = data\n        self.inputs = self.df.question_text.tolist() # list of questions\n        self.labels = self.df.target.tolist() # list of labels\n\n    def __getitem__(self, i):\n        # return the ith sample's string and label\n        return self.inputs[i], self.labels[i]\n\n    def __len__(self):\n        return len(self.labels)\nfrom torchtext.data import get_tokenizer\n\nmax_words = 25\nembed_len = 50\n\ntokenizer = get_tokenizer(\"basic_english\")\nfile='/kaggle/input/quora-insincere-questions-classification/embeddings.zip'\nwith ZipFile(file, 'r') as zip:\n    # printing all the contents of the zip file\n    zip.printdir()\n  \n    # extracting all the files\n    print('Extracting all the files now...')\n    zip.extract(\"glove.840B.300d/glove.840B.300d.txt\")\n    print('Done!')\n# https://coderzcolumn.com/tutorials/artificial-intelligence/how-to-use-glove-embeddings-with-pytorch\n\n#torch.save(model,'/content/drive/MyDrive')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-17T08:52:52.038685Z","iopub.execute_input":"2022-11-17T08:52:52.039152Z","iopub.status.idle":"2022-11-17T08:54:05.818233Z","shell.execute_reply.started":"2022-11-17T08:52:52.039111Z","shell.execute_reply":"2022-11-17T08:54:05.816915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-11-17T08:52:16.065081Z","iopub.execute_input":"2022-11-17T08:52:16.066345Z","iopub.status.idle":"2022-11-17T08:52:32.306153Z","shell.execute_reply.started":"2022-11-17T08:52:16.066281Z","shell.execute_reply":"2022-11-17T08:52:32.304904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torchtext\nfrom pathlib import Path\nstr_paths = '/kaggle/working/glove.840B.300d/glove.840B.300d.txt'\npath = Path(str_paths)\n\n\nglobal_vectors = torchtext.vocab.Vectors(path)\nprint('GloVe data loaded')\n","metadata":{"execution":{"iopub.status.busy":"2022-11-17T08:54:05.820782Z","iopub.execute_input":"2022-11-17T08:54:05.821310Z","iopub.status.idle":"2022-11-17T08:58:58.683105Z","shell.execute_reply.started":"2022-11-17T08:54:05.821258Z","shell.execute_reply":"2022-11-17T08:58:58.681330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embed_len = 300\ndef vectorize_batch(X):\n    # separate the question into individual tokens (words)\n    X = [tokenizer(x) for x in X]\n    # make all sentences have the same number of tokens, pad with empty string or cut as needed\n    X = [tokens+[\"\"] * (max_words-len(tokens))  if len(tokens)<max_words else tokens[:max_words] for tokens in X]\n    # note that this shape will require batch_first = true for the lstm, so we will transpose it at the end\n    X_tensor = torch.zeros(len(X), max_words, embed_len)\n    for i, tokens in enumerate(X):\n        X_tensor[i] = global_vectors.get_vecs_by_tokens(tokens)\n    # with the transpose, we can have batch_first = false for the lstm\n    return torch.transpose(X_tensor, 0, 1)\ndef glove_train(train_dataset, val_dataset, model, hyperparameters, n_eval):\n    \"\"\"\n    Trains and evaluates a model.\n\n    Args:\n        train_dataset:   PyTorch dataset containing training data.\n        val_dataset:     PyTorch dataset containing validation data.\n        model:           PyTorch model to be trained.\n        hyperparameters: Dictionary containing hyperparameters.\n        n_eval:          Interval at which we evaluate our model.\n    \"\"\"\n\n    # Get keyword arguments\n    batch_size, epochs = hyperparameters[\"batch_size\"], hyperparameters[\"epochs\"]\n\n    # Initialize dataloaders\n    train_loader = torch.utils.data.DataLoader(\n        train_dataset, batch_size=batch_size, shuffle=True\n    )\n\n    # Note: batch_size = len(val_dataset), so that's the whole validation set\n    val_loader = torch.utils.data.DataLoader(\n        val_dataset, batch_size=len(val_dataset), shuffle=True\n    )\n\n    # Initalize optimizer (for gradient descent) and loss function\n    optimizer = optim.Adam(model.parameters())\n    loss_fn = nn.BCELoss()\n\n    for epoch in range(epochs):\n        print(f\"Epoch {epoch + 1} of {epochs}\")\n\n        # Loop over each batch in the dataset\n        for batch, (X, y) in tqdm(enumerate(train_loader)):\n            # Predictions and loss\n            '''\n            inputs = Embeddings.vectorize_batch(X)\n            '''\n\n            inputs = vectorize_batch(X)\n            y = y.type(torch.float)\n\n            \n            pred = model(inputs)\n            pred = torch.flatten(pred)\n            loss = loss_fn(pred, y)\n\n            # Backpropagation and optimization\n            optimizer.zero_grad()\n            loss.backward()\n            optimizer.step()\n\n            # Periodically evaluate our model + log to Tensorboard\n            if batch % n_eval == 0:\n\n\n                # Compute training loss and accuracy.\n                accuracy = compute_accuracy(pred, y)\n                print(\"loss: \", loss)\n                print(\"accuracy: \", accuracy)\n\n                # Compute validation loss and accuracy.\n                val_loss, val_accuracy,val_f1 = evaluate(val_loader, model, loss_fn)\n                print(\"validation loss: \", val_loss)\n                print(\"validation accuracy: \", val_accuracy)\n                print(\"f1 score: \", val_f1)\n                # TODO: Log the results to Tensorboard.\n\n\n\ndef compute_accuracy(outputs, labels):\n    n_correct = (torch.round(outputs) == labels).sum().item()\n    n_total = len(outputs)\n    return n_correct / n_total\n\n\ndef evaluate(val_loader, model, loss_fn):\n    with torch.no_grad():\n        # There should only be one batch (the entire validation set)\n        for (X, y) in val_loader:\n            '''\n            inputs = Embeddings.vectorize_batch(X)\n            '''\n\n            inputs = vectorize_batch(X)\n            y = y.type(torch.float)\n\n\n            pred = model(inputs)\n            pred = torch.flatten(pred)\n            loss = loss_fn(pred, y)\n            f1 = f1_score(torch.round(pred).cpu(), y.cpu(), average='macro')\n            accuracy = compute_accuracy(pred, y)\n            return loss, accuracy, f1\ndata_path = '/kaggle/input/quora-insincere-questions-classification/train.csv'\ndata_pd = pd.read_csv(data_path)\ndata, val = train_test_split(data_pd, test_size = 0.05, stratify = data_pd['target'], shuffle = True, random_state = SEED)\ntrain_dataset = RawWordsDataset(data)\nval_dataset = RawWordsDataset(val)\nmodel = LSTMNetwork(embed_len, HIDDEN_DIM, 1, max_words)\n\nglove_train(\n    train_dataset=train_dataset,\n    val_dataset=val_dataset,\n    model=model,\n    hyperparameters={\"epochs\": EPOCHS, \"batch_size\": BATCH_SIZE},\n    n_eval=N_EVAL,\n)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-11-17T09:01:23.615012Z","iopub.execute_input":"2022-11-17T09:01:23.615461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_path = '/kaggle/input/quora-insincere-questions-classification/test.csv'\ntest_pd = pd.read_csv(test_path)\nguesses = []\ntest_size = test_pd[\"qid\"].size\nfor i in tqdm(range(test_size)):\n  input = test_pd.loc[i, \"question_text\"]\n  input_tokenized = tokenizer(input)\n  input_tokenized = input_tokenized+[\"\"] * (max_words-len(input_tokenized))  if len(input_tokenized)<max_words else input_tokenized[:max_words]\n  input_vectorized = global_vectors.get_vecs_by_tokens(input_tokenized)\n  pred = model(input_vectorized)\n  pred = torch.round(torch.squeeze(pred)).item()\n  guesses.append((bool)(pred))","metadata":{"execution":{"iopub.status.busy":"2022-11-17T21:48:27.729737Z","iopub.execute_input":"2022-11-17T21:48:27.730487Z","iopub.status.idle":"2022-11-17T21:48:27.808362Z","shell.execute_reply.started":"2022-11-17T21:48:27.730362Z","shell.execute_reply":"2022-11-17T21:48:27.807138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_answers_pd = pd.DataFrame(guesses, columns = [\"prediction\"])\nqids = test_pd[[\"qid\"]]\nsubmission_pd = pd.concat([qids, final_answers_pd], axis = 1)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-17T06:16:32.245799Z","iopub.execute_input":"2022-11-17T06:16:32.247451Z","iopub.status.idle":"2022-11-17T06:16:32.299410Z","shell.execute_reply.started":"2022-11-17T06:16:32.247396Z","shell.execute_reply":"2022-11-17T06:16:32.297574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_pd","metadata":{"execution":{"iopub.status.busy":"2022-11-17T06:16:42.281689Z","iopub.execute_input":"2022-11-17T06:16:42.282109Z","iopub.status.idle":"2022-11-17T06:16:42.309700Z","shell.execute_reply.started":"2022-11-17T06:16:42.282076Z","shell.execute_reply":"2022-11-17T06:16:42.308504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_pd.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T06:17:04.065292Z","iopub.execute_input":"2022-11-17T06:17:04.065676Z","iopub.status.idle":"2022-11-17T06:17:04.638445Z","shell.execute_reply.started":"2022-11-17T06:17:04.065646Z","shell.execute_reply":"2022-11-17T06:17:04.637423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}