{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Quora Insincere Questions Classification**","metadata":{}},{"cell_type":"markdown","source":"<font size = \"4\">**Importing Necessary Libraries**</font>","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:03:36.516769Z","iopub.execute_input":"2023-08-01T06:03:36.517965Z","iopub.status.idle":"2023-08-01T06:03:36.523674Z","shell.execute_reply.started":"2023-08-01T06:03:36.517895Z","shell.execute_reply":"2023-08-01T06:03:36.522594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')\ntest_df = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/test.csv')\nsub_df = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:03:42.387669Z","iopub.execute_input":"2023-08-01T06:03:42.388105Z","iopub.status.idle":"2023-08-01T06:03:49.022086Z","shell.execute_reply.started":"2023-08-01T06:03:42.388070Z","shell.execute_reply":"2023-08-01T06:03:49.020887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:04:49.462726Z","iopub.execute_input":"2023-08-01T06:04:49.463121Z","iopub.status.idle":"2023-08-01T06:04:49.475456Z","shell.execute_reply.started":"2023-08-01T06:04:49.463089Z","shell.execute_reply":"2023-08-01T06:04:49.474181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.columns","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:04:53.361879Z","iopub.execute_input":"2023-08-01T06:04:53.363123Z","iopub.status.idle":"2023-08-01T06:04:53.370984Z","shell.execute_reply.started":"2023-08-01T06:04:53.363075Z","shell.execute_reply":"2023-08-01T06:04:53.369654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-31T16:55:13.561943Z","iopub.execute_input":"2023-07-31T16:55:13.562401Z","iopub.status.idle":"2023-07-31T16:55:13.575020Z","shell.execute_reply.started":"2023-07-31T16:55:13.562364Z","shell.execute_reply":"2023-07-31T16:55:13.573945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-07-31T16:55:27.403126Z","iopub.execute_input":"2023-07-31T16:55:27.403484Z","iopub.status.idle":"2023-07-31T16:55:27.429375Z","shell.execute_reply.started":"2023-07-31T16:55:27.403455Z","shell.execute_reply":"2023-07-31T16:55:27.428320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.catplot(x='target', kind='count', data=train_df)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T16:55:31.751196Z","iopub.execute_input":"2023-07-31T16:55:31.751556Z","iopub.status.idle":"2023-07-31T16:55:32.627235Z","shell.execute_reply.started":"2023-07-31T16:55:31.751526Z","shell.execute_reply":"2023-07-31T16:55:32.626185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size='5'>The above catplot shows that there is a significant class imbalance present in the dataset. It is dealt in the further stage.<font/>","metadata":{}},{"cell_type":"markdown","source":"## **Data Preparation**","metadata":{}},{"cell_type":"markdown","source":"<font size = '3'>Before proceeding, first task is to preprocess the text data. For this purpose we use the nltk library.<font/>","metadata":{}},{"cell_type":"code","source":"import nltk\nnltk.download('punkt')\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\nfrom nltk.stem import SnowballStemmer\n\nnltk.download('stopwords')","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:05:00.871520Z","iopub.execute_input":"2023-08-01T06:05:00.872079Z","iopub.status.idle":"2023-08-01T06:05:02.246647Z","shell.execute_reply.started":"2023-08-01T06:05:00.872021Z","shell.execute_reply":"2023-08-01T06:05:02.245570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eng_stopwords = stopwords.words('english')\nprint(eng_stopwords)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:05:05.980139Z","iopub.execute_input":"2023-08-01T06:05:05.980561Z","iopub.status.idle":"2023-08-01T06:05:05.990442Z","shell.execute_reply.started":"2023-08-01T06:05:05.980527Z","shell.execute_reply":"2023-08-01T06:05:05.989162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stemmer = SnowballStemmer('english')","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:05:08.206385Z","iopub.execute_input":"2023-08-01T06:05:08.206993Z","iopub.status.idle":"2023-08-01T06:05:08.212379Z","shell.execute_reply.started":"2023-08-01T06:05:08.206946Z","shell.execute_reply":"2023-08-01T06:05:08.211260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def stem_tokenizer(sentence):\n    tokens = word_tokenize(sentence)\n    final_tokens = []\n    for token in tokens:\n        final_tokens.append(stemmer.stem(token))\n    return final_tokens\n\nprint(stem_tokenizer('My name is Kamlesh. I am working for the data analysis industry'))","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:05:09.238902Z","iopub.execute_input":"2023-08-01T06:05:09.239331Z","iopub.status.idle":"2023-08-01T06:05:09.266163Z","shell.execute_reply.started":"2023-08-01T06:05:09.239295Z","shell.execute_reply":"2023-08-01T06:05:09.265063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:06:26.512064Z","iopub.execute_input":"2023-08-01T06:06:26.512794Z","iopub.status.idle":"2023-08-01T06:06:26.518055Z","shell.execute_reply.started":"2023-08-01T06:06:26.512756Z","shell.execute_reply":"2023-08-01T06:06:26.516899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorizer = TfidfVectorizer(tokenizer = stem_tokenizer, stop_words = eng_stopwords, max_features=2000)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:06:27.841221Z","iopub.execute_input":"2023-08-01T06:06:27.841589Z","iopub.status.idle":"2023-08-01T06:06:27.846974Z","shell.execute_reply.started":"2023-08-01T06:06:27.841558Z","shell.execute_reply":"2023-08-01T06:06:27.845802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df = train_df.sample(200000)\nsample_df","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:06:29.706433Z","iopub.execute_input":"2023-08-01T06:06:29.707248Z","iopub.status.idle":"2023-08-01T06:06:29.834103Z","shell.execute_reply.started":"2023-08-01T06:06:29.707204Z","shell.execute_reply":"2023-08-01T06:06:29.832957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorizer.fit(sample_df['question_text'])","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:06:33.026095Z","iopub.execute_input":"2023-08-01T06:06:33.026479Z","iopub.status.idle":"2023-08-01T06:08:08.835298Z","shell.execute_reply.started":"2023-08-01T06:06:33.026446Z","shell.execute_reply":"2023-08-01T06:08:08.833964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorizer.get_feature_names_out()[:100] #viewing 100 features from the vocabulary","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:09:02.291232Z","iopub.execute_input":"2023-08-01T06:09:02.291625Z","iopub.status.idle":"2023-08-01T06:09:02.301501Z","shell.execute_reply.started":"2023-08-01T06:09:02.291591Z","shell.execute_reply":"2023-08-01T06:09:02.300360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs = vectorizer.transform(sample_df['question_text'])\ninputs.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:10:49.851594Z","iopub.execute_input":"2023-08-01T06:10:49.852157Z","iopub.status.idle":"2023-08-01T06:12:25.996075Z","shell.execute_reply.started":"2023-08-01T06:10:49.852112Z","shell.execute_reply":"2023-08-01T06:12:25.994979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_inputs = vectorizer.transform(test_df['question_text'])","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:13:02.176277Z","iopub.execute_input":"2023-08-01T06:13:02.176673Z","iopub.status.idle":"2023-08-01T06:16:03.684530Z","shell.execute_reply.started":"2023-08-01T06:13:02.176641Z","shell.execute_reply":"2023-08-01T06:16:03.683339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = sample_df['target']\ntargets","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:18:40.551633Z","iopub.execute_input":"2023-08-01T06:18:40.552042Z","iopub.status.idle":"2023-08-01T06:18:40.560422Z","shell.execute_reply.started":"2023-08-01T06:18:40.552009Z","shell.execute_reply":"2023-08-01T06:18:40.559396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_val, y_train, y_val = train_test_split(inputs, targets, train_size=0.75, random_state = 42)\nx_train","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:18:46.531048Z","iopub.execute_input":"2023-08-01T06:18:46.532206Z","iopub.status.idle":"2023-08-01T06:18:46.574304Z","shell.execute_reply.started":"2023-08-01T06:18:46.532159Z","shell.execute_reply":"2023-08-01T06:18:46.573051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:18:48.906401Z","iopub.execute_input":"2023-08-01T06:18:48.906786Z","iopub.status.idle":"2023-08-01T06:18:48.915105Z","shell.execute_reply.started":"2023-08-01T06:18:48.906753Z","shell.execute_reply":"2023-08-01T06:18:48.913987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import TensorDataset, DataLoader","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:18:51.451309Z","iopub.execute_input":"2023-08-01T06:18:51.452632Z","iopub.status.idle":"2023-08-01T06:18:57.055944Z","shell.execute_reply.started":"2023-08-01T06:18:51.452590Z","shell.execute_reply":"2023-08-01T06:18:57.054776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if torch.cuda.is_available():\n    device = 'cuda'\nelse:\n    device = 'cpu'","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:19:02.002216Z","iopub.execute_input":"2023-08-01T06:19:02.003355Z","iopub.status.idle":"2023-08-01T06:19:02.087699Z","shell.execute_reply.started":"2023-08-01T06:19:02.003316Z","shell.execute_reply":"2023-08-01T06:19:02.086271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:19:04.003881Z","iopub.execute_input":"2023-08-01T06:19:04.004330Z","iopub.status.idle":"2023-08-01T06:19:04.012396Z","shell.execute_reply.started":"2023-08-01T06:19:04.004294Z","shell.execute_reply":"2023-08-01T06:19:04.010804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_input_tensor = torch.tensor(x_train.toarray()).float().to(device)\nvalidation_input_tensor = torch.tensor(x_val.toarray()).float().to(device)\n\ntrain_target_tensor = torch.tensor(y_train.values).float().to(device)\nvalidation_target_tensor = torch.tensor(y_val.values).float().to(device)\n\ntest_tensor = torch.tensor(test_inputs.toarray()).float().to(device)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:19:31.116513Z","iopub.execute_input":"2023-08-01T06:19:31.116930Z","iopub.status.idle":"2023-08-01T06:19:54.796051Z","shell.execute_reply.started":"2023-08-01T06:19:31.116878Z","shell.execute_reply":"2023-08-01T06:19:54.794868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_target_tensor)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:22:20.395527Z","iopub.execute_input":"2023-08-01T06:22:20.396538Z","iopub.status.idle":"2023-08-01T06:22:20.532714Z","shell.execute_reply.started":"2023-08-01T06:22:20.396500Z","shell.execute_reply":"2023-08-01T06:22:20.530669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_tensor_dataset = TensorDataset(train_input_tensor, train_target_tensor)\nvalidation_tensor_dataset = TensorDataset(validation_input_tensor, validation_target_tensor)\ntest_tensor_dataset = TensorDataset(test_tensor)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:22:23.564231Z","iopub.execute_input":"2023-08-01T06:22:23.564989Z","iopub.status.idle":"2023-08-01T06:22:23.570086Z","shell.execute_reply.started":"2023-08-01T06:22:23.564943Z","shell.execute_reply":"2023-08-01T06:22:23.568985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_tensor_dataset[0:5]","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:22:25.645732Z","iopub.execute_input":"2023-08-01T06:22:25.649412Z","iopub.status.idle":"2023-08-01T06:22:25.681177Z","shell.execute_reply.started":"2023-08-01T06:22:25.649356Z","shell.execute_reply":"2023-08-01T06:22:25.679870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l_rate = 0.01","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:28:03.595923Z","iopub.execute_input":"2023-08-01T06:28:03.596308Z","iopub.status.idle":"2023-08-01T06:28:03.600902Z","shell.execute_reply.started":"2023-08-01T06:28:03.596275Z","shell.execute_reply":"2023-08-01T06:28:03.599895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_loader = DataLoader(train_tensor_dataset, batch_size=128)\nval_loader = DataLoader(validation_tensor_dataset, batch_size=128)\ntest_loader = DataLoader(test_tensor_dataset, batch_size=128)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:22:30.155444Z","iopub.execute_input":"2023-08-01T06:22:30.155854Z","iopub.status.idle":"2023-08-01T06:22:30.162230Z","shell.execute_reply.started":"2023-08-01T06:22:30.155820Z","shell.execute_reply":"2023-08-01T06:22:30.160831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, j in train_loader:\n    print(i.shape)\n    print(j.shape)\n    break","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:22:33.541246Z","iopub.execute_input":"2023-08-01T06:22:33.541669Z","iopub.status.idle":"2023-08-01T06:22:33.563128Z","shell.execute_reply.started":"2023-08-01T06:22:33.541633Z","shell.execute_reply":"2023-08-01T06:22:33.561994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn as nn\nimport torch.nn.functional as F","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:22:36.146148Z","iopub.execute_input":"2023-08-01T06:22:36.146897Z","iopub.status.idle":"2023-08-01T06:22:36.152021Z","shell.execute_reply.started":"2023-08-01T06:22:36.146861Z","shell.execute_reply":"2023-08-01T06:22:36.150897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class QuoraModel(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.layer1 = nn.Linear(2000, 512)\n        self.layer2 = nn.Linear(512, 256)\n        self.layer3 = nn.Linear(256, 64)\n        self.layer4 = nn.Linear(64, 1)\n    \n    def forward(self, inputs):\n        out = self.layer1(inputs)\n        out = F.relu(out)       # introducing non linearity\n        \n        out = self.layer2(out)\n        out = F.relu(out)\n        \n        out = self.layer3(out)\n        out = F.relu(out)\n        \n        out = self.layer4(out)\n\n        \n        return out\n    \nmodel = QuoraModel()\nmodel.to(device)       ","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:22:38.556353Z","iopub.execute_input":"2023-08-01T06:22:38.557194Z","iopub.status.idle":"2023-08-01T06:22:38.587332Z","shell.execute_reply.started":"2023-08-01T06:22:38.557151Z","shell.execute_reply":"2023-08-01T06:22:38.586376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torchinfo\ntorchinfo.summary(model = QuoraModel())","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:22:40.212095Z","iopub.execute_input":"2023-08-01T06:22:40.212509Z","iopub.status.idle":"2023-08-01T06:22:40.255928Z","shell.execute_reply.started":"2023-08-01T06:22:40.212477Z","shell.execute_reply":"2023-08-01T06:22:40.253958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score, accuracy_score","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:22:42.426402Z","iopub.execute_input":"2023-08-01T06:22:42.426796Z","iopub.status.idle":"2023-08-01T06:22:42.432192Z","shell.execute_reply.started":"2023-08-01T06:22:42.426764Z","shell.execute_reply":"2023-08-01T06:22:42.430988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training and Evaluation","metadata":{}},{"cell_type":"code","source":"def evaluator(model, dataLoader):\n    f1_s, accs, losses = [], [], []\n    \n    for batch in dataLoader:\n        batch_in, batch_targets = batch\n        batch_out = model(batch_in)\n        \n        probablities = torch.sigmoid(batch_out[:,0])\n\n    predicted = (probablities>0.5).int()\n    \n    batch_targets = batch_targets.cpu()\n    predicted = predicted.cpu()\n    probablities = probablities.cpu()\n\n    f1 = f1_score(batch_targets, predicted)\n    loss = F.binary_cross_entropy(probablities, batch_targets)\n    acc = accuracy_score(batch_targets, predicted)\n\n    f1_s.append(f1)\n    losses.append(loss)\n    accs.append(acc)\n    \n    dict = {'f1_s':torch.mean(torch.tensor(f1_s)).item(), 'accs':torch.mean(torch.tensor(accs)).item(), 'losses':torch.mean(torch.tensor(losses)).item()}\n    return dict['f1_s'], dict['accs'], dict['losses']","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:22:53.396333Z","iopub.execute_input":"2023-08-01T06:22:53.397136Z","iopub.status.idle":"2023-08-01T06:22:53.408435Z","shell.execute_reply.started":"2023-08-01T06:22:53.397079Z","shell.execute_reply":"2023-08-01T06:22:53.407306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluator(model, train_loader)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:22:54.426408Z","iopub.execute_input":"2023-08-01T06:22:54.427212Z","iopub.status.idle":"2023-08-01T06:23:01.014927Z","shell.execute_reply.started":"2023-08-01T06:22:54.427172Z","shell.execute_reply":"2023-08-01T06:23:01.013820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ADAM_optimizer = torch.optim.Adam(model.parameters(), lr = l_rate)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:23:03.668484Z","iopub.execute_input":"2023-08-01T06:23:03.669321Z","iopub.status.idle":"2023-08-01T06:23:03.675301Z","shell.execute_reply.started":"2023-08-01T06:23:03.669273Z","shell.execute_reply":"2023-08-01T06:23:03.673870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = []\ndef model_fit(epochs, model, train_dataLoader,validation_dataLoader):\n    optimizer = torch.optim.Adam(model.parameters(), lr = l_rate)\n    for epoch in range(epochs):\n        for batch in train_dataLoader:\n            batch_in, batch_targets = batch\n            \n            # calculate predicted outputs\n            batch_out = model(batch_in)\n            probablities = torch.sigmoid(batch_out[:,0])        # batch_out is represented like this [0]. To remove the brackets, [:,0] is used.\n            predicted = (probablities>0.5).int()\n            \n            # calculate loss\n            loss = F.binary_cross_entropy(probablities, batch_targets)\n            \n            # back propagration\n            loss.backward()\n            optimizer.step()\n            optimizer.zero_grad()\n            \n        \n        # evaluation part \n        f1, acc, loss = evaluator(model, validation_dataLoader)\n        print(f\"Epoch {epoch+1} | Loss {loss:.4f} | Accuracy {acc:.4f} | F-1 Score {f1:.4f}\")\n        history.append([acc, f1, loss])\n    return history","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:28:10.402370Z","iopub.execute_input":"2023-08-01T06:28:10.402754Z","iopub.status.idle":"2023-08-01T06:28:10.412421Z","shell.execute_reply.started":"2023-08-01T06:28:10.402721Z","shell.execute_reply":"2023-08-01T06:28:10.411267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_fit(20, model, train_loader, val_loader)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:28:15.062796Z","iopub.execute_input":"2023-08-01T06:28:15.063198Z","iopub.status.idle":"2023-08-01T06:29:44.766561Z","shell.execute_reply.started":"2023-08-01T06:28:15.063165Z","shell.execute_reply":"2023-08-01T06:29:44.765452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"losses = [item[2] for item in history]\nplt.plot(losses)\nplt.xlabel('epochs')\nplt.ylabel('losses')\nplt.title('losses vs no.of epochs')","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:30:01.338703Z","iopub.execute_input":"2023-08-01T06:30:01.339164Z","iopub.status.idle":"2023-08-01T06:30:01.824705Z","shell.execute_reply.started":"2023-08-01T06:30:01.339126Z","shell.execute_reply":"2023-08-01T06:30:01.823667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plotting accuracies\naccs = [item[0] for item in history]\nplt.plot(accs)\nplt.xlabel('epochs')\nplt.ylabel('accuracy')\nplt.title('accuracy vs no.of epochs')","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:30:05.752690Z","iopub.execute_input":"2023-08-01T06:30:05.753376Z","iopub.status.idle":"2023-08-01T06:30:06.115309Z","shell.execute_reply.started":"2023-08-01T06:30:05.753325Z","shell.execute_reply":"2023-08-01T06:30:06.114251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_score = [item[1] for item in history]\nplt.plot(f1_score)\nplt.xlabel('epochs')\nplt.ylabel('f1_score')\nplt.title('f1_score vs no.of epochs') ","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:30:09.573476Z","iopub.execute_input":"2023-08-01T06:30:09.574276Z","iopub.status.idle":"2023-08-01T06:30:09.960671Z","shell.execute_reply.started":"2023-08-01T06:30:09.574230Z","shell.execute_reply":"2023-08-01T06:30:09.959619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Testing","metadata":{}},{"cell_type":"code","source":"def predictionOutput(df):\n    inputs = vectorizer.transform(df['question_text'])\n    input_tensors = torch.tensor(inputs.toarray()).float().to(device)\n    output = model(input_tensors)\n    probablities = torch.sigmoid(output)[:,0]\n    prediction = (probablities>0.5).int()\n    return prediction","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:30:28.456337Z","iopub.execute_input":"2023-08-01T06:30:28.456747Z","iopub.status.idle":"2023-08-01T06:30:28.463050Z","shell.execute_reply.started":"2023-08-01T06:30:28.456713Z","shell.execute_reply":"2023-08-01T06:30:28.461931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predicted = predictionOutput(test_df)\ntest_predicted = test_predicted.cpu().numpy()","metadata":{"execution":{"iopub.status.busy":"2023-08-01T06:57:19.518844Z","iopub.execute_input":"2023-08-01T06:57:19.519274Z","iopub.status.idle":"2023-08-01T07:00:37.352459Z","shell.execute_reply.started":"2023-08-01T06:57:19.519240Z","shell.execute_reply":"2023-08-01T07:00:37.351340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df","metadata":{"execution":{"iopub.status.busy":"2023-08-01T07:00:44.098800Z","iopub.execute_input":"2023-08-01T07:00:44.099951Z","iopub.status.idle":"2023-08-01T07:00:44.115165Z","shell.execute_reply.started":"2023-08-01T07:00:44.099884Z","shell.execute_reply":"2023-08-01T07:00:44.113965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.prediction = test_predicted\nsub_df.to_csv('submission.csv',index=None)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T07:00:47.499335Z","iopub.execute_input":"2023-08-01T07:00:47.500175Z","iopub.status.idle":"2023-08-01T07:00:48.850687Z","shell.execute_reply.started":"2023-08-01T07:00:47.500130Z","shell.execute_reply":"2023-08-01T07:00:48.849415Z"},"trusted":true},"execution_count":null,"outputs":[]}]}