{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":19018,"databundleVersionId":2703900,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Load Libraries","metadata":{}},{"cell_type":"code","source":"import time\nimport numpy as np\nimport pandas as pd\nfrom torchtext import vocab ## for glove vectors\nimport torch\nfrom torch import nn\nfrom torch.utils.data import TensorDataset, DataLoader\n# from torchtext.data import TabularDataset\nfrom nltk import word_tokenize","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-09T10:50:59.355880Z","iopub.execute_input":"2023-12-09T10:50:59.356483Z","iopub.status.idle":"2023-12-09T10:51:05.973855Z","shell.execute_reply.started":"2023-12-09T10:50:59.356428Z","shell.execute_reply":"2023-12-09T10:51:05.972754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(torch.__version__)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:05.976012Z","iopub.execute_input":"2023-12-09T10:51:05.976701Z","iopub.status.idle":"2023-12-09T10:51:05.982548Z","shell.execute_reply.started":"2023-12-09T10:51:05.976666Z","shell.execute_reply":"2023-12-09T10:51:05.981316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(r'/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\n# validation = pd.read_csv(r'/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:05.984331Z","iopub.execute_input":"2023-12-09T10:51:05.984771Z","iopub.status.idle":"2023-12-09T10:51:08.926170Z","shell.execute_reply.started":"2023-12-09T10:51:05.984733Z","shell.execute_reply":"2023-12-09T10:51:08.924970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n## Take small sample \ntrain = train[:1000]\ntrain2 = train.copy()\n# validation = validation[:1000]","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:08.927982Z","iopub.execute_input":"2023-12-09T10:51:08.929235Z","iopub.status.idle":"2023-12-09T10:51:08.935136Z","shell.execute_reply.started":"2023-12-09T10:51:08.929187Z","shell.execute_reply":"2023-12-09T10:51:08.933976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Preprocessing\n\n### A. train data","metadata":{}},{"cell_type":"code","source":"## preprocessing \nimport nltk \nfrom nltk.corpus import stopwords\n\nstopword = stopwords.words('english')\n\n# define a function to remove stopwords\ndef remove_stopwords(txt):\n    filtered_word = []\n    words = txt.split(' ')\n    for word in words:\n        if word not in stopword:\n            filtered_word.append(word)\n    return ' '.join(filtered_word)\n\n## clean text remove all special characters\nimport re, string\n\ndef clean_text(txt):\n    txt = str(txt).lower()\n    txt = re.sub('\\[.*?\\]','',txt)\n    txt = re.sub('https?://\\S+|www\\.\\S+','',txt)\n    txt = re.sub('<.*?>+','',txt)\n    txt =re.sub('[%s]' % re.escape(string.punctuation),'',txt)\n    txt = re.sub('\\n','',txt)\n    txt = re.sub('\\w*\\d\\w*','',txt)\n    return txt","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:08.939000Z","iopub.execute_input":"2023-12-09T10:51:08.939722Z","iopub.status.idle":"2023-12-09T10:51:08.955178Z","shell.execute_reply.started":"2023-12-09T10:51:08.939683Z","shell.execute_reply":"2023-12-09T10:51:08.954041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['comment_text'] = train['comment_text'].apply(clean_text)\ntrain['comment_text'] = train['comment_text'].apply(remove_stopwords)\n\n# train.to_csv('/kaggle/working/' + 'train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:08.956580Z","iopub.execute_input":"2023-12-09T10:51:08.957010Z","iopub.status.idle":"2023-12-09T10:51:09.264731Z","shell.execute_reply.started":"2023-12-09T10:51:08.956980Z","shell.execute_reply":"2023-12-09T10:51:09.263657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Load Data into DataLoader","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:09.266201Z","iopub.execute_input":"2023-12-09T10:51:09.266580Z","iopub.status.idle":"2023-12-09T10:51:09.271617Z","shell.execute_reply.started":"2023-12-09T10:51:09.266551Z","shell.execute_reply":"2023-12-09T10:51:09.270393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## B. Validation Data\n\n","metadata":{}},{"cell_type":"code","source":"# ## we need to get the stopwords for all five languagee\n# #append all the other languages stopwords\n\n# tuk_stopwords = stopwords.words('turkish')\n# rus_stopwords = stopwords.words('russian')\n# ital_stopwords = stopwords.words('italian')\n# french_stopwords = stopwords.words('french')\n# portugese_stopwords = stopwords.words('portuguese')\n# spanish_stopwords = stopwords.words('spanish')\n\n# test_stopwords = tuk_stopwords + rus_stopwords + ital_stopwords + french_stopwords + portugese_stopwords + spanish_stopwords\n\n# ## define a funct to remove stopwords from test data\n# def test_stopword(txt):\n#     words = txt.split(' ')\n#     filtered_word = []\n#     for word in words:\n#         if word not in test_stopwords:\n#             filtered_word.append(word)\n#     return ' '.join(filtered_word)\n\n# ## remove stopwords from test data \n# # test['content'] = test['content'].apply(test_stopword)\n# # test['content'] = test['content'].apply(clean_text)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:09.272978Z","iopub.execute_input":"2023-12-09T10:51:09.273386Z","iopub.status.idle":"2023-12-09T10:51:09.283376Z","shell.execute_reply.started":"2023-12-09T10:51:09.273354Z","shell.execute_reply":"2023-12-09T10:51:09.282104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ## validation stopwords\n# validation_stopword = tuk_stopwords + ital_stopwords + spanish_stopwords\n\n# def valid_stop(txt):\n#     words = txt.split(' ')\n#     filtered_word = []\n#     for word in words:\n#         if word not in validation_stopword:\n#             filtered_word.append(word)\n#     return ' '.join(filtered_word)\n\n# validation['comment_text'] = validation['comment_text'].apply(valid_stop)\n# validation['comment_text'] = validation['comment_text'].apply(clean_text)\n\n# validation.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:09.285040Z","iopub.execute_input":"2023-12-09T10:51:09.285535Z","iopub.status.idle":"2023-12-09T10:51:09.299296Z","shell.execute_reply.started":"2023-12-09T10:51:09.285489Z","shell.execute_reply":"2023-12-09T10:51:09.298332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #Approach 1 \n# # tokenize the comment_text,\n# ## use toxic as label \n# ## convert to tensor \n# ## \n\n# from nltk import word_tokenize\n\n# # tokenize function\n# def tokenize_text(txt):\n#     txt = txt.lower()\n#     tokens = []\n#     words = txt.split(' ')\n#     for word in words:\n#         token = word_tokenize(word)\n#         tokens.append(token)\n#     return tokens\n\n\n# ## use train['comment_text'] \n# ## get all the words \n# ## get  unique words  create vocab (tokenization is done of unique words)\n# ## tokenize them \n# ## use those to create index \n# ## \n\n# ## Builduing vocabulary \n# # count the words\n\n# tokenized_texts = train['comment_text'].apply(tokenize_text)\n        \n    \n","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:09.300353Z","iopub.execute_input":"2023-12-09T10:51:09.300692Z","iopub.status.idle":"2023-12-09T10:51:09.311989Z","shell.execute_reply.started":"2023-12-09T10:51:09.300664Z","shell.execute_reply":"2023-12-09T10:51:09.310970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ## create a library of unique_words\n# unique_word = []\n# for i, j in enumerate(train['comment_text']):\n#     words = j.split(' ')\n#     for word in words:\n#         if word not in unique_words:\n#             unique_word.append(word)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:09.313158Z","iopub.execute_input":"2023-12-09T10:51:09.313556Z","iopub.status.idle":"2023-12-09T10:51:09.326041Z","shell.execute_reply.started":"2023-12-09T10:51:09.313525Z","shell.execute_reply":"2023-12-09T10:51:09.325068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from nltk.tokenize import word_tokenize\n\n# #1. tokenize each comment\n# tokenized_text = []\n# for i,j in enumerate(train['comment_text']):\n#         word = j.lower()\n#         x = word_tokenize(word)\n#         tokenized_text.append(x)\n            \n            \n# #2. Get unique words\n# unique_word = set()\n\n# for sentence in tokenized_text:\n#     for word in sentence:\n#         unique_word.add(word)\n        \n                       \n# # 3. sort alphabetically \n# unique_word = list[unique_word]\n# unique_word = sorted(unique_word)\n\n# #4. Assign id to each token\n# word_to_id = {}\n# for id, word in enumerate(unique_word):\n#     word_to_id[word] = id\n\n# #6. Assign id to word\n# id_to_word = {}\n# for token,id in word_to_id.items():\n#     id_to_word[id] =token\n    \n# #7. Add padding to the snetences which has less length\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:09.327395Z","iopub.execute_input":"2023-12-09T10:51:09.327810Z","iopub.status.idle":"2023-12-09T10:51:09.338618Z","shell.execute_reply.started":"2023-12-09T10:51:09.327776Z","shell.execute_reply":"2023-12-09T10:51:09.337478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train['comment_text']","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:09.340379Z","iopub.execute_input":"2023-12-09T10:51:09.340876Z","iopub.status.idle":"2023-12-09T10:51:09.349142Z","shell.execute_reply.started":"2023-12-09T10:51:09.340825Z","shell.execute_reply":"2023-12-09T10:51:09.347986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Convert the text to tokenize for the model ","metadata":{}},{"cell_type":"code","source":"#define tokenizer\ndef tokenizer(txt):\n    word = word_tokenize(txt.lower())\n    return word\n\n#tokenize sentences\ntrain['comment_text'] = train['comment_text'].apply(tokenizer)\n\n# print(train['comment_text'])\n\nunique_word = set()\nfor sentence in train['comment_text']:\n    for word in sentence:\n        unique_word.add(word)\n        \n#convert unique words to list\nunique_word = list(unique_word)\n\n# print(unique_word)\n#create a dictionary of indexes of tokens\nword_to_idx = {}\nfor idx,word in enumerate(sorted(unique_word)):\n    word_to_idx[word] = idx\n    \n    \n#convert word to ids\ndef convert_words_to_ids(tokens):\n    idx = []\n    for token in tokens:\n        token_id = word_to_idx.get(token)\n        idx.append(token_id)\n    return idx\n        \n    \n\n# assign ids to words\ntrain['comment_text'] = train['comment_text'].apply(convert_words_to_ids)\n\nprint(train['comment_text'])\n#max_length of sentence\nlength_of_sequences = []\nfor i,j in enumerate(train['comment_text']):\n    x = len(train['comment_text'][i])\n    length_of_sequences.append(x)\nmax_length = max(length_of_sequences)\n\n#pad the sentences\npadded_sentence = []\nfor i ,sentence in enumerate(train['comment_text']):\n    length = len(sentence)\n    pad = sentence + [0]*(max_length-length)\n    padded_sentence.append(pad)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-12-09T10:51:09.356154Z","iopub.execute_input":"2023-12-09T10:51:09.356826Z","iopub.status.idle":"2023-12-09T10:51:09.942477Z","shell.execute_reply.started":"2023-12-09T10:51:09.356781Z","shell.execute_reply":"2023-12-09T10:51:09.941306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_length","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:09.943961Z","iopub.execute_input":"2023-12-09T10:51:09.944413Z","iopub.status.idle":"2023-12-09T10:51:09.952155Z","shell.execute_reply.started":"2023-12-09T10:51:09.944383Z","shell.execute_reply":"2023-12-09T10:51:09.950990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"padded_sentence[0]","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-12-09T10:51:09.953589Z","iopub.execute_input":"2023-12-09T10:51:09.954077Z","iopub.status.idle":"2023-12-09T10:51:09.981217Z","shell.execute_reply.started":"2023-12-09T10:51:09.954033Z","shell.execute_reply":"2023-12-09T10:51:09.979979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assign the sentences to \ntrain['comment_text'] = padded_sentence","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:09.982988Z","iopub.execute_input":"2023-12-09T10:51:09.983498Z","iopub.status.idle":"2023-12-09T10:51:09.990077Z","shell.execute_reply.started":"2023-12-09T10:51:09.983454Z","shell.execute_reply":"2023-12-09T10:51:09.988972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## CHECK if the words are not padded\nfor i,j in enumerate(train['comment_text']):\n    if len(train['comment_text'][i]) != max_length:\n        print(i)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:09.991704Z","iopub.execute_input":"2023-12-09T10:51:09.993017Z","iopub.status.idle":"2023-12-09T10:51:10.014410Z","shell.execute_reply.started":"2023-12-09T10:51:09.992979Z","shell.execute_reply":"2023-12-09T10:51:10.012979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. DEFINE RNN ARCHITECTURE\n\nRNN ARCHITECTURE : INPUT AND OUTPUT\n\nhttps://towardsdatascience.com/pytorch-basics-how-to-train-your-neural-net-intro-to-rnn-cb6ebc594677#:~:text=torch.-,nn.,batch%2C%20num_directions%20*%20hidden_size)%20.\n\nrnn class\n\n1. **for unbatched** \ninput tensor (sequence_length,input_size)\noutput tensor(sequence_length,D * hidden_size)\n\nD = 2 IF bidirectional =True, otherwise 1\n\n\n2. **when batch_first = False**\ninput tensor(sequence_length,batch_size, input_size), \noutput tensor(sequence_length,batch_size, D*hidden_size)\n\nwhen batch_first = True\ninput tensor(batch_size,sequence_length,input_size)\noutput tensor(batch_size,sequence_length,D*hidden_size)\n\n\n3. **Output of rnn is output and h_n**\n\n'output' contains hidden states for each time step of each sequence in the batch\nShape of the output varies according to batch_first parameter\n\n'h_n' contains the final hidden state for each elements in the batch \n\nfor unbatched input\nhn is (D*num_layers,hidden_size)\nh","metadata":{}},{"cell_type":"code","source":"# ## design model\n# class RNNmodel(nn.Module): \n#     def __init__(self,vocab_size,embedding_dimension,hidden_dim,output_dim):\n#         super(RNNmodel,self).__init__()\n#         self.embeddinglayer = nn.Embedding(num_embeddings = vocab_size,embedding_dim = embedding_dimension)\n#         '''num_embeddings -> size of dictionary of embeddings(total number of unique words or tokens\n#         in your vocabulary).\n#         embedding_dim -> size of each embedding vector(dimensionality of embeddings)'''\n#         self.rnn = nn.RNN(input_size=embedding_dimension,hidden_size=hidden_dim)\n#         self.linear = nn.Linear(hidden_dim,output_dim)\n#         self.sigmoid = nn.Sigmoid()\n        \n#     def forward(self,text):\n#         embedded_layer = self.embeddinglayer(text)\n#         out,hidden  = self.rnn(embedded_layer)\n#         output2 = self.linear(hidden)\n#             # for classification task we use the output value of all time steps in last rnn\n#             ## for other seq 2 seq task we use hidden value of last time step \n#         final = self.sigmoid(output2)\n#         return final ","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:10.016981Z","iopub.execute_input":"2023-12-09T10:51:10.017393Z","iopub.status.idle":"2023-12-09T10:51:10.024980Z","shell.execute_reply.started":"2023-12-09T10:51:10.017360Z","shell.execute_reply":"2023-12-09T10:51:10.023976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The output of the `self.rnn(embedded)` statement in PyTorch, when using a basic RNN layer, consists of two main components: **`output`** and **`hidden`**. Let's break down what each of these represents and their shapes:\n\n### `output`\n- **`output` contains the hidden states for each time step of each sequence in the batch.**\n- Its shape varies depending on whether the input was batched and the `batch_first` parameter.\n\n#### Shape of `output`\n1. **Unbatched Input (`batch_first=False`)**\n   - Shape: `(seq_len, batch_size, num_directions * hidden_size)`\n   - Here, `seq_len` is the length of the sequence, `batch_size` is the number of sequences (if batched), `num_directions` is 1 for a simple RNN (2 for bidirectional RNNs), and `hidden_size` is the size of the hidden layer.\n\n2. **Batched Input (`batch_first=True`)**\n   - Shape: `(batch_size, seq_len, num_directions * hidden_size)`\n   - The same dimensions as above, but the batch size and sequence length dimensions are swapped.\n\n### `hidden`\n- **`hidden` represents the final hidden state for each element in the batch.**\n- It's particularly important for understanding the final state of the sequence, which is often used in sequence-to-label tasks (like classification).\n\n#### Shape of `hidden`\n- Shape: `(num_layers * num_directions, batch_size, hidden_size)`\n- `num_layers` is the number of layers in the RNN, `num_directions` is 1 for a simple RNN (and 2 for bidirectional), `batch_size` is the number of sequences, and `hidden_size` is the size of the hidden layer.\n\n### Explanation\n1. **For Each Time Step (`output`):**\n   - The RNN processes each element of the sequence one by one. At each time step, it updates its hidden state based on the current input and the previous hidden state.\n   - The **`output` tensor collects these hidden states for each time step across all sequences in the batch. It's useful if you need to do further processing at each time step (e.g., in sequence-to-sequence models).**\n\n2. **Final State of the Sequence (`hidden`):**\n   - After processing the last element of the sequence, the RNN's hidden state represents the final encoding of the entire sequence.\n   - The **`hidden` tensor captures this final state. It's especially useful for tasks like classification where the entire sequence's context needs to be considered for a single prediction.**\n\n### In Summary\n- `output`: Provides a full sequence of hidden states, useful for tasks where every time step's output is important.\n- `hidden`: Provides the final hidden state, useful for tasks where the final state of the sequence is critical (like classification).\n\nThe structure and shape of these outputs allow RNNs to be versatile for a variety of sequence processing tasks in PyTorch.","metadata":{}},{"cell_type":"markdown","source":"## I was making mistake by not squeezing the hidden dimsion and my output from forward function is coming of size (16,793)\n\nThe issue you're encountering seems to be related to the dimensions of the output from your RNN model. When using an RNN for binary classification, the typical approach is to use the last hidden state to make a prediction. However, the shape of the output seems to be inconsistent with what is expected for binary classification.\n\nLet's break down the potential issues and solutions:\n\n1. **Understanding RNN Output:** The `self.rnn` layer in your code returns two outputs: `out` and `hidden`. The `out` contains the hidden states from all timesteps, while `hidden` is the last hidden state. For binary classification, you generally use the last hidden state.\n\n2. **Shape of RNN Output:** The output shape of `hidden` from an RNN in PyTorch is `(num_layers * num_directions, batch, hidden_size)`. If you are using a single layer, unidirectional RNN, this shape becomes `(1, batch_size, hidden_size)`. \n\n3. **Linear Layer Dimension Issue:** You have defined your linear layer with an output dimension of 2 (`self.linear = nn.Linear(hidden_dim, 2)`). This is suitable for a multi-class classification problem with 2 classes, but for binary classification, you typically have a single output unit which predicts the probability of one class (with the other class's probability being `1 - predicted_probability`). Therefore, your linear layer should have an output dimension of 1 for binary classification.\n\n4. **Squeezing the Hidden State:** Before passing the hidden state to the linear layer, you should squeeze the first dimension (which is 1) to match the input shape expectation of the linear layer. \n\n5. **Sigmoid Activation:** The sigmoid activation is correctly used for binary classification. However, you should apply it after the linear layer on its output, ensuring that the output is a single value representing the probability of the positive class for each instance in the batch.\n\nHere's how you can modify your `forward` method:\n\n```python\ndef forward(self, text):\n    embedded_layer = self.embeddinglayer(text)\n    out, hidden = self.rnn(embedded_layer)\n    # hidden is of shape (1, batch_size, hidden_size), squeeze the first dimension\n    hidden = hidden.squeeze(0) # Now hidden is of shape (batch_size, hidden_size)\n    output2 = self.linear(hidden) # Now output2 is of shape (batch_size, 1)\n    final = self.sigmoid(output2)\n    return final.squeeze()  # Ensuring the output is of shape (batch_size)\n```\n\nAnd make sure your linear layer is defined for binary classification:\n\n```python\nself.linear = nn.Linear(hidden_dim, 1)\n```\n\nThis should correct the dimensionality issue and produce an output tensor that matches the size of your batch, thus resolving the mismatch error.","metadata":{}},{"cell_type":"code","source":"# vocab_size","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:54:03.529043Z","iopub.execute_input":"2023-12-09T10:54:03.529511Z","iopub.status.idle":"2023-12-09T10:54:03.536491Z","shell.execute_reply.started":"2023-12-09T10:54:03.529477Z","shell.execute_reply":"2023-12-09T10:54:03.535321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## design model\nclass RNNmodel(nn.Module): \n    def __init__(self,vocab_size,embedding_dimension,hidden_dim,output_dim):\n        super(RNNmodel,self).__init__()\n        self.embeddinglayer = nn.Embedding(num_embeddings = vocab_size,embedding_dim = embedding_dimension)\n        '''num_embeddings -> size of dictionary of embeddings(total number of unique words or tokens\n        in your vocabulary).\n        embedding_dim -> size of each embedding vector(dimensionality of embeddings)'''\n        self.rnn = nn.RNN(input_size=embedding_dimension,hidden_size=hidden_dim,batch_first =True)\n        self.linear = nn.Linear(hidden_dim,output_dim)\n        self.sigmoid = nn.Sigmoid()\n        \n    def forward(self,text):\n        embedded_layer = self.embeddinglayer(text)\n        print(f\"shape of embedded_layer: {embedded_layer.shape}\")\n        out,hidden  = self.rnn(embedded_layer)\n        print(f\"Shape of hidden state after rnn ,before squeeze: {hidden.shape}\")\n        # hidden is of shape (1,batch_size,hidden_size), will squeeze the first dimesnion\n        hidden = hidden.squeeze(0) # now hidden is of shape (batch_size,hidden_size)\n        print(f\"shape of input to linear{hidden.shape}\")\n        output2 = self.linear(hidden)# output 2 is of shape (batch_size,1)\n        # for seq 2 seq task we use the output value of all time steps in last rnn\n        ## for classification task we use hidden value of last time step \n        print(f\"shape of output after linear:{output2.shape}\")\n        final = self.sigmoid(output2)\n        return final.squeeze()","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:10.026188Z","iopub.execute_input":"2023-12-09T10:51:10.026630Z","iopub.status.idle":"2023-12-09T10:51:10.042596Z","shell.execute_reply.started":"2023-12-09T10:51:10.026587Z","shell.execute_reply":"2023-12-09T10:51:10.041471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Mistake : I was not putting batch_first = true in nn.rnn\n(\nself.rnn = nn.RNN(input_size=embedding_dimension,hidden_size=hidden_dim,batch_first =True)\n) \nwhich caused a error that hidden dimension to be of shape(1,763,256) (batch,sequence_length,hidden_dimesnion), but the hidden state shpuld have a shape of (1,16,256) i.e (1,batch_size,hidden_size) after the rnn**\n\nFrom your provided code and the issue description, it seems that the problem might not be in the model architecture itself, but rather in how the data is being processed or fed into the model. The output size of `793` is particularly intriguing as it does not directly correspond to a typical dimension in an RNN for binary classification (like batch size or hidden size).\n\nHere are some potential issues to check:\n\n1. **Data Loader and Batch Processing:**\n   - Ensure that `X_train` and `y_train` in your training loop are tensors of the correct shape. `X_train` should be a tensor of shape `[batch_size, seq_len]`, where `seq_len` is the length of your sequences, and `y_train` should be of shape `[batch_size]`.\n   - In your training loop, you are iterating over `train_loader` but only using the last batch for training and validation. The line `X_train, y_train = batch` and similar for validation should be inside the loop body, not outside.\n\n2. **Input to the Model:**\n   - Verify that the input text data (`X_train` and `X_val`) is correctly preprocessed and converted to tensor format suitable for input to your RNN. This includes tokenization, numericalization (converting words to indices), and padding if necessary.\n\n3. **Dimension of Hidden State in RNN:**\n   - Double-check that the `hidden_dim` parameter in your RNNmodel class matches the expected size. The hidden state should have a shape of `[1, batch_size, hidden_size]` after the RNN and before squeezing.\n\n4. **Output Dimension of Linear Layer:**\n   - For binary classification, the `output_dim` in your model's constructor should be `1`, and it looks like it might be correctly set. Just verify it.\n\n5. **Check Model Output:**\n   - After getting `y_pred` from your model, print its shape to confirm it is `[batch_size]`. If it's not, there might be an issue in the forward pass of your model.\n\n6. **Error in Data Preparation or Model Call:**\n   - There might be an issue in how the data is being prepared or an inconsistency in how the model is called with the data. Double-check the preprocessing steps.\n\n\n","metadata":{}},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:10.044288Z","iopub.execute_input":"2023-12-09T10:51:10.044773Z","iopub.status.idle":"2023-12-09T10:51:10.087056Z","shell.execute_reply.started":"2023-12-09T10:51:10.044729Z","shell.execute_reply":"2023-12-09T10:51:10.085844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## load into dataloaders\nX = train['comment_text']\ny = train['toxic'].values\nX_tensor = torch.tensor(X,dtype = torch.long)\ny_tensor= torch.tensor(y,dtype =torch.float)\n\ndataset = TensorDataset(X_tensor,y_tensor)\n# train_dataset = DataLoader(dataset,batch_size=16,shuffle=True)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:10.088569Z","iopub.execute_input":"2023-12-09T10:51:10.089370Z","iopub.status.idle":"2023-12-09T10:51:10.261492Z","shell.execute_reply.started":"2023-12-09T10:51:10.089335Z","shell.execute_reply":"2023-12-09T10:51:10.260364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## USing the train data only as validation data because for validation data the unique words needs tonbe added into vocabulary \n\nfrom torch.utils.data import random_split\n# Split the dataset into training and testing sets\ntrain_size = int(0.8 * len(dataset))\ntest_size = len(dataset) - train_size\ntrain_dataset, val_dataset = random_split(dataset, [train_size, test_size])\n\n# Create data loaders for the train and test sets\ntrain_loader = DataLoader(train_dataset, batch_size=16, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=16, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:10.263320Z","iopub.execute_input":"2023-12-09T10:51:10.263847Z","iopub.status.idle":"2023-12-09T10:51:10.283525Z","shell.execute_reply.started":"2023-12-09T10:51:10.263812Z","shell.execute_reply":"2023-12-09T10:51:10.282207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i,j in train_dataset:\n    print(i,j)\n    break\n## i -> X_tensor","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:10.284972Z","iopub.execute_input":"2023-12-09T10:51:10.285421Z","iopub.status.idle":"2023-12-09T10:51:10.339532Z","shell.execute_reply.started":"2023-12-09T10:51:10.285381Z","shell.execute_reply":"2023-12-09T10:51:10.338056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"i -> X_tensor contains tokenized  text data, i is a subset of this tensor corresponding to the batch size.\n\nj -> batch of labels from y_tensor","metadata":{}},{"cell_type":"code","source":"vocab_size = len(unique_word)\nembedding_dimension = 100\nhidden_dim = 256\noutput_dim = 1\n\nmodel = RNNmodel(vocab_size = vocab_size,embedding_dimension=embedding_dimension,hidden_dim=hidden_dim,output_dim=output_dim)\n# model.to(device)\n\n\n## define a levaluation function\nloss_fn = nn.BCELoss()\n## optimizer\noptimizer = torch.optim.Adam(model.parameters(),\n                            lr = 0.001)\n# define evaluation \nfrom sklearn.metrics import roc_auc_score\ndef roc_score(y_train,y_pred):\n    y_train = y_train.detach().cpu().numpy()\n    y_pred = y_pred.detach().cpu().numpy()\n    roc = roc_auc_score(y_train,y_pred)\n    return roc","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:10.341166Z","iopub.execute_input":"2023-12-09T10:51:10.341886Z","iopub.status.idle":"2023-12-09T10:51:10.366589Z","shell.execute_reply.started":"2023-12-09T10:51:10.341843Z","shell.execute_reply":"2023-12-09T10:51:10.365386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n## Training\n\nepochs = 1\n\n\n\nfor epoch in range(epochs):\n    model.train()\n    for batch in train_loader:\n        X_train , y_train = batch\n        y_pred = model(X_train)\n        y_pred = y_pred.squeeze()\n        loss = loss_fn(y_pred,y_train)\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n    \n    model.eval()\n    with torch.no_grad():\n        for batches in val_loader:\n            X_val , y_val = batches\n            pred = model(X_val)\n            val_loss = loss_fn(pred,y_val)\n            #val_roc = roc_score(y_val,pred)\n    \n    if epoch%10==0:\n        print(f\"epoch {epoch} | train_loss {loss} | validation loss {val_loss} \")\n        \n        ","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:10.368541Z","iopub.execute_input":"2023-12-09T10:51:10.369026Z","iopub.status.idle":"2023-12-09T10:51:38.004212Z","shell.execute_reply.started":"2023-12-09T10:51:10.368982Z","shell.execute_reply":"2023-12-09T10:51:38.002949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred.shape,y_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:38.006299Z","iopub.execute_input":"2023-12-09T10:51:38.007185Z","iopub.status.idle":"2023-12-09T10:51:38.014849Z","shell.execute_reply.started":"2023-12-09T10:51:38.007140Z","shell.execute_reply":"2023-12-09T10:51:38.013656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:38.017176Z","iopub.execute_input":"2023-12-09T10:51:38.018094Z","iopub.status.idle":"2023-12-09T10:51:38.028198Z","shell.execute_reply.started":"2023-12-09T10:51:38.018051Z","shell.execute_reply":"2023-12-09T10:51:38.026868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Architecture of LSTM\n\n- 1. Output of **`nn.lstm`**\n\n- A. **output** - Contains the output feature(h_t) from last layer of the LSTM, for each t.\n        -- Used for **`sequence labelling`**(where we want an output for each input in the sequence)\n\n\n- B. **hidden_state(h_n)/short term memory** - Contains the final hidden state for each element in the sequence.\n        -- Used for **`sentiment analysis`** where we a single output for the entire sequence.\n\n\n- C. **cell_state(c_n)/Long term memory** - Contains the final cell state for each element in the state","metadata":{}},{"cell_type":"code","source":"class lstmmodel(nn.Module):\n    def __init__(self,vocab_size,\n                     embedding_dimension,\n                    hidden_dimension,\n                output_dimesnion):\n        super(lstmmodel).__init__(self)\n        self.embedding = nn.Embedding(num_embeddings=vocab_size,\n                                       embedding_dim = embedding_dimension)\n#         self.rnn = self.RNN(input_size=embedding_dimension,#input features\n#                            hidden_size= hidden_dimension,#number of neurons in hidden state h\n#                            num_layers = 1,## number of stacked rnn\n#                            bias = True,\n#                            batch_first = True)\n        self.lstm = nn.LSTM(input_size = embedding_dimension,#input features\n                             hidden_size = hidden_dimension,# neurons in hidden layer\n                             num_layers = 1,#number of lstm layers\n                             bias= True,\n                             batch_first = True,\n                             bidirectional = False# not bidirectional \n                             )\n        self.linear = nn.Linear(in_features=hidden_dimension,\n                                  output_features = output_dimension,\n                                 bias = True)\n        self.sigmoid = nn.Sigmoid()\n        \n    def forward(self,text):\n        embedding = self.embedding(text)\n        out,(hidden_state,cell_state) = self.lstm(embedding) # output of lstm is output,(hidden,)\n        #squeeze hidden\n        hidden_state = hidden_state.squeeze(0)\n        output2 = self.linear(hidden_state)\n        final = self.sigmoid(output2)\n        return final\n        ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tokenize using Count Vectorizer","metadata":{}},{"cell_type":"code","source":"## create Words into vectors using Count Vectorizer\ntrain = train2\n\nfrom sklearn.feature_extraction.text import CountVectorizer\ncvec = CountVectorizer()\n## tokenize using count vectorizer\nX = cvec.fit_transform(train['comment_text'])\ny = train['toxic'].values\n\n# convert to pytorch tensors\nfrom torch.utils.data import TensorDataset, DataLoader\nX_tensor = torch.tensor(X.toarray()).float()\ny_tensor = torch.tensor(y).long()\n\n## create Tensor Dataset\ndataset = TensorDataset(X_tensor,y_tensor)\n\ntrain_loader = DataLoader(dataset,batch_size=32,shuffle = True)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:38.029827Z","iopub.execute_input":"2023-12-09T10:51:38.030382Z","iopub.status.idle":"2023-12-09T10:51:38.273824Z","shell.execute_reply.started":"2023-12-09T10:51:38.030310Z","shell.execute_reply":"2023-12-09T10:51:38.272847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Same way tfidf , glove can be used** \n\n- from sklearn.feature selection import tfidf\n- use it to tokenize the words.\n- convert the feature and labels to tensors\n- load it into tensor dataset\n- load it into dataloaders\n- design an architecture rnn, lstm \n- tokens need to be fed into embeddings\n- first design embedding layer\n- feed those embeddings to architecture (RNN,LSTM)\n- **The correct output needs to be used depending on the type of task performed**\n- Feed those output as per specific task into further nn.modules\n- Above classification task the output is fed into linear and then sigmoid to convert it into between o to 1 for binary classification.\n- write the training loop \n- call the features and label tensors for feeding into model using for loop on dataloaders.\n- train the loop and call evaluation function to evaluate the model.","metadata":{}},{"cell_type":"code","source":"## same way tfifd , glove can be used ","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:38.275267Z","iopub.execute_input":"2023-12-09T10:51:38.276185Z","iopub.status.idle":"2023-12-09T10:51:38.282159Z","shell.execute_reply.started":"2023-12-09T10:51:38.276122Z","shell.execute_reply":"2023-12-09T10:51:38.281038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## use token","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:38.283994Z","iopub.execute_input":"2023-12-09T10:51:38.284388Z","iopub.status.idle":"2023-12-09T10:51:38.292548Z","shell.execute_reply.started":"2023-12-09T10:51:38.284354Z","shell.execute_reply":"2023-12-09T10:51:38.291159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## SIMPLE RNN \n## DATASET CLASS\n## load data into dataloader\n## define architecture\n## dfine loss function\n## define optimizer \n## train\n## evaluate\n## predict","metadata":{"execution":{"iopub.status.busy":"2023-12-09T10:51:38.294183Z","iopub.execute_input":"2023-12-09T10:51:38.295329Z","iopub.status.idle":"2023-12-09T10:51:38.303720Z","shell.execute_reply.started":"2023-12-09T10:51:38.295292Z","shell.execute_reply":"2023-12-09T10:51:38.302763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}