{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# HW1: Frame-Level Speech Recognition","metadata":{"id":"F9ERgBpbcMmB"}},{"cell_type":"markdown","source":"In this homework, you will be working with MFCC data consisting of 28 features at each time step/frame. Your model should be able to recognize the phoneme occured in that frame.","metadata":{"id":"CLkH6GMGcWcE"}},{"cell_type":"markdown","source":"# Libraries","metadata":{"id":"z4vZbDmJvMp1"}},{"cell_type":"code","source":"!pip install torchsummary wandb --quiet","metadata":{"id":"rwYu9sSUnSho","outputId":"cb512053-c0c5-4f0d-e75d-39dcb02b7a96","execution":{"iopub.status.busy":"2023-09-20T21:17:36.780732Z","iopub.execute_input":"2023-09-20T21:17:36.781385Z","iopub.status.idle":"2023-09-20T21:17:50.737485Z","shell.execute_reply.started":"2023-09-20T21:17:36.781352Z","shell.execute_reply":"2023-09-20T21:17:50.736251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport numpy as np\nfrom torchsummary import summary\nfrom torch.optim.lr_scheduler import StepLR\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\nimport sklearn\nimport gc\nimport zipfile\nimport pandas as pd\nfrom tqdm.auto import tqdm\nimport os\nimport datetime\nimport wandb\ndevice = 'cuda' if torch.cuda.is_available() else 'cpu'\nprint(\"Device: \", device)","metadata":{"id":"qI4qfx7tiBZt","outputId":"b12e2f48-d8d1-4aed-ead1-2b69c1af1248","execution":{"iopub.status.busy":"2023-09-20T21:17:50.740729Z","iopub.execute_input":"2023-09-20T21:17:50.741449Z","iopub.status.idle":"2023-09-20T21:17:55.119458Z","shell.execute_reply.started":"2023-09-20T21:17:50.741418Z","shell.execute_reply":"2023-09-20T21:17:55.118441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### If you are using colab, you can import google drive to save model checkpoints in a folder\n# from google.colab import drive\n# drive.mount('/content/drive')","metadata":{"id":"8yBgXjKV1O0Z","execution":{"iopub.status.busy":"2023-09-20T21:17:55.121068Z","iopub.execute_input":"2023-09-20T21:17:55.121722Z","iopub.status.idle":"2023-09-20T21:17:55.125710Z","shell.execute_reply.started":"2023-09-20T21:17:55.121686Z","shell.execute_reply":"2023-09-20T21:17:55.124727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### PHONEME LIST\nPHONEMES = [\n            '[SIL]',   'AA',    'AE',    'AH',    'AO',    'AW',    'AY',\n            'B',     'CH',    'D',     'DH',    'EH',    'ER',    'EY',\n            'F',     'G',     'HH',    'IH',    'IY',    'JH',    'K',\n            'L',     'M',     'N',     'NG',    'OW',    'OY',    'P',\n            'R',     'S',     'SH',    'T',     'TH',    'UH',    'UW',\n            'V',     'W',     'Y',     'Z',     'ZH',    '[SOS]', '[EOS]']","metadata":{"id":"N-9qE20hmCgQ","execution":{"iopub.status.busy":"2023-09-20T21:17:55.128664Z","iopub.execute_input":"2023-09-20T21:17:55.129230Z","iopub.status.idle":"2023-09-20T21:17:55.142871Z","shell.execute_reply.started":"2023-09-20T21:17:55.129173Z","shell.execute_reply":"2023-09-20T21:17:55.142139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Kaggle","metadata":{"id":"ZIi0Big7vPa9"}},{"cell_type":"markdown","source":"This section contains code that helps you install kaggle's API, creating kaggle.json with you username and API key details. Make sure to input those in the given code to ensure you can download data from the competition successfully.","metadata":{"id":"BBCbeRhixGM7"}},{"cell_type":"code","source":"#!pip install --upgrade --force-reinstall --no-deps kaggle==1.5.8\n#!mkdir /root/.kaggle\n\n#with open(\"/root/.kaggle/kaggle.json\", \"w+\") as f:\n#    f.write('{\"username\":\"cemadatepe\",\"key\":\"1758b98ee2098ea1266f43b9afaa3b16\"}')\n    # Put your kaggle username & key here\n\n#!chmod 600 /root/.kaggle/kaggle.json","metadata":{"id":"GIw_oVlUvJ3A","outputId":"912d7737-cdf1-4870-a363-41ab0793526f","execution":{"iopub.status.busy":"2023-09-20T21:17:55.144413Z","iopub.execute_input":"2023-09-20T21:17:55.145139Z","iopub.status.idle":"2023-09-20T21:17:55.153271Z","shell.execute_reply.started":"2023-09-20T21:17:55.145106Z","shell.execute_reply":"2023-09-20T21:17:55.152442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# commands to download data from kaggle\n\n#!kaggle competitions download -c 11785-hw1p2-f23\n#!mkdir '/content/data'\n\n#!unzip -qo /content/11785-hw1p2-f23.zip -d '/content/data'","metadata":{"id":"ixbs634lvOaR","outputId":"403b97b4-81fb-4212-f52b-24acd777ffe8","execution":{"iopub.status.busy":"2023-09-20T21:17:55.154745Z","iopub.execute_input":"2023-09-20T21:17:55.155183Z","iopub.status.idle":"2023-09-20T21:17:55.163347Z","shell.execute_reply.started":"2023-09-20T21:17:55.155152Z","shell.execute_reply":"2023-09-20T21:17:55.162454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset","metadata":{"id":"Vuzce0_TdcaR"}},{"cell_type":"markdown","source":"This section covers the dataset/dataloader class for speech data. You will have to spend time writing code to create this class successfully. We have given you a lot of comments guiding you on what code to write at each stage, from top to bottom of the class. Please try and take your time figuring this out, as it will immensely help in creating dataset/dataloader classes for future homeworks.\n\nBefore running the following cells, please take some time to analyse the structure of data. Try loading a single MFCC and its transcipt, print out the shapes and print out the values. Do the transcripts look like phonemes?","metadata":{"id":"2_7QgMbBdgPp"}},{"cell_type":"code","source":"# Dataset class to load train and validation data\nclass AudioDataset(torch.utils.data.Dataset):\n\n    def __init__(self, root, phonemes = PHONEMES, context=0, partition= \"train-clean-100\"): # Feel free to add more arguments\n\n        self.context    = context\n        self.phonemes   = phonemes\n        # TODO: MFCC directory - use partition to acces train/dev directories from kaggle data using root\n        self.mfcc_dir       = root + '/' + partition + '/mfcc/'\n        # TODO: Transcripts directory - use partition to acces train/dev directories from kaggle data using root\n        self.transcript_dir = root + '/' + partition + '/transcript/'\n\n        # TODO: List files in sefl.mfcc_dir using os.listdir in sorted order\n        mfcc_names          = sorted(os.listdir(self.mfcc_dir))\n        # TODO: List files in self.transcript_dir using os.listdir in sorted order\n        transcript_names    = sorted(os.listdir(self.transcript_dir))\n\n        # Making sure that we have the same no. of mfcc and transcripts\n        total_timestamps = 0\n        assert len(mfcc_names) == len(transcript_names)\n        for mfcc in mfcc_names:\n            total_timestamps += len(np.load(self.mfcc_dir + mfcc))\n        self.length = total_timestamps\n        print(total_timestamps)\n\n        print(\"HERE\")\n        self.mfccs, self.transcripts = np.zeros((2*context+total_timestamps, 28), dtype=np.float32), np.zeros((total_timestamps), dtype=np.uint8)\n        #self.mfccs, self.transcripts = [], []\n        # TODO: Iterate through mfccs and transcripts\n        current_index = context\n        for i in range(len(mfcc_names)):\n        #   Load a single mfcc\n            mfcc        = np.load(self.mfcc_dir + mfcc_names[i])\n            mfcc_mean = np.mean(mfcc, axis = 0)\n            mfcc_stddev =  np.std(mfcc, axis = 0)\n            cepstral_norm = (mfcc - mfcc_mean)/mfcc_stddev\n        #   Do Cepstral Normalization of mfcc (explained in writeup)\n        #   Load the corresponding transcript\n            transcript  = np.load(self.transcript_dir + transcript_names[i]) # Remove [SOS] and [EOS] from the transcript\n            transcript = transcript[1: -1]\n            # (Is there an efficient way to do this without traversing through the transcript?)\n            # Note that SOS will always be in the starting and EOS at end, as the name suggests.\n        #   Append each mfcc to self.mfcc, transcript to self.transcript\n            self.mfccs[current_index: current_index + len(cepstral_norm)] = cepstral_norm\n            self.transcripts[current_index - context: current_index + len(transcript) - context] = np.array(list(map(lambda x : self.phonemes.index(x), transcript)))\n            current_index += len(cepstral_norm)\n            #self.mfccs.append(cepstral_norm)\n            #self.transcripts.append(transcript)\n        print(current_index)\n        # NOTE:\n        # Each mfcc is of shape T1 x 28, T2 x 28, ...\n        # Each transcript is of shape (T1+2), (T2+2),... before removing [SOS] and [EOS]\n\n        # TODO: Concatenate all mfccs in self.mfccs such that\n        # the final shape is T x 28 (Where T = T1 + T2 + ...)\n        #self.mfccs          = np.concatenate(self.mfccs, axis = 0)\n\n        # TODO: Concatenate all transcripts in self.transcripts such that\n        # the final shape is (T,) meaning, each time step has one phoneme output\n        #self.transcripts    = np.concatenate(self.transcripts, axis = 0)\n        # Hint: Use numpy to concatenate\n\n        # Length of the dataset is now the length of concatenated mfccs/transcripts\n\n        # Take some time to think about what we have done.\n        # self.mfcc is an array of the format (Frames x Features).\n        # Our goal is to recognize phonemes of each frame\n        # We can introduce context by padding zeros on top and bottom of self.mfcc\n        #self.mfccs = np.concatenate((np.zeros((context, 28)), self.mfccs, np.zeros((context, 28))), axis = 0) # TODO\n\n        # The available phonemes in the transcript are of string data type\n        # But the neural network cannot predict strings as such.\n        # Hence, we map these phonemes to integers\n\n        # TODO: Map the phonemes to their corresponding list indexes in self.phonemes\n        #def phoneme_to_index(p):\n        #  return self.phonemes.index(p)\n        #self.transcripts = np.array(list(map(lambda x : self.phonemes.index(x), self.transcripts)))\n        # Now, if an element in self.transcript is 0, it means that it is 'SIL' (as per the above example)\n\n    def __len__(self):\n        return self.length\n\n    def __getitem__(self, ind):\n\n        # TODO: Based on context and offset, return a frame at given index with context frames to the left, and right.\n        frames = self.mfccs[ind: ind + 2*self.context+1]\n        # After slicing, you get an array of shape 2*context+1 x 28. But our MLP needs 1d data and not 2d.\n        frames = frames.flatten() # TODO: Flatten to get 1d data\n\n        frames      = torch.FloatTensor(frames) # Convert to tensors\n        phonemes    = torch.tensor(self.transcripts[ind])\n\n        return frames, phonemes","metadata":{"id":"YpLCvi3AJC5z","execution":{"iopub.status.busy":"2023-09-20T21:17:55.166995Z","iopub.execute_input":"2023-09-20T21:17:55.167280Z","iopub.status.idle":"2023-09-20T21:17:55.183864Z","shell.execute_reply.started":"2023-09-20T21:17:55.167257Z","shell.execute_reply":"2023-09-20T21:17:55.182627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AudioTestDataset(torch.utils.data.Dataset):\n    def __init__(self, root, phonemes = PHONEMES, context=0, partition= \"train-clean-100\"): # Feel free to add more arguments\n\n        self.context    = context\n        self.phonemes   = phonemes\n        # TODO: MFCC directory - use partition to acces train/dev directories from kaggle data using root\n        self.mfcc_dir       = root + '/' + partition + '/mfcc/'\n\n\n        # TODO: List files in sefl.mfcc_dir using os.listdir in sorted order\n        mfcc_names          = sorted(os.listdir(self.mfcc_dir))\n\n\n        total_timestamps = 0\n        for mfcc in mfcc_names:\n            total_timestamps += len(np.load(self.mfcc_dir + mfcc))\n        self.length = total_timestamps\n        print(total_timestamps)\n        #self.mfccs = []\n        self.mfccs = np.zeros((2*context+total_timestamps, 28), dtype=np.float32)\n\n        # TODO: Iterate through mfccs and transcripts\n        current_index = context\n        for i in range(len(mfcc_names)):\n        #   Load a single mfcc\n            mfcc        = np.load(self.mfcc_dir + mfcc_names[i])\n            mfcc_mean = np.mean(mfcc, axis = 0)\n            mfcc_stddev =  np.std(mfcc, axis = 0)\n            cepstral_norm = (mfcc - mfcc_mean)/mfcc_stddev\n        #   Do Cepstral Normalization of mfcc (explained in writeup)\n        #   Load the corresponding transcript\n\n            #self.mfccs.append(cepstral_norm)\n        #   Append each mfcc to self.mfcc, transcript to self.transcript\n            self.mfccs[current_index: current_index + len(cepstral_norm)] = cepstral_norm\n            current_index += len(cepstral_norm)\n        print(current_index)\n        # NOTE:\n        # Each mfcc is of shape T1 x 28, T2 x 28, ...\n        # Each transcript is of shape (T1+2), (T2+2),... before removing [SOS] and [EOS]\n\n        # TODO: Concatenate all mfccs in self.mfccs such that\n        # the final shape is T x 28 (Where T = T1 + T2 + ...)\n        #self.mfccs          = np.concatenate(self.mfccs, axis = 0)\n\n        # TODO: Concatenate all transcripts in self.transcripts such that\n        # the final shape is (T,) meaning, each time step has one phoneme output\n        #self.transcripts    = np.concatenate(self.transcripts, axis = 0)\n        # Hint: Use numpy to concatenate\n\n        # Length of the dataset is now the length of concatenated mfccs/transcripts\n\n        # Take some time to think about what we have done.\n        # self.mfcc is an array of the format (Frames x Features).\n        # Our goal is to recognize phonemes of each frame\n        # We can introduce context by padding zeros on top and bottom of self.mfcc\n        #self.mfccs = np.concatenate((np.zeros((context, 28)), self.mfccs, np.zeros((context, 28))), axis = 0) # TODO\n\n        # The available phonemes in the transcript are of string data type\n        # But the neural network cannot predict strings as such.\n        # Hence, we map these phonemes to integers\n\n\n    def __len__(self):\n        return self.length\n\n    def __getitem__(self, ind):\n\n        # TODO: Based on context and offset, return a frame at given index with context frames to the left, and right.\n        frames = self.mfccs[ind: ind + 2*self.context+1]\n        # After slicing, you get an array of shape 2*context+1 x 28. But our MLP needs 1d data and not 2d.\n        frames = frames.flatten() # TODO: Flatten to get 1d data\n\n        frames      = torch.FloatTensor(frames) # Convert to tensors\n\n        return frames\n\n    # TODO: Create a test dataset class similar to the previous class but you dont have transcripts for this\n    # Imp: Read the mfccs in sorted order, do NOT shuffle the data here or in your dataloader.","metadata":{"id":"e8KfVP39S6o7","execution":{"iopub.status.busy":"2023-09-20T21:17:55.186226Z","iopub.execute_input":"2023-09-20T21:17:55.187331Z","iopub.status.idle":"2023-09-20T21:17:55.208456Z","shell.execute_reply.started":"2023-09-20T21:17:55.187292Z","shell.execute_reply":"2023-09-20T21:17:55.207472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Parameters Configuration","metadata":{"id":"qNacQ8bpt9nw"}},{"cell_type":"markdown","source":"Storing your parameters and hyperparameters in a single configuration dictionary makes it easier to keep track of them during each experiment. It can also be used with weights and biases to log your parameters for each experiment and keep track of them across multiple experiments.","metadata":{"id":"WE7tsinAuLNy"}},{"cell_type":"code","source":"config = {\n    'epochs'        : 20,\n    'batch_size'    : 1024,\n    'context'       : 35,\n    'init_lr'       : 1e-3,\n    'architecture'  : 'very-low-cutoff'\n    # Add more as you need them - e.g dropout values, weight decay, scheduler parameters\n}","metadata":{"id":"PmKwlFqgt_Zq","execution":{"iopub.status.busy":"2023-09-20T21:17:55.209921Z","iopub.execute_input":"2023-09-20T21:17:55.210738Z","iopub.status.idle":"2023-09-20T21:17:55.222146Z","shell.execute_reply.started":"2023-09-20T21:17:55.210707Z","shell.execute_reply":"2023-09-20T21:17:55.221321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{"id":"IT0E12EFvBV2"}},{"cell_type":"markdown","source":"# Create Datasets","metadata":{"id":"2mlwaKlDt_2c"}},{"cell_type":"code","source":"#TODO: Create a dataset object using the AudioDataset class for the training data\ntrain_data = AudioDataset('/kaggle/input/11785-hw1p2-f23/11-785-f23-hw1p2', context = config['context'])\n\n# TODO: Create a dataset object using the AudioDataset class for the validation data\nval_data = AudioDataset('/kaggle/input/11785-hw1p2-f23/11-785-f23-hw1p2', context = config['context'], partition='dev-clean')\n\n# TODO: Create a dataset object using the AudioTestDataset class for the test data\ntest_data = AudioTestDataset('/kaggle/input/11785-hw1p2-f23/11-785-f23-hw1p2', context = config['context'], partition='test-clean')\n","metadata":{"id":"7xi7V8x8W9z4","outputId":"a36828f6-5ee6-4d78-ec3b-89e8497843fe","execution":{"iopub.status.busy":"2023-09-20T21:17:55.226574Z","iopub.execute_input":"2023-09-20T21:17:55.226825Z","iopub.status.idle":"2023-09-20T21:25:45.105675Z","shell.execute_reply.started":"2023-09-20T21:17:55.226803Z","shell.execute_reply":"2023-09-20T21:25:45.104672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#TODO: Create a dataset object using the AudioDataset class for the training data\n#train_data = AudioDataset('/content/data/11-785-f23-hw1p2', context = config['context'])\n\n# TODO: Create a dataset object using the AudioDataset class for the validation data\n#val_data = AudioDataset('/content/data/11-785-f23-hw1p2', context = config['context'], partition='dev-clean')\n\n# TODO: Create a dataset object using the AudioTestDataset class for the test data\n#test_data = AudioTestDataset('/content/data/11-785-f23-hw1p2', context = config['context'], partition='test-clean')","metadata":{"id":"nzSfwitqvV3S","outputId":"01e09af1-f8ef-4daf-aa98-f16b92522440","execution":{"iopub.status.busy":"2023-09-20T21:25:45.106943Z","iopub.execute_input":"2023-09-20T21:25:45.107309Z","iopub.status.idle":"2023-09-20T21:25:45.112979Z","shell.execute_reply.started":"2023-09-20T21:25:45.107274Z","shell.execute_reply":"2023-09-20T21:25:45.111839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define dataloaders for train, val and test datasets\n# Dataloaders will yield a batch of frames and phonemes of given batch_size at every iteration\n# We shuffle train dataloader but not val & test dataloader. Why?\n\ntrain_loader = torch.utils.data.DataLoader(\n    dataset     = train_data,\n    num_workers = 2,\n    batch_size  = config['batch_size'],\n    pin_memory  = True,\n    shuffle     = True\n)\n\nval_loader = torch.utils.data.DataLoader(\n    dataset     = val_data,\n    num_workers = 2,\n    batch_size  = config['batch_size'],\n    pin_memory  = True,\n    shuffle     = False\n)\n\ntest_loader = torch.utils.data.DataLoader(\n    dataset     = test_data,\n    num_workers = 2,\n    batch_size  = config['batch_size'],\n    pin_memory  = True,\n    shuffle     = False\n)\n\n\nprint(\"Batch size     : \", config['batch_size'])\nprint(\"Context        : \", config['context'])\nprint(\"Input size     : \", (2*config['context']+1)*28)\nprint(\"Output symbols : \", len(PHONEMES))\n\nprint(\"Train dataset samples = {}, batches = {}\".format(train_data.__len__(), len(train_loader)))\nprint(\"Validation dataset samples = {}, batches = {}\".format(val_data.__len__(), len(val_loader)))\nprint(\"Test dataset samples = {}, batches = {}\".format(test_data.__len__(), len(test_loader)))","metadata":{"id":"4mzoYfTKu14s","outputId":"8a69b328-907f-4d11-9fc5-54ca45106a15","execution":{"iopub.status.busy":"2023-09-20T21:25:45.114416Z","iopub.execute_input":"2023-09-20T21:25:45.115419Z","iopub.status.idle":"2023-09-20T21:25:45.127299Z","shell.execute_reply.started":"2023-09-20T21:25:45.115387Z","shell.execute_reply":"2023-09-20T21:25:45.126188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Testing code to check if your data loaders are working\nfor i, data in enumerate(train_loader):\n    frames, phoneme = data\n    print(frames.shape, phoneme.shape)\n    break","metadata":{"id":"n-GV3UvgLSoF","outputId":"08bdb4b7-9bff-47ea-f777-19ad6dfa8609","execution":{"iopub.status.busy":"2023-09-20T21:25:45.128731Z","iopub.execute_input":"2023-09-20T21:25:45.129114Z","iopub.status.idle":"2023-09-20T21:25:53.739915Z","shell.execute_reply.started":"2023-09-20T21:25:45.129077Z","shell.execute_reply":"2023-09-20T21:25:53.738792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Network Architecture\n","metadata":{"id":"Nxjwve20JRJ2"}},{"cell_type":"markdown","source":"This section defines your network architecture for the homework. We have given you a sample architecture that can easily clear the very low cutoff for the early submission deadline.","metadata":{"id":"3NJzT-mRw6iy"}},{"cell_type":"markdown","source":"# This architecture will make you cross the very low cutoff\n# However, you need to run a lot of experiments to cross the medium or high cutoff\n","metadata":{"id":"OvcpontXQq9j"}},{"cell_type":"code","source":"# This architecture will make you cross the very low cutoff\n# However, you need to run a lot of experiments to cross the medium or high cutoff\nclass Network(torch.nn.Module):\n\n    def __init__(self, input_size, output_size):\n\n        super(Network, self).__init__()\n\n        self.model = torch.nn.Sequential(\n            \n            torch.nn.Linear(input_size, 2048),\n            torch.nn.BatchNorm1d(2048),\n            torch.nn.GELU(),\n            torch.nn.Dropout(p=0.25),\n            \n            torch.nn.Linear(2048, 2048),\n            torch.nn.GELU(),\n            torch.nn.Dropout(p=0.25),\n            \n            torch.nn.Linear(2048, 2048),\n            torch.nn.BatchNorm1d(2048),\n            torch.nn.GELU(),\n            torch.nn.Dropout(p=0.25),\n            \n            torch.nn.Linear(2048, 2048),\n            torch.nn.GELU(),\n            torch.nn.Dropout(p=0.25),\n            \n            torch.nn.Linear(2048, 2048),\n            torch.nn.BatchNorm1d(2048),\n            torch.nn.GELU(),\n            torch.nn.Dropout(p=0.25),\n            \n            torch.nn.Linear(2048, 2048),\n            torch.nn.GELU(),\n            torch.nn.Dropout(p=0.25),\n            \n            torch.nn.Linear(2048, output_size),\n        )\n\n    def forward(self, x):\n        out = self.model(x)\n\n        return out","metadata":{"execution":{"iopub.status.busy":"2023-09-20T21:25:53.741838Z","iopub.execute_input":"2023-09-20T21:25:53.742489Z","iopub.status.idle":"2023-09-20T21:25:53.754007Z","shell.execute_reply.started":"2023-09-20T21:25:53.742448Z","shell.execute_reply":"2023-09-20T21:25:53.753143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define Model, Loss Function and Optimizer","metadata":{"id":"HejoSXe3vMVU"}},{"cell_type":"markdown","source":"Here we define the model, loss function, optimizer and optionally a learning rate scheduler.","metadata":{"id":"xAhGBH7-xxth"}},{"cell_type":"code","source":"INPUT_SIZE  = (2*config['context'] + 1) * 28 # Why is this the case?\nmodel       = Network(INPUT_SIZE, len(train_data.phonemes)).to(device)\n#summary(model, frames.to(device))\n# Check number of parameters of your network\n# Remember, you are limited to 25 million parameters for HW1 (including ensembles)","metadata":{"id":"_qtrEM1ZvLje","execution":{"iopub.status.busy":"2023-09-20T21:25:53.755190Z","iopub.execute_input":"2023-09-20T21:25:53.755620Z","iopub.status.idle":"2023-09-20T21:25:54.664677Z","shell.execute_reply.started":"2023-09-20T21:25:53.755587Z","shell.execute_reply":"2023-09-20T21:25:54.663678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"zVPT7fSVvBV6"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"criterion = torch.nn.CrossEntropyLoss() # Defining Loss function.\n# We use CE because the task is multi-class classification\n\noptimizer = torch.optim.AdamW(model.parameters(), lr= config['init_lr']) #Defining Optimizer\n\n# Create a StepLR scheduler with a step size of 2 and a gamma of 0.5\n#scheduler = StepLR(optimizer, step_size=2, gamma=0.5)\nscheduler = ReduceLROnPlateau(optimizer, factor = 0.1, patience = 1)\n# Recommended : Define Scheduler for Learning Rate,\n# including but not limited to StepLR, MultiStepLR, CosineAnnealingLR, ReduceLROnPlateau, etc.\n# You can refer to Pytorch documentation for more information on how to use them.\n\n# Is your training time very high?\n# Look into mixed precision training if your GPU (Tesla T4, V100, etc) can make use of it\n# Refer - https://pytorch.org/docs/stable/notes/amp_examples.html","metadata":{"id":"UROGEVJevKD-","execution":{"iopub.status.busy":"2023-09-20T21:25:54.666089Z","iopub.execute_input":"2023-09-20T21:25:54.666545Z","iopub.status.idle":"2023-09-20T21:25:54.673255Z","shell.execute_reply.started":"2023-09-20T21:25:54.666507Z","shell.execute_reply":"2023-09-20T21:25:54.672194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training and Validation Functions","metadata":{"id":"IBwunYpyugFg"}},{"cell_type":"markdown","source":"This section covers the training, and validation functions for each epoch of running your experiment with a given model architecture. The code has been provided to you, but we recommend going through the comments to understand the workflow to enable you to write these loops for future HWs.","metadata":{"id":"1JgeNhx4x2-P"}},{"cell_type":"code","source":"torch.cuda.empty_cache()\ngc.collect()","metadata":{"id":"XblOHEVtKab2","outputId":"a412c456-9c5d-4e55-c8ea-368820b989d6","execution":{"iopub.status.busy":"2023-09-20T21:25:54.674837Z","iopub.execute_input":"2023-09-20T21:25:54.675398Z","iopub.status.idle":"2023-09-20T21:25:54.834423Z","shell.execute_reply.started":"2023-09-20T21:25:54.675354Z","shell.execute_reply":"2023-09-20T21:25:54.833519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train(model, dataloader, optimizer, criterion):\n\n    model.train()\n    tloss, tacc = 0, 0 # Monitoring loss and accuracy\n    batch_bar   = tqdm(total=len(train_loader), dynamic_ncols=True, leave=False, position=0, desc='Train')\n\n    for i, (frames, phonemes) in enumerate(dataloader):\n\n        ### Initialize Gradients\n        optimizer.zero_grad()\n\n        ### Move Data to Device (Ideally GPU)\n        frames      = frames.to(device)\n        phonemes    = phonemes.to(device)\n\n        ### Forward Propagation\n        logits  = model(frames)\n\n        ### Loss Calculation\n        loss    = criterion(logits, phonemes)\n\n        ### Backward Propagation\n        loss.backward()\n\n        ### Gradient Descent\n        optimizer.step()\n\n        tloss   += loss.item()\n        tacc    += torch.sum(torch.argmax(logits, dim= 1) == phonemes).item()/logits.shape[0]\n\n        batch_bar.set_postfix(loss=\"{:.04f}\".format(float(tloss / (i + 1))),\n                              acc=\"{:.04f}%\".format(float(tacc*100 / (i + 1))))\n        batch_bar.update()\n\n        ### Release memory\n        del frames, phonemes, logits\n        torch.cuda.empty_cache()\n\n    batch_bar.close()\n    tloss   /= len(train_loader)\n    tacc    /= len(train_loader)\n\n    return tloss, tacc","metadata":{"id":"8wjPz7DHqKcL","execution":{"iopub.status.busy":"2023-09-20T21:25:54.835978Z","iopub.execute_input":"2023-09-20T21:25:54.836394Z","iopub.status.idle":"2023-09-20T21:25:54.848856Z","shell.execute_reply.started":"2023-09-20T21:25:54.836362Z","shell.execute_reply":"2023-09-20T21:25:54.847966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def eval(model, dataloader):\n\n    model.eval() # set model in evaluation mode\n    vloss, vacc = 0, 0 # Monitoring loss and accuracy\n    batch_bar   = tqdm(total=len(val_loader), dynamic_ncols=True, position=0, leave=False, desc='Val')\n\n    for i, (frames, phonemes) in enumerate(dataloader):\n\n        ### Move data to device (ideally GPU)\n        frames      = frames.to(device)\n        phonemes    = phonemes.to(device)\n\n        # makes sure that there are no gradients computed as we are not training the model now\n        with torch.inference_mode():\n            ### Forward Propagation\n            logits  = model(frames)\n            ### Loss Calculation\n            loss    = criterion(logits, phonemes)\n\n        vloss   += loss.item()\n        vacc    += torch.sum(torch.argmax(logits, dim= 1) == phonemes).item()/logits.shape[0]\n\n        # Do you think we need loss.backward() and optimizer.step() here?\n\n        batch_bar.set_postfix(loss=\"{:.04f}\".format(float(vloss / (i + 1))),\n                              acc=\"{:.04f}%\".format(float(vacc*100 / (i + 1))))\n        batch_bar.update()\n\n        ### Release memory\n        del frames, phonemes, logits\n        torch.cuda.empty_cache()\n\n    batch_bar.close()\n    vloss   /= len(val_loader)\n    vacc    /= len(val_loader)\n\n    return vloss, vacc","metadata":{"id":"Q5npQNFH315V","execution":{"iopub.status.busy":"2023-09-20T21:25:54.850188Z","iopub.execute_input":"2023-09-20T21:25:54.850656Z","iopub.status.idle":"2023-09-20T21:25:54.863114Z","shell.execute_reply.started":"2023-09-20T21:25:54.850624Z","shell.execute_reply":"2023-09-20T21:25:54.862255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Weights and Biases Setup","metadata":{"id":"yMd_XxPku5qp"}},{"cell_type":"markdown","source":"This section is to enable logging metrics and files with Weights and Biases. Please refer to wandb documentationa and recitation 0 that covers the use of weights and biases for logging, hyperparameter tuning and monitoring your runs for your homeworks. Using this tool makes it very easy to show results when submitting your code and models for homeworks, and also extremely useful for study groups to organize and run ablations under a single team in wandb.\n\nWe have written code for you to make use of it out of the box, so that you start using wandb for all your HWs from the beginning.","metadata":{"id":"tjIbhR1wwbgI"}},{"cell_type":"markdown","source":"# Experiment","metadata":{"id":"nclx_04fu7Dd"}},{"cell_type":"markdown","source":"Now, it is time to finally run your ablations! Have fun!","metadata":{"id":"MdLMWfEpyGOB"}},{"cell_type":"code","source":"# Iterate over number of epochs to train and evaluate your model\ntorch.cuda.empty_cache()\ngc.collect()\n\nfor epoch in range(config['epochs']):\n\n    print(\"\\nEpoch {}/{}\".format(epoch+1, config['epochs']))\n\n    curr_lr                 = float(optimizer.param_groups[0]['lr'])\n    train_loss, train_acc   = train(model, train_loader, optimizer, criterion)\n    val_loss, val_acc       = eval(model, val_loader)\n    scheduler.step(val_loss)\n    print(\"\\tTrain Acc {:.04f}%\\tTrain Loss {:.04f}\\t Learning Rate {:.07f}\".format(train_acc*100, train_loss, curr_lr))\n    print(\"\\tVal Acc {:.04f}%\\tVal Loss {:.04f}\".format(val_acc*100, val_loss))\n\n    ### Log metrics at each epoch in your run\n    # Optionally, you can log at each batch inside train/eval functions\n    # (explore wandb documentation/wandb recitation)\n    ### Highly Recommended: Save checkpoint in drive and/or wandb if accuracy is better than your current best\n\n### Finish your wandb run\n#run.finish()","metadata":{"id":"MG4F77Nm0Am9","outputId":"f5b15ca3-a963-4532-8223-731c96d9c8ce","execution":{"iopub.status.busy":"2023-09-20T21:25:54.864598Z","iopub.execute_input":"2023-09-20T21:25:54.865029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Testing and submission to Kaggle","metadata":{"id":"_kXwf5YUo_4A"}},{"cell_type":"markdown","source":"Before we get to the following code, make sure to see the format of submission given in *sample_submission.csv*. Once you have done so, it is time to fill the following function to complete your inference on test data. Refer the eval function from previous cells to get an idea of how to go about completing this function.","metadata":{"id":"WI1hSFYLpJvH"}},{"cell_type":"code","source":"def test(model, test_loader):\n    ### What you call for model to perform inference?\n    model.eval() # TODO train or eval?\n\n    ### List to store predicted phonemes of test data\n    test_predictions = []\n\n    ### Which mode do you need to avoid gradients?\n    with torch.inference_mode(): # TODO\n\n        for i, mfccs in enumerate(tqdm(test_loader)):\n\n            mfccs   = mfccs.to(device)\n\n            logits  = model(mfccs)\n\n            ### Get most likely predicted phoneme with argmax\n            predicted_phonemes = [PHONEMES[i] for i in torch.argmax(logits, dim=1)]\n\n            ### How do you store predicted_phonemes with test_predictions? Hint, look at eval\n            test_predictions.extend(predicted_phonemes)\n            # TODO\n\n    return test_predictions","metadata":{"id":"R-SU9fZ3xHtk","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = test(model, test_loader)","metadata":{"id":"wG9v6Xmxu7wp","outputId":"edbd51c5-abd9-4d41-d9e0-3520f113eb71","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Create CSV file with predictions\nwith open(\"./submission.csv\", \"w+\") as f:\n    f.write(\"id,label\\n\")\n    for i in range(len(predictions)):\n        f.write(\"{},{}\\n\".format(i, predictions[i]))","metadata":{"id":"ZE1hRnvf0bFz","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Submit to kaggle competition using kaggle API (Uncomment below to use)\n!kaggle competitions submit -c 11785-hw1p2-f23 -f ./submission.csv -m \"Test Submission\"\n\n### However, its always safer to download the csv file and then upload to kaggle","metadata":{"id":"LjcammuCxMKN","outputId":"ccd7f69a-b6ca-4369-a661-0ecf0d8ee860","trusted":true},"execution_count":null,"outputs":[]}]}