{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392733,"sourceType":"datasetVersion","datasetId":4297749},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782},{"sourceId":7447509,"sourceType":"datasetVersion","datasetId":4334995},{"sourceId":165784130,"sourceType":"kernelVersion"}],"dockerImageVersionId":30699,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nimport pandas as pd, numpy as np, os\nimport matplotlib.pyplot as plt, gc\n\nfrom sklearn.model_selection import train_test_split\n\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom IPython.display import Image\n!pip install torchview\nfrom torchview import draw_graph","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:31:02.239956Z","iopub.execute_input":"2024-04-23T05:31:02.240419Z","iopub.status.idle":"2024-04-23T05:31:23.409431Z","shell.execute_reply.started":"2024-04-23T05:31:02.240393Z","shell.execute_reply":"2024-04-23T05:31:23.408335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nTARGETS = df.columns[-6:]\nprint('Train shape:', df.shape )\nprint('Targets', list(TARGETS))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:31:23.411702Z","iopub.execute_input":"2024-04-23T05:31:23.412333Z","iopub.status.idle":"2024-04-23T05:31:23.692188Z","shell.execute_reply.started":"2024-04-23T05:31:23.412301Z","shell.execute_reply":"2024-04-23T05:31:23.691170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spec_id','min']\n\ntmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max'] = tmp\n\ntmp = df.groupby('eeg_id')[['patient_id']].agg('first')\ntrain['patient_id'] = tmp\n\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n    \ny_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['target'] = tmp\n\ntrain = train.reset_index()\nprint('Train non-overlapp eeg_id shape:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:31:23.693443Z","iopub.execute_input":"2024-04-23T05:31:23.693801Z","iopub.status.idle":"2024-04-23T05:31:23.797684Z","shell.execute_reply.started":"2024-04-23T05:31:23.693771Z","shell.execute_reply":"2024-04-23T05:31:23.796630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spectrograms = np.load('/kaggle/input/brain-spectrograms/specs.npy',allow_pickle=True).item()\n# all_eegs = np.load('/kaggle/input/brain-eeg-spectrograms/eeg_specs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:31:23.800594Z","iopub.execute_input":"2024-04-23T05:31:23.800993Z","iopub.status.idle":"2024-04-23T05:32:16.627000Z","shell.execute_reply.started":"2024-04-23T05:31:23.800957Z","shell.execute_reply":"2024-04-23T05:32:16.626162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(spectrograms.keys())","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:32:16.628171Z","iopub.execute_input":"2024-04-23T05:32:16.628467Z","iopub.status.idle":"2024-04-23T05:32:16.634490Z","shell.execute_reply.started":"2024-04-23T05:32:16.628441Z","shell.execute_reply":"2024-04-23T05:32:16.633542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\nTARS2 = {x:y for y,x in TARS.items()}","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:32:16.635810Z","iopub.execute_input":"2024-04-23T05:32:16.636281Z","iopub.status.idle":"2024-04-23T05:32:16.644352Z","shell.execute_reply.started":"2024-04-23T05:32:16.636247Z","shell.execute_reply":"2024-04-23T05:32:16.643590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Stacking channels vertically","metadata":{}},{"cell_type":"code","source":"import torch\nimport numpy as np\nfrom torch.utils.data import Dataset\n\nclass StackedSpectrogramDataset(Dataset):\n    def __init__(self, data, specs, mode='train'):\n        self.data = data\n        self.specs = specs\n        self.mode = mode\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        row = self.data.iloc[idx]\n        \n        # Initialize the array that will store the vertically stacked images\n        X = np.zeros((1, 512, 256), dtype='float32')  # Single sample, 1 channel, 512 height, 256 width\n\n        # Determine the row index for slicing if not in test mode\n        r = int((row['min'] + row['max']) // 4) if self.mode != 'test' else 0\n\n        for k in range(4):\n            # Extract spectrogram segment\n            img = self.specs[row.spec_id][r:r+300, k*100:(k+1)*100].T\n            img = np.clip(img, np.exp(-4), np.exp(8))\n            img = np.log(img)\n\n            # Standardize per image\n            ep = 1e-6\n            m = np.nanmean(img.flatten())\n            s = np.nanstd(img.flatten())\n            img = (img - m) / (s + ep)\n            img = np.nan_to_num(img, nan=0.0)\n\n            # Assign each processed channel to the corresponding vertical slot\n            X[0, k*128+14:(k+1)*128-14, :] = img[:, 22:-22] / 2.0  # Adjusting the slicing based on cropping\n\n        # Convert X to a tensor\n        X_tensor = torch.from_numpy(X).float()\n\n        # Get labels, assuming they're stored in a column named 'targets'\n        y = torch.tensor(row[TARGETS], dtype=torch.float32) if self.mode != 'test' else torch.tensor([])\n\n        return X_tensor, y\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:32:16.645441Z","iopub.execute_input":"2024-04-23T05:32:16.645733Z","iopub.status.idle":"2024-04-23T05:32:16.657019Z","shell.execute_reply.started":"2024-04-23T05:32:16.645709Z","shell.execute_reply":"2024-04-23T05:32:16.656267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stacked_train = StackedSpectrogramDataset(train, spectrograms, mode='train')\ntrain_loader = DataLoader(stacked_train, batch_size=32, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:32:16.657958Z","iopub.execute_input":"2024-04-23T05:32:16.658222Z","iopub.status.idle":"2024-04-23T05:32:16.669020Z","shell.execute_reply.started":"2024-04-23T05:32:16.658185Z","shell.execute_reply":"2024-04-23T05:32:16.668133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(stacked_train[0]))","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:32:16.670194Z","iopub.execute_input":"2024-04-23T05:32:16.670948Z","iopub.status.idle":"2024-04-23T05:32:16.704791Z","shell.execute_reply.started":"2024-04-23T05:32:16.670916Z","shell.execute_reply":"2024-04-23T05:32:16.703907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROWS = 2\nCOLS = 3\nBATCHES = 2\nfor i, (X_batch, y_batch) in enumerate(train_loader):\n    plt.figure(figsize=(20, 20))\n    for j in range(ROWS):\n        for k in range(COLS):\n            plt.subplot(ROWS, COLS, j * COLS + k + 1)\n            \n            # Get the image and labels\n            img = X_batch[j * COLS + k].numpy()  # Convert the tensor to numpy\n            img = img[0]  # Selecting the first channel (assuming it's similar to the Keras version)\n            img = img[::-1, :]  # Flipping the image vertically\n            \n            # Normalize the image for better visualization\n            mn = img.min()\n            mx = img.max()\n            img = (img - mn) / (mx - mn)\n            \n            # Get the corresponding target labels\n            t = y_batch[j * COLS + k].numpy()\n            tars = f'[{t[0]:0.2f}'\n            for s in t[1:]:\n                tars += f', {s:0.2f}'\n            tars += ']'\n            \n            # Get EEG ID if available or set a placeholder\n            eeg = train.eeg_id.values[i*32+j*COLS+k]  # Replace 'Placeholder' with actual EEG ID retrieval logic if needed\n            \n            # Plotting the image\n            plt.imshow(img, cmap='viridis')  # viridis colormap to match EEG style\n            plt.title(f'EEG = {eeg}\\nTarget = {tars}', size=12)\n            plt.yticks([])\n            plt.ylabel('Frequencies (Hz)', size=14)\n            plt.xlabel('Time (sec)', size=16)\n\n    plt.show()\n    if i == BATCHES - 1:\n        break\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:32:16.707004Z","iopub.execute_input":"2024-04-23T05:32:16.707258Z","iopub.status.idle":"2024-04-23T05:32:20.366038Z","shell.execute_reply.started":"2024-04-23T05:32:16.707236Z","shell.execute_reply":"2024-04-23T05:32:20.364517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Baseline CNN","metadata":{}},{"cell_type":"code","source":"class Net(nn.Module):\n    def __init__(self):\n        super(Net, self).__init__()\n        self.conv1 = nn.Conv2d(1, 64, kernel_size=5, padding='same')  # set the size of the convolution to 5x5, and have 12 of them\n        self.conv2 = nn.Conv2d(64, 128, kernel_size=3, padding='same')\n        self.fc1 = nn.Linear(1048576, 50)\n        self.fc2 = nn.Linear(50, 6)\n        \n        self.global_avg_pool = nn.AdaptiveAvgPool2d((1, 1))\n    \n    def forward(self, x):\n        x = F.relu(F.max_pool2d(self.conv1(x), 2))  # Convolution then pooling\n        #print(\"Shape after conv1 and pool:\", x.shape)\n        x = F.relu(F.max_pool2d(self.conv2(x), 2))  # Second convolution and pooling\n        #print(\"Shape after conv2 and pool:\", x.shape)\n        \n        #x = self.global_avg_pool(x)\n        #print(\"Shape after Pool:\", x.shape)\n        \n        x = x.view(-1, x.shape[1] * x.shape[2] * x.shape[3])\n        #print(\"Shape before entering fc1:\", x.shape)\n\n        x = F.relu(self.fc1(x))\n        x = self.fc2(x)\n        return F.log_softmax(x, dim=1)\n\n    \nnetwork = Net()\n\nimport torch.optim as optim\n\n# create the learning rule\n#optimizer = optim.SGD(network.parameters(),\n #                     lr=0.1,   # learning rate\n #                     momentum=0.5)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-22T23:56:40.690478Z","iopub.execute_input":"2024-04-22T23:56:40.690804Z","iopub.status.idle":"2024-04-22T23:56:41.242204Z","shell.execute_reply.started":"2024-04-22T23:56:40.690777Z","shell.execute_reply":"2024-04-22T23:56:41.241356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Net with fewer params since it takes too long to train","metadata":{}},{"cell_type":"code","source":"class Net(nn.Module):\n    def __init__(self):\n        super(Net, self).__init__()\n        self.conv1 = nn.Conv2d(1, 32, kernel_size=7, padding='same')  # set the size of the convolution to 5x5, and have 12 of them\n        self.conv2 = nn.Conv2d(32, 64, kernel_size=5, padding='same')\n        self.conv3 = nn.Conv2d(64, 128, kernel_size=3, padding='same')\n        self.conv4 = nn.Conv2d(128, 256, kernel_size=3, padding='same')\n        self.fc1 = nn.Linear(256, 50)\n        self.fc2 = nn.Linear(50, 6)\n        \n        self.global_avg_pool = nn.AdaptiveAvgPool2d((1, 1))\n        \n        self.bn1 = nn.BatchNorm2d(32)\n        self.bn2 = nn.BatchNorm2d(64)\n        self.bn3 = nn.BatchNorm2d(128)\n        self.bn4 nn.BatchNorm2d(256)\n    \n    def forward(self, x):\n        x = F.relu(self.bn1(F.max_pool2d(self.conv1(x), 2)))\n        x = F.relu(self.bn2(F.max_pool2d(self.conv2(x), 2)))\n        x = self.global_avg_pool(x)\n        x = x.view(x.size(0), -1)\n        x = F.relu(self.fc1(x))\n        x = self.fc2(x)\n        return F.log_softmax(x, dim=1)\n\n    \nnetwork = Net()\n\nimport torch.optim as optim\n\n# create the learning rule\n#optimizer = optim.SGD(network.parameters(),\n #                     lr=0.1,   # learning rate\n #                     momentum=0.5)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-22T23:56:41.243472Z","iopub.execute_input":"2024-04-22T23:56:41.243770Z","iopub.status.idle":"2024-04-22T23:56:41.280081Z","shell.execute_reply.started":"2024-04-22T23:56:41.243744Z","shell.execute_reply":"2024-04-22T23:56:41.279206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"more params to match densenet","metadata":{}},{"cell_type":"code","source":"class Net(nn.Module):\n    def __init__(self):\n        super(Net, self).__init__()\n        self.conv1 = nn.Conv2d(1, 32, kernel_size=7, padding='same')  # set the size of the convolution to 5x5, and have 12 of them\n        self.conv2 = nn.Conv2d(32, 64, kernel_size=5, padding='same')\n        self.conv3 = nn.Conv2d(64, 128, kernel_size=3, padding='same')\n        self.conv4 = nn.Conv2d(128, 256, kernel_size=3, padding='same')\n        self.fc1 = nn.Linear(256, 50)\n        self.fc2 = nn.Linear(50, 6)\n        \n        self.global_avg_pool = nn.AdaptiveAvgPool2d((1, 1))\n        \n        self.bn1 = nn.BatchNorm2d(32)\n        self.bn2 = nn.BatchNorm2d(64)\n        self.bn3 = nn.BatchNorm2d(128)\n        self.bn4 = nn.BatchNorm2d(256)\n    \n    def forward(self, x):\n        x = F.relu(self.bn1(F.max_pool2d(self.conv1(x), 2)))\n        x = F.relu(self.bn2(F.max_pool2d(self.conv2(x), 2)))\n        x = F.relu(self.bn3(F.max_pool2d(self.conv3(x), 2)))\n        x = F.relu(self.bn4(F.max_pool2d(self.conv4(x), 2)))\n        x = self.global_avg_pool(x)\n        x = x.view(x.size(0), -1)\n        x = F.relu(self.fc1(x))\n        x = self.fc2(x)\n        return x\n\n    \nnetwork = Net()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T01:34:46.179310Z","iopub.execute_input":"2024-04-23T01:34:46.180200Z","iopub.status.idle":"2024-04-23T01:34:46.198209Z","shell.execute_reply.started":"2024-04-23T01:34:46.180168Z","shell.execute_reply":"2024-04-23T01:34:46.197055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def test_Net():\n    x = torch.randn(1,1,512,256)\n    model = Net()\n    # print(model(x).shape)\n    # print(model)\n    # del model\n    return model","metadata":{"execution":{"iopub.status.busy":"2024-04-23T01:34:47.023176Z","iopub.execute_input":"2024-04-23T01:34:47.024072Z","iopub.status.idle":"2024-04-23T01:34:47.028785Z","shell.execute_reply.started":"2024-04-23T01:34:47.024038Z","shell.execute_reply":"2024-04-23T01:34:47.027812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = test_Net()\n\narchitecture = 'Net'\nmodel_graph = draw_graph(model, input_size=(1,1,512,256), graph_dir ='TB' , roll=True, expand_nested=True, graph_name=f'self_{architecture}',save_graph=True,filename=f'self_{architecture}')\nmodel_graph.visual_graph","metadata":{"execution":{"iopub.status.busy":"2024-04-22T23:56:41.289857Z","iopub.execute_input":"2024-04-22T23:56:41.290494Z","iopub.status.idle":"2024-04-22T23:56:42.476562Z","shell.execute_reply.started":"2024-04-22T23:56:41.290454Z","shell.execute_reply":"2024-04-22T23:56:42.475609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def continue_training():\n    network.train()      # configure the network for training\n    for i in range(10):  # train the network 10 times\n        correct = 0\n        for data, target in train_loader:       # working in batchs of 1000\n            optimizer.zero_grad()               # initialize the learning system\n            output = network(data)              # feed in the data\n            #print(\"Output shape:\", output.shape)\n            #print(\"Target shape:\", target.shape)\n            print(f'First pred: {output[0]} ---- First target: {target[0]}')\n            loss = F.kl_div(output, target)   # compute how wrong the output is\n            loss.backward()                     # change the weights to reduce error\n            optimizer.step()                    # update the learning rule\n            \n            print(f'Loss: {loss}')\n            #pred = output.data.max(1, keepdim=True)[1]           # compute which output is largest\n            #correct += pred.eq(target.data.view_as(pred)).sum()  # compute the number of correct outputs\n    # update the list of training accuracy values\n    score = float(correct/len(train_loader.dataset))\n    accuracy_train.append(score)\n    print('Iteration', len(accuracy_train), 'Training accuracy:', score)\n\n    correct = 0\n    network.eval()\n    for data, target in test_loader:    # go through the test data once (in groups of 1000)\n        output = network(data)                               # feed in the data\n        pred = output.data.max(1, keepdim=True)[1]           # compute which output is largest\n        correct += pred.eq(target.data.view_as(pred)).sum()  # compute the number of correct outputs\n    # update the list of testing accuracy values\n    score = float(correct/len(test_loader.dataset))\n    accuracy_test.append(score)\n    print('Iteration', len(accuracy_test), 'Testing accuracy:', score)","metadata":{"execution":{"iopub.status.busy":"2024-04-22T23:56:42.478083Z","iopub.execute_input":"2024-04-22T23:56:42.478445Z","iopub.status.idle":"2024-04-22T23:56:42.488204Z","shell.execute_reply.started":"2024-04-22T23:56:42.478413Z","shell.execute_reply":"2024-04-22T23:56:42.487140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model(network, train_loader, val_loader, optimizer, num_epochs=10):\n    # Lists to store losses for plotting\n    epoch_train_losses = []\n    epoch_val_losses = []\n    \n    for epoch in range(num_epochs):\n        network.train()\n        train_loss = 0\n        for data, target in train_loader:\n            data, target = data.to(device), target.to(device) # ------------------------------------------------- TO JULIET KERN: THIS IS TO LOAD DATA TO GPU\n            optimizer.zero_grad()\n            output = network(data)\n            \n            output_log_prob = F.log_softmax(output, dim=1)\n            target_prob = target \n            #print(f'output probs: {output[0]} ------- target probs: {target[0]}')\n            loss = F.kl_div(output_log_prob, target_prob, reduction='batchmean')\n            loss.backward()\n            optimizer.step()\n            train_loss += loss.item()\n\n        avg_train_loss = train_loss / len(train_loader)\n        epoch_train_losses.append(avg_train_loss)\n        print(f'Epoch {epoch+1}/{num_epochs}, Train Loss: {avg_train_loss:.4f}')\n\n        # Validation\n        network.eval()\n        val_loss = 0\n        with torch.no_grad():\n            for data, target in val_loader:\n                data, target = data.to(device), target.to(device) # -------------------------------------------------TO JULIET KERN: NEED THIS LINE TOO FOR VALIDATION DATA TO GPU!\n                output = network(data)\n                output_log_prob = F.log_softmax(output, dim=1)\n                target_prob = target\n                loss = F.kl_div(output_log_prob, target_prob, reduction='batchmean')\n                val_loss += loss.item()\n\n        avg_val_loss = val_loss / len(val_loader)\n        epoch_val_losses.append(avg_val_loss)\n        print(f'Epoch {epoch+1}/{num_epochs}, Validation Loss: {avg_val_loss:.4f}')\n\n    plt.figure(figsize=(10, 5))\n    plt.plot(epoch_train_losses, label='Training Loss')\n    plt.plot(epoch_val_losses, label='Validation Loss')\n    plt.title('Training and Validation Losses Over Epochs')\n    plt.xlabel('Epochs')\n    plt.ylabel('KL Divergence Loss')\n    plt.legend()\n    plt.grid(True)\n    plt.show()\n\n    return epoch_train_losses, epoch_val_losses","metadata":{"execution":{"iopub.status.busy":"2024-04-23T02:11:48.290690Z","iopub.execute_input":"2024-04-23T02:11:48.291504Z","iopub.status.idle":"2024-04-23T02:11:48.304824Z","shell.execute_reply.started":"2024-04-23T02:11:48.291473Z","shell.execute_reply":"2024-04-23T02:11:48.303817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import DataLoader, random_split\n\nfrom torch.utils.data import random_split\n\n# Define the proportions for each set\ntrain_ratio = 0.7\nval_ratio = 0.15\ntest_ratio = 0.15\n\n# Ensure the ratios sum to 1\nassert train_ratio + val_ratio + test_ratio == 1, \"The ratios must sum to 1!\"\n\ntotal_size = len(stacked_train)\ntrain_size = int(train_ratio * total_size)\nval_size = int(val_ratio * total_size)\n# Ensure that the sum of the splits equals the total size\n# This is necessary because using int can sometimes lead to off-by-one errors due to rounding\ntest_size = total_size - train_size - val_size\n\n# Split the dataset\ntrain_dataset, val_dataset, test_dataset = random_split(stacked_train, [train_size, val_size, test_size])\n\n# Create the data loaders for each set\ntrain_loader = DataLoader(train_dataset, batch_size=32, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=32, shuffle=False)\ntest_loader = DataLoader(test_dataset, batch_size=32, shuffle=False)\n\n\nnetwork = Net()\noptimizer = torch.optim.Adam(network.parameters(), lr=0.001)\n\nprint(torch.cuda.is_available()) # -------------------------------------------------------------------------TO JULIET KERN:  THESE LINES OF CODE BELOW ARE TO SET DEVICE TO GPU, and NETWORK TO GPU\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nnetwork = network.to(device)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T02:11:53.110089Z","iopub.execute_input":"2024-04-23T02:11:53.110447Z","iopub.status.idle":"2024-04-23T02:11:53.129392Z","shell.execute_reply.started":"2024-04-23T02:11:53.110419Z","shell.execute_reply":"2024-04-23T02:11:53.128528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_losses, val_losses = train_model(network, train_loader, val_loader, optimizer)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T01:35:15.067486Z","iopub.execute_input":"2024-04-23T01:35:15.068309Z","iopub.status.idle":"2024-04-23T01:55:24.276632Z","shell.execute_reply.started":"2024-04-23T01:35:15.068276Z","shell.execute_reply":"2024-04-23T01:55:24.275606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"adding weight decay (l2 norm)","metadata":{}},{"cell_type":"code","source":"network = Net()\noptimizer = torch.optim.Adam(network.parameters(), lr=0.001, weight_decay=1e-4)\nprint(torch.cuda.is_available()) # -------------------------------------------------------------------------TO JULIET KERN:  THESE LINES OF CODE BELOW ARE TO SET DEVICE TO GPU, and NETWORK TO GPU\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nnetwork = network.to(device)\n\ntrain_losses, val_losses = train_model(network, train_loader, val_loader, optimizer)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T02:42:14.302643Z","iopub.execute_input":"2024-04-23T02:42:14.303044Z","iopub.status.idle":"2024-04-23T03:02:58.402891Z","shell.execute_reply.started":"2024-04-23T02:42:14.303013Z","shell.execute_reply":"2024-04-23T03:02:58.402031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"network = Net()\noptimizer = torch.optim.Adam(network.parameters(), lr=0.001, weight_decay=1e-3)\nprint(torch.cuda.is_available()) # -------------------------------------------------------------------------TO JULIET KERN:  THESE LINES OF CODE BELOW ARE TO SET DEVICE TO GPU, and NETWORK TO GPU\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nnetwork = network.to(device)\n\ntrain_losses, val_losses = train_model(network, train_loader, val_loader, optimizer)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:03:47.222437Z","iopub.execute_input":"2024-04-23T03:03:47.223140Z","iopub.status.idle":"2024-04-23T03:24:05.027175Z","shell.execute_reply.started":"2024-04-23T03:03:47.223108Z","shell.execute_reply":"2024-04-23T03:24:05.026204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Lateral Connections (DenseNet) ","metadata":{}},{"cell_type":"code","source":"# growth rate\nbn_size = 4  # bn stands for bottle neck.... not batch norm i guess\nk = 32  # k = growth rate  (increases # of feature maps across dense blocks)\ncompression_factor = 0.5","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:39:23.252696Z","iopub.execute_input":"2024-04-23T03:39:23.253714Z","iopub.status.idle":"2024-04-23T03:39:23.258030Z","shell.execute_reply.started":"2024-04-23T03:39:23.253679Z","shell.execute_reply":"2024-04-23T03:39:23.256983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DenseLayer(nn.Module):\n    def __init__(self,in_channels):\n        super(DenseLayer,self).__init__()\n        self.norm1 = nn.BatchNorm2d(num_features = in_channels)\n        self.conv1 = nn.Conv2d( in_channels=in_channels , out_channels=4*k , kernel_size=1 , stride=1 , padding=0 , bias = False )\n\n        self.norm2 = nn.BatchNorm2d(num_features = 4*k)\n        self.conv2 = nn.Conv2d( in_channels=4*k , out_channels=k , kernel_size=3 , stride=1 , padding=1 , bias = False )\n\n        self.relu = nn.ReLU(inplace=True)\n\n    def forward(self,x):\n        out = self.conv1(self.relu(self.norm1(x)))\n        out = self.conv2(self.relu(self.norm2(out)))\n        return torch.cat([x, out], 1) ","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:39:24.467478Z","iopub.execute_input":"2024-04-23T03:39:24.467837Z","iopub.status.idle":"2024-04-23T03:39:24.476514Z","shell.execute_reply.started":"2024-04-23T03:39:24.467807Z","shell.execute_reply":"2024-04-23T03:39:24.475422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def test_DenseLayer():\n    x = torch.randn(1,64,224,224)\n    model = DenseLayer(64)\n    # print(model(x).shape)\n    # print(model)\n    # del model\n    return model","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:39:24.682414Z","iopub.execute_input":"2024-04-23T03:39:24.682726Z","iopub.status.idle":"2024-04-23T03:39:24.687603Z","shell.execute_reply.started":"2024-04-23T03:39:24.682701Z","shell.execute_reply":"2024-04-23T03:39:24.686653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = test_DenseLayer()\n\narchitecture = 'denselayer'\nmodel_graph = draw_graph(model, input_size=(1,64,224,224), graph_dir ='TB' , roll=True, expand_nested=True, graph_name=f'self_{architecture}',save_graph=True,filename=f'self_{architecture}')\nmodel_graph.visual_graph","metadata":{"execution":{"iopub.status.busy":"2024-04-23T00:24:35.336276Z","iopub.execute_input":"2024-04-23T00:24:35.336660Z","iopub.status.idle":"2024-04-23T00:24:35.507411Z","shell.execute_reply.started":"2024-04-23T00:24:35.336629Z","shell.execute_reply":"2024-04-23T00:24:35.506424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = test_DenseLayer()\n\narchitecture = 'denselayer2'\nmodel_graph = draw_graph(model, input_size=(1,64,224,224), graph_dir ='TB' , roll=True, expand_nested=True, graph_name=f'self_{architecture}',save_graph=True,filename=f'self_{architecture}')\nmodel_graph.visual_graph","metadata":{"execution":{"iopub.status.busy":"2024-04-23T00:24:35.704713Z","iopub.execute_input":"2024-04-23T00:24:35.705246Z","iopub.status.idle":"2024-04-23T00:24:35.845769Z","shell.execute_reply.started":"2024-04-23T00:24:35.705209Z","shell.execute_reply":"2024-04-23T00:24:35.844846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TransitionLayer(nn.Module):\n    def __init__(self, num_input_features, num_output_features):\n        super(TransitionLayer, self).__init__()\n        self.norm = nn.BatchNorm2d(num_input_features)\n        self.relu = nn.ReLU(inplace=True)\n        self.conv = nn.Conv2d(num_input_features, num_output_features,\n                              kernel_size=1, stride=1, bias=False)\n        self.pool = nn.AvgPool2d(kernel_size=2, stride=2)\n\n    def forward(self, x):\n        out = self.conv(self.relu(self.norm(x)))\n        out = self.pool(out)\n        return out","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:39:27.785259Z","iopub.execute_input":"2024-04-23T03:39:27.785989Z","iopub.status.idle":"2024-04-23T03:39:27.792594Z","shell.execute_reply.started":"2024-04-23T03:39:27.785956Z","shell.execute_reply":"2024-04-23T03:39:27.791744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def test_TransitionLayer():\n    x = torch.randn(1,64,224,224)\n    model = TransitionLayer(64,32)\n    print('Transition Layer Output shape : ',model(x).shape)\n    print('Model : ',model)\n    return model\n\n\nmodel = test_TransitionLayer()\narchitecture = 'transition'\nmodel_graph = draw_graph(model, input_size=(1,64,224,224), graph_dir ='TB' , roll=True, expand_nested=True, graph_name=f'self_{architecture}',save_graph=True,filename=f'self_{architecture}')\nmodel_graph.visual_graph","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:39:28.235342Z","iopub.execute_input":"2024-04-23T03:39:28.235697Z","iopub.status.idle":"2024-04-23T03:39:28.443974Z","shell.execute_reply.started":"2024-04-23T03:39:28.235670Z","shell.execute_reply":"2024-04-23T03:39:28.443043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DenseBlock(nn.Module):\n    def __init__(self, num_layers, num_input_features):\n        super(DenseBlock, self).__init__()\n        self.layers = nn.ModuleList()\n        for i in range(num_layers):\n            layer = DenseLayer(num_input_features + i * k)\n            self.layers.append(layer)\n\n    def forward(self, x):\n        for layer in self.layers:\n            x = layer(x)\n            # After each layer, update in_planes to reflect the new total number of channels\n        return x\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:39:30.433467Z","iopub.execute_input":"2024-04-23T03:39:30.434352Z","iopub.status.idle":"2024-04-23T03:39:30.440586Z","shell.execute_reply.started":"2024-04-23T03:39:30.434319Z","shell.execute_reply":"2024-04-23T03:39:30.439500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def test_DenseBlock():\n    x = torch.randn(1,3,224,224)\n    model = DenseBlock(3,3)\n    print('Denseblock Output shape : ',model(x).shape)\n    print('Model ',model)\n    # del model\n    return model\n\nmodel = test_DenseBlock()\n\narchitecture = 'denseblock'\nmodel_graph = draw_graph(model, input_size=(1,3,224,224), graph_dir ='TB' , roll=True, expand_nested=True, graph_name=f'self_{architecture}',save_graph=True,filename=f'self_{architecture}')\nmodel_graph.visual_graph","metadata":{"execution":{"iopub.status.busy":"2024-04-21T23:47:00.944081Z","iopub.execute_input":"2024-04-21T23:47:00.944539Z","iopub.status.idle":"2024-04-21T23:47:01.615345Z","shell.execute_reply.started":"2024-04-21T23:47:00.944504Z","shell.execute_reply":"2024-04-21T23:47:01.614037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DenseNet(nn.Module):\n    def __init__(self, in_channels, num_classes=6, growth_rate=32, block_config=(6, 12, 24, 16)):\n        super(DenseNet, self).__init__()\n\n        # Initial convolution\n        self.init_conv = nn.Conv2d(in_channels, growth_rate * 2, kernel_size=7, stride=2, padding=3, bias=False)\n        self.init_bn = nn.BatchNorm2d(growth_rate * 2)\n        self.init_relu = nn.ReLU(inplace=True)\n        self.init_pool = nn.MaxPool2d(kernel_size=3, stride=2, padding=1)\n\n        # Dense Blocks and Transition Layers\n        num_features = growth_rate * 2\n        self.blocks = nn.ModuleList()\n        self.transitions = nn.ModuleList()\n\n        for i, num_layers in enumerate(block_config):\n            block = DenseBlock(num_layers, num_features)\n            self.blocks.append(block)\n            num_features += num_layers * growth_rate  # Increment feature map count\n            \n            # Don't add a transition layer after the last block\n            if i != len(block_config) - 1:\n                trans = TransitionLayer(num_input_features=num_features, num_output_features=num_features // 2)\n                self.transitions.append(trans)\n                num_features = num_features // 2\n\n        # Final batch norm\n        self.final_bn = nn.BatchNorm2d(num_features)\n        self.final_relu = nn.ReLU(inplace=True)\n\n        # Global Average Pooling\n        self.global_avg_pool = nn.AdaptiveAvgPool2d((1, 1))\n\n        # Classifier\n        self.classifier = nn.Linear(num_features, num_classes)\n\n    def forward(self, x):\n        x = self.init_conv(x)\n        x = self.init_bn(x)\n        x = self.init_relu(x)\n        x = self.init_pool(x)\n\n        for block, trans in zip(self.blocks, self.transitions):\n            x = block(x)\n            x = trans(x)\n        \n        # Handle the last block separately since it doesn't have a transition layer\n        x = self.blocks[-1](x)\n\n        x = self.final_bn(x)\n        x = self.final_relu(x)\n        x = self.global_avg_pool(x)\n        x = torch.flatten(x, 1)\n        x = self.classifier(x)\n        return F.log_softmax(x, dim=1)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-22T00:07:35.099860Z","iopub.execute_input":"2024-04-22T00:07:35.100595Z","iopub.status.idle":"2024-04-22T00:07:35.125722Z","shell.execute_reply.started":"2024-04-22T00:07:35.100536Z","shell.execute_reply":"2024-04-22T00:07:35.124102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"new smaller densenet","metadata":{}},{"cell_type":"code","source":"class DenseNet(nn.Module):\n    def __init__(self, growth_rate=32, block_config=(2, 2, 2, 2), num_init_features=32, num_classes=6):\n        super(DenseNet, self).__init__()\n\n        # Initial convolution, here we're assuming that the initial number of features is 32.\n        self.init_conv = nn.Conv2d(1, num_init_features, kernel_size=7, stride=2, padding=3, bias=False)\n        self.init_bn = nn.BatchNorm2d(num_init_features)\n        self.init_relu = nn.ReLU(inplace=True)\n        self.init_pool = nn.MaxPool2d(kernel_size=3, stride=2, padding=1)\n\n        # DenseBlocks and TransitionLayers\n        num_features = num_init_features\n        self.dense_blocks = nn.ModuleList()\n        self.trans_layers = nn.ModuleList()\n\n        for i, num_layers in enumerate(block_config):\n            block = DenseBlock(num_layers=num_layers, num_input_features=num_features)\n            self.dense_blocks.append(block)\n            num_features += num_layers * growth_rate\n\n            if i != len(block_config) - 1:  # Do not perform downsampling after the last block\n                trans_layer = TransitionLayer(num_input_features=num_features, num_output_features=num_features // 2)\n                self.trans_layers.append(trans_layer)\n                num_features = num_features // 2\n\n        # Final batch norm\n        self.final_bn = nn.BatchNorm2d(num_features)\n        self.final_relu = nn.ReLU(inplace=True)\n\n        # Global Average Pooling and Classifier\n        self.global_avg_pool = nn.AdaptiveAvgPool2d((1, 1))\n        self.classifier = nn.Linear(num_features, num_classes)\n\n    def forward(self, x):\n        x = self.init_bn(self.init_conv(x))\n        x = self.init_relu(x)\n        x = self.init_pool(x)\n\n        for block, trans_layer in zip(self.dense_blocks, self.trans_layers):\n            x = block(x)\n            x = trans_layer(x)\n        x = self.dense_blocks[-1](x)  # Apply the last dense block\n\n        x = self.final_bn(x)\n        x = self.final_relu(x)\n        x = self.global_avg_pool(x)\n        x = x.view(x.size(0), -1)\n        x = self.classifier(x)\n        return F.log_softmax(x, dim=1)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T03:39:35.259651Z","iopub.execute_input":"2024-04-23T03:39:35.260352Z","iopub.status.idle":"2024-04-23T03:39:35.274110Z","shell.execute_reply.started":"2024-04-23T03:39:35.260318Z","shell.execute_reply":"2024-04-23T03:39:35.272939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = torch.randn(1,3,224,224)\nmodel = DenseNet(in_channels=1, num_classes=6)\n\narchitecture = 'denseNet'\nmodel_graph = draw_graph(model, input_size=(1,1,224,224), graph_dir ='TB' , roll=False, expand_nested=True, show_shapes=True, graph_name=f'self_{architecture}',save_graph=True,filename=f'self_{architecture}')\n#model_graph.visual_graph","metadata":{"execution":{"iopub.status.busy":"2024-04-22T00:07:37.405417Z","iopub.execute_input":"2024-04-22T00:07:37.405841Z","iopub.status.idle":"2024-04-22T00:07:39.530911Z","shell.execute_reply.started":"2024-04-22T00:07:37.405805Z","shell.execute_reply":"2024-04-22T00:07:39.529560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install torchsummary\nfrom torchsummary import summary\nsummary(model, (1, 224, 224))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"network = DenseNet(in_channels=1, num_classes=6)\n\nimport torch.optim as optim\n\n# create the learning rule\noptimizer = optim.SGD(network.parameters(),\n                      lr=0.1,   # learning rate\n                      momentum=0.5)\n\n\naccuracy_test = []\naccuracy_train = []\n\nfor i in range(10): # each call of continue_training trains for 1 epochs\n  #continue_training()","metadata":{"execution":{"iopub.status.busy":"2024-04-22T00:08:00.966377Z","iopub.execute_input":"2024-04-22T00:08:00.966792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DenseNet(nn.Module):\n    def __init__(self, growth_rate=32, block_config=(2, 2, 2, 2), num_init_features=32, num_classes=6):\n        super(DenseNet, self).__init__()\n\n        # Initial convolution, here we're assuming that the initial number of features is 32.\n        self.init_conv = nn.Conv2d(1, num_init_features, kernel_size=7, stride=2, padding=3, bias=False)\n        self.init_bn = nn.BatchNorm2d(num_init_features)\n        self.init_relu = nn.ReLU(inplace=True)\n        self.init_pool = nn.MaxPool2d(kernel_size=3, stride=2, padding=1)\n\n        # DenseBlocks and TransitionLayers\n        num_features = num_init_features\n        self.dense_blocks = nn.ModuleList()\n        self.trans_layers = nn.ModuleList()\n\n        for i, num_layers in enumerate(block_config):\n            block = DenseBlock(num_layers=num_layers, num_input_features=num_features)\n            self.dense_blocks.append(block)\n            num_features += num_layers * growth_rate\n\n            if i != len(block_config) - 1:  # Do not perform downsampling after the last block\n                trans_layer = TransitionLayer(num_input_features=num_features, num_output_features=num_features // 2)\n                self.trans_layers.append(trans_layer)\n                num_features = num_features // 2\n\n        # Final batch norm\n        self.final_bn = nn.BatchNorm2d(num_features)\n        self.final_relu = nn.ReLU(inplace=True)\n\n        # Global Average Pooling and Classifier\n        self.global_avg_pool = nn.AdaptiveAvgPool2d((1, 1))\n        self.classifier = nn.Linear(num_features, num_classes)\n\n    def forward(self, x):\n        x = self.init_bn(self.init_conv(x))\n        x = self.init_relu(x)\n        x = self.init_pool(x)\n\n        for block, trans_layer in zip(self.dense_blocks, self.trans_layers):\n            x = block(x)\n            x = trans_layer(x)\n        x = self.dense_blocks[-1](x)  # Apply the last dense block\n\n        x = self.final_bn(x)\n        x = self.final_relu(x)\n        x = self.global_avg_pool(x)\n        x = x.view(x.size(0), -1)\n        x = self.classifier(x)\n        return x\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T04:02:04.187948Z","iopub.execute_input":"2024-04-23T04:02:04.188597Z","iopub.status.idle":"2024-04-23T04:02:04.201797Z","shell.execute_reply.started":"2024-04-23T04:02:04.188566Z","shell.execute_reply":"2024-04-23T04:02:04.200557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"network = DenseNet(growth_rate=32, block_config=(2, 2, 2, 2), num_init_features=32, num_classes=6)\noptimizer = torch.optim.Adam(network.parameters(), lr=0.001)\nprint(torch.cuda.is_available()) # -------------------------------------------------------------------------TO JULIET KERN:  THESE LINES OF CODE BELOW ARE TO SET DEVICE TO GPU, and NETWORK TO GPU\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nnetwork = network.to(device)\n\ntrain_losses, val_losses = train_model(network, train_loader, val_loader, optimizer)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T04:02:07.104034Z","iopub.execute_input":"2024-04-23T04:02:07.104762Z","iopub.status.idle":"2024-04-23T04:28:23.692828Z","shell.execute_reply.started":"2024-04-23T04:02:07.104733Z","shell.execute_reply":"2024-04-23T04:28:23.691956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming pred_labels and true_labels are numpy arrays as provided\n# Number of classes (columns)\nnum_classes = pred_labels.shape[1]\n\n# basically the same thing as TARS but opposite to get labels for the graphs\ninverse_TARS = {v: k for k, v in TARS.items()}\nnon_log_values = np.exp(pred_labels)\nnon_log_pred_labels = non_log_values / non_log_values.sum(axis=1, keepdims=True)\n\n# Create a histogram for each class\nfig, axes = plt.subplots(num_classes, 1, figsize=(5, 3 * num_classes))\n\nfor i in range(num_classes):\n    # Plot histogram of predicted probabilities\n    axes[i].hist(true_labels[:, i], bins=20, alpha=0.5, label='True Probabilities')\n    axes[i].hist(non_log_pred_labels[:, i], bins=20, alpha=0.5, label='Predicted Probabilities')\n    # Overlay true probabilities (scaled for visualization if necessary)\n    \n    axes[i].set_title(f'Class: {inverse_TARS[i]}')\n    axes[i].set_xlabel('Probability')\n    axes[i].set_ylabel('Frequency')\n    axes[i].legend()\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T04:34:10.355875Z","iopub.execute_input":"2024-04-23T04:34:10.356282Z","iopub.status.idle":"2024-04-23T04:34:10.403055Z","shell.execute_reply.started":"2024-04-23T04:34:10.356254Z","shell.execute_reply":"2024-04-23T04:34:10.401617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}