{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport torch\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport typing\nimport PIL\nfrom torchvision.transforms.functional import to_pil_image\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n\"\"\"\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\"\"\"\n\n\ndirectory = \"/kaggle/input/hms-harmful-brain-activity-classification/\"\nspectrogram_directory = directory + \"train_spectrograms/\"\nspectrogram: pd.DataFrame = pd.read_parquet(spectrogram_directory+\"1000086677.parquet\")\ntensor = torch.tensor(spectrogram.values)[:,1:]\n\n\nprint(tensor)\nimage = to_pil_image(tensor)\nimage\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-08T20:55:18.533259Z","iopub.execute_input":"2024-07-08T20:55:18.534404Z","iopub.status.idle":"2024-07-08T20:55:26.342070Z","shell.execute_reply.started":"2024-07-08T20:55:18.534339Z","shell.execute_reply":"2024-07-08T20:55:26.340073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(directory+\"train.csv\")\nprint(data.info())","metadata":{"execution":{"iopub.status.busy":"2024-07-08T20:55:26.344591Z","iopub.execute_input":"2024-07-08T20:55:26.345193Z","iopub.status.idle":"2024-07-08T20:55:26.693694Z","shell.execute_reply.started":"2024-07-08T20:55:26.345159Z","shell.execute_reply":"2024-07-08T20:55:26.692314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import Dataset\nclass SpectrogramDataset(Dataset):\n    def __init__(self):\n        self.directory = \"/kaggle/input/hms-harmful-brain-activity-classification/\"\n        self.spectrogram_directory = self.directory + \"train_spectrograms/\"\n        self.data = pd.read_csv(self.directory+\"train.csv\")\n    \n    def __len__(self):\n        return len(self.data)\n    \n    def create_vote_tensor(self, row):\n        vote_list = pd.to_numeric(row[[\"seizure_vote\", \"lpd_vote\", \"gpd_vote\", \"lrda_vote\", \"grda_vote\", \"other_vote\"]])\n        vote_tensor = torch.tensor(vote_list.values)\n        total = torch.sum(vote_tensor)\n        return vote_tensor/total\n        #Finish this by returning the probabilities instead of the \n        #numbers of votes\n        \n    def create_spectrogram_tensor(self, row):\n        spectrogram_id = row[\"spectrogram_id\"]\n        start_time = row[\"spectrogram_label_offset_seconds\"]\n        end_time = start_time + 600\n        spectrogram: pd.DataFrame = pd.read_parquet(\n            spectrogram_directory+f\"{spectrogram_id}.parquet\"\n        )\n        spectrogram = spectrogram[(spectrogram[\"time\"]>=start_time)&(spectrogram[\"time\"]<=end_time)]\n        tensor = torch.tensor(spectrogram.values)[:,1:]\n        tensor = torch.nan_to_num(tensor)\n        return tensor[:300]\n    \n    def __getitem__(self, idx):\n        #self.data.iloc is essentially a list of the rows of self.data\n        #\"iloc\" means index location\n        row = self.data.iloc[idx]\n        return self.create_spectrogram_tensor(row), self.create_vote_tensor(row)\n        \n        \n        #row[\"spectrogram_id\"], or for multiple columns\n        #row[[\"spectrogram_id\", \"spectrogram_label_offset_seconds\"]]\n        #Create a variable \"start_time\" which has the value of\n        #spectrogram_label_offset_seconds\n        #And another variable which is a list of the votes\n        #(seizure_vote, lpd_vote,...)\n        # return row\n        \n \n#__init__ defines this function\n\n\n\"\"\"\n#__len__ defines this\nlen(dataset)\n\n#__getitem defines this\ndataset[0]\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-07-08T20:55:26.695497Z","iopub.execute_input":"2024-07-08T20:55:26.696008Z","iopub.status.idle":"2024-07-08T20:55:26.714889Z","shell.execute_reply.started":"2024-07-08T20:55:26.695968Z","shell.execute_reply":"2024-07-08T20:55:26.713587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"dataset = SpectrogramDataset()\nfor spectrogram, votes in dataset:\n    print(spectrogram.shape)\n    print(votes)\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-07-08T20:55:26.717560Z","iopub.execute_input":"2024-07-08T20:55:26.717930Z","iopub.status.idle":"2024-07-08T20:55:26.736542Z","shell.execute_reply.started":"2024-07-08T20:55:26.717899Z","shell.execute_reply":"2024-07-08T20:55:26.735166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.nn import Module\nclass Model(Module):\n    def __init__(self):\n        super().__init__()\n        self.conv1 = torch.nn.Conv2d(1,8,5)\n        self.pool = torch.nn.MaxPool2d(2)\n        self.conv2 = torch.nn.Conv2d(8,16,5)\n        self.conv3 = torch.nn.Conv2d(16,32,5)\n        self.layer = torch.nn.LazyLinear(200)\n        self.activation = torch.nn.GELU()\n        self.layer2 = torch.nn.Linear(200, 100)\n        self.layer3=torch.nn.Linear(100,6)\n        self.dtype = self.layer.weight.dtype\n        \n    \n    def forward(self, spectrogram):\n        spectrogram = spectrogram.to(dtype=self.dtype)\n        spectrogram = spectrogram/torch.max(spectrogram)\n\n        \n        #(batch_size, rows, columns) => (batch_size, 1,rows, columns)\n        spectrogram = spectrogram.unsqueeze(1)\n        \n        #This next part is some convolutional layers, I was experimenting\n        spectrogram = self.conv1(spectrogram)\n        spectrogram = self.pool(spectrogram)\n        spectrogram = self.activation(spectrogram)\n        spectrogram = self.conv2(spectrogram)\n        spectrogram = self.pool(spectrogram)\n        spectrogram = self.activation(spectrogram)\n        spectrogram = self.conv3(spectrogram)\n        spectrogram = self.pool(spectrogram)\n        spectrogram = self.activation(spectrogram)\n        \n        shape = spectrogram.shape\n        batch_size = shape[0]\n        spectrogram = spectrogram.reshape((batch_size, shape[1]*shape[2]*shape[3]))\n        spectrogram = self.layer(spectrogram)\n        spectrogram = self.activation(spectrogram)\n        spectrogram = self.layer2(spectrogram)\n        spectrogram = self.activation(spectrogram)\n        spectrogram = self.layer3(spectrogram)\n        return spectrogram\n        ","metadata":{"execution":{"iopub.status.busy":"2024-07-08T20:58:55.439700Z","iopub.execute_input":"2024-07-08T20:58:55.440162Z","iopub.status.idle":"2024-07-08T20:58:55.455299Z","shell.execute_reply.started":"2024-07-08T20:58:55.440127Z","shell.execute_reply":"2024-07-08T20:58:55.454117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculate_accuracy(prediction, votes):\n    prediction = torch.argmax(prediction, -1)\n    votes = torch.argmax(votes, -1)\n    return 100*torch.sum(prediction == votes)/votes.shape[0]","metadata":{"execution":{"iopub.status.busy":"2024-07-08T20:55:26.756366Z","iopub.execute_input":"2024-07-08T20:55:26.756813Z","iopub.status.idle":"2024-07-08T20:55:26.777036Z","shell.execute_reply.started":"2024-07-08T20:55:26.756779Z","shell.execute_reply":"2024-07-08T20:55:26.775721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import wandb\n\nclass Trainer:\n    def __init__(self, model, num_epochs, dataloader, loss, optimizer, scheduler, run):\n        self.model = model\n        self.num_epochs = num_epochs\n        self.dataloader = dataloader\n        self.loss = loss\n        self.optimizer = optimizer\n        self.run = run\n        self.scheduler = scheduler\n        self.step=0\n        \n    def training_loop(self):\n        for epoch in range(self.num_epochs):\n            for spectrogram, votes in self.dataloader:\n                self.training_step(spectrogram, votes)\n    \n    def training_step(self, spectrogram, votes):\n        self.optimizer.zero_grad()\n        prediction = self.model.forward(spectrogram)\n        loss = self.loss(prediction, votes)\n        loss.backward()\n        self.loss_value=float(loss.item())\n        self.optimizer.step()\n        self.scheduler.step()\n        print(self.scheduler.get_last_lr())\n        accuracy = calculate_accuracy(prediction, votes)\n        self.run.log({\"Loss\": loss, \"Accuracy\": accuracy})\n        \n        \n        ","metadata":{"execution":{"iopub.status.busy":"2024-07-08T20:58:59.271001Z","iopub.execute_input":"2024-07-08T20:58:59.271469Z","iopub.status.idle":"2024-07-08T20:58:59.286210Z","shell.execute_reply.started":"2024-07-08T20:58:59.271435Z","shell.execute_reply":"2024-07-08T20:58:59.285075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import DataLoader\nimport wandb\nmodel = Model()\nnum_epochs = 100\ndataset = SpectrogramDataset()\nbatch_size = 20\ndataloader = DataLoader(dataset, batch_size, shuffle=True)\nloss = torch.nn.CrossEntropyLoss()\nlr = 0.05\noptimizer = torch.optim.Adam(model.parameters(), lr = lr)\nrun = wandb.init(project=\"hms\")\nconfig = wandb.config\nconfig.learning_rate = lr\n#We didn't use a learning rate scheduler in class, but it's always good to use one. It's not clear if this one is the best\nscheduler = torch.optim.lr_scheduler.ExponentialLR(optimizer, .9995)\ntrainer = Trainer(\n    model,\n    num_epochs,\n    dataloader,\n    loss,\n    optimizer,\n    scheduler,\n    run\n)\ntrainer.training_loop()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-07-08T20:59:02.515247Z","iopub.execute_input":"2024-07-08T20:59:02.515667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%tensorboard.notebook --logdir /kaggle/working/logs","metadata":{"execution":{"iopub.status.busy":"2024-07-08T20:56:40.297499Z","iopub.status.idle":"2024-07-08T20:56:40.298077Z","shell.execute_reply.started":"2024-07-08T20:56:40.297846Z","shell.execute_reply":"2024-07-08T20:56:40.297873Z"},"trusted":true},"execution_count":null,"outputs":[]}]}