{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":21669,"databundleVersionId":1692278,"sourceType":"competition"},{"sourceId":1314904,"sourceType":"datasetVersion","datasetId":761895}],"dockerImageVersionId":30588,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install  resnest > /dev/null","metadata":{"execution":{"iopub.status.busy":"2023-11-28T17:30:52.458357Z","iopub.execute_input":"2023-11-28T17:30:52.459021Z","iopub.status.idle":"2023-11-28T17:31:04.191325Z","shell.execute_reply.started":"2023-11-28T17:30:52.458989Z","shell.execute_reply":"2023-11-28T17:31:04.190142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport librosa \nimport numpy as np\nfrom skimage.transform import resize\n\nsr = 48000\nlength = 10 * sr\ndata = pd.read_csv(\"../input/rfcx-species-audio-detection/train_tp.csv\")\n\ndef resize_image(spec):\n    return resize(spec, (224, 400))\n\ndef colorize(X, eps=1e-6, mean=None, std=None):\n    mean = mean or X.mean()\n    std = std or X.std()\n    X = (X - mean) / (std + eps)\n    _min, _max = X.min(), X.max()\n    if (_max - _min) > eps:\n        V = np.clip(X, _min, _max)\n        V = 255 * (V - _min) / (_max - _min)\n        V = V.astype(np.uint8)\n    else:\n        V = np.zeros_like(X, dtype=np.uint8)\n    return V\n\ndef normalize(image):\n    image = image.astype(\"float32\", copy=False) / 255.0\n    image = np.stack([image, image, image])\n    return image\n\nfmin = 100000\nfmax = 0\nfor i in range(0, len(data)):\n    if fmin > float(data.iloc[i]['f_min']):\n        fmin = float(data.iloc[i]['f_min'])\n    if fmax < float(data.iloc[i]['f_max']):\n        fmax = float(data.iloc[i]['f_max'])\n        \nlabel_list = []\ndata_list = []\naudio_data = {}\nfor i in range(0, len(data)):\n    if i % 100 == 0:\n        print(str(i) + '/' + str(len(data)))\n    recording_id = data.recording_id.values[i]\n    species_id = int(data.species_id.values[i])\n    data_list.append(recording_id)\n    label_list.append(species_id)\n\n    wav, sr = librosa.load('../input/rfcx-species-audio-detection/train/' + recording_id + '.flac', sr=None)\n    t_min = float(data.t_min.values[i]) * sr\n    t_max = float(data.t_max.values[i]) * sr\n    center = np.round((t_min + t_max) / 2)\n    beginning = center - length / 2\n    if beginning < 0:\n        beginning = 0\n    ending = beginning + length\n    if ending > len(wav):\n        ending = len(wav)\n        beginning = ending - length\n    slice = wav[int(beginning):int(ending)]\n    \n    spec = librosa.feature.melspectrogram(y=slice, sr=sr, fmin=fmin, fmax=fmax)\n    spec_db = librosa.power_to_db(spec, top_db=80)\n    \n    img = normalize(colorize(resize_image(spec_db)))\n    \n    audio_data[recording_id] = img\n ","metadata":{"execution":{"iopub.status.busy":"2023-11-28T17:31:04.193617Z","iopub.execute_input":"2023-11-28T17:31:04.193932Z","iopub.status.idle":"2023-11-28T17:35:50.027526Z","shell.execute_reply.started":"2023-11-28T17:31:04.193903Z","shell.execute_reply":"2023-11-28T17:35:50.026083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nimport random\nfrom torch.utils.data import Dataset, DataLoader\n\nclass CustomDataset(Dataset):\n    def __init__(self, X, y, funcs=[]):\n        self.data = []\n        self.labels = []\n        self.funcs=funcs\n        for i in range(0, len(X)):\n            recording_id = X[i]\n            label = y[i]\n            mel_spec = audio_data[recording_id]\n            self.data.append(mel_spec)\n            self.labels.append(label)\n\n                \n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        if len(self.funcs):\n            func = random.choice(self.funcs)\n            data = func(self.data[idx])\n        else:\n            data = self.data[idx]\n        return data, self.labels[idx]","metadata":{"execution":{"iopub.status.busy":"2023-11-28T17:35:50.029302Z","iopub.execute_input":"2023-11-28T17:35:50.029864Z","iopub.status.idle":"2023-11-28T17:35:50.044987Z","shell.execute_reply.started":"2023-11-28T17:35:50.029815Z","shell.execute_reply":"2023-11-28T17:35:50.042728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom skimage import exposure, util\nfrom resnest.torch import resnest101\nimport torch\nimport torch.nn as nn\n\ndevice = 'cuda' if torch.cuda.is_available() else 'cpu'\nweights = torch.load('../input/resnest-package/resnest101-22405ba7.pth', map_location=device)\nmodel = resnest101()\nmodel.load_state_dict(weights)\nmodel.fc = nn.Linear(model.fc.in_features, 24)\nmodel = model.to(device)\n\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.Adam(model.parameters(), lr =  0.001)\n\ndef change_gamma(img):\n    output = exposure.adjust_gamma(img, (random.randint(3, 7) / 5))\n    return output\n\ndef add_noise(img):\n    output = util.random_noise(img)\n    return output\n\nX_train, X_val, y_train, y_val = train_test_split(data_list, label_list, test_size=0.1)\ntrain_data = CustomDataset(X_train, y_train)\ntrain_data2 = CustomDataset(X_train, y_train, [change_gamma, add_noise])\nvalid_data = CustomDataset(X_val, y_val)\ntrain_loader = DataLoader(\n    torch.utils.data.ConcatDataset([train_data, train_data2]), \n    batch_size=32, shuffle=True\n)\nvalid_loader = DataLoader(valid_data, batch_size=32, shuffle=True)\ntrain_accuracies = []\ntest_accuracies = []\nnum_epochs = 9\nfor epoch in range(num_epochs):\n    # Training\n    train_loss = 0.0\n    correct_train = 0\n    total_train = 0\n    for images, labels in tqdm(train_loader):\n        images = images.to(device)\n        labels = labels.to(device)\n\n        # Forward pass\n        outputs = model(images.float())\n        loss = criterion(outputs, labels)\n\n        # Backward pass and optimization\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        # Track train loss and accuracy\n        train_loss += loss.item() * images.size(0)\n        _, predicted = torch.max(outputs.data, 1)\n        total_train += labels.size(0)\n        correct_train += (predicted == labels).sum().item()\n\n    # Calculate train accuracy and loss\n    train_accuracy = correct_train / total_train\n    train_loss = train_loss / total_train\n    \n    train_accuracies.append(train_accuracy)\n    # Evaluation (Test)\n    test_loss = 0.0\n    correct_test = 0\n    total_test = 0\n    all_predictions = []\n    all_targets = []\n    with torch.no_grad():\n        for images, labels in valid_loader:\n            images = images.to(device)\n            labels = labels.to(device)\n\n            # Forward pass\n            outputs = model(images.float())\n            loss = criterion(outputs, labels)\n\n            # Track test loss and accuracy\n            test_loss += loss.item() * images.size(0)\n            _, predicted = torch.max(outputs.data, 1)\n            total_test += labels.size(0)\n            correct_test += (predicted == labels).sum().item()\n            all_predictions.extend(predicted.cpu().numpy())\n            all_targets.extend(labels.cpu().numpy())\n\n    # Calculate test accuracy and loss\n    test_accuracy = correct_test / total_test\n    test_loss = test_loss / total_test\n    \n    test_accuracies.append(test_accuracy)\n\n    # Print epoch results\n    print(f\"Epoch {epoch+1}/{num_epochs} - Train Loss: {train_loss:.4f} - Train Acc: {train_accuracy:.4f} - Test Loss: {test_loss:.4f} - Test Acc: {test_accuracy:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-28T18:07:57.845884Z","iopub.execute_input":"2023-11-28T18:07:57.846260Z","iopub.status.idle":"2023-11-28T18:25:42.687198Z","shell.execute_reply.started":"2023-11-28T18:07:57.846230Z","shell.execute_reply":"2023-11-28T18:25:42.686188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\n\n# Assuming you have a trained model object named \"model\"\n# and you want to save it to a file named \"model.pkl\"\n\n# Save the model to a pickle file\nwith open('model.pkl', 'wb') as file:\n    pickle.dump(model, file)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T18:25:46.198057Z","iopub.execute_input":"2023-11-28T18:25:46.198747Z","iopub.status.idle":"2023-11-28T18:25:46.702931Z","shell.execute_reply.started":"2023-11-28T18:25:46.198717Z","shell.execute_reply":"2023-11-28T18:25:46.701698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_test_file(f):\n    wav, sr = librosa.load('/kaggle/input/rfcx-species-audio-detection/test/' + f, sr=None)\n    \n    segments = len(wav) / length\n    segments = int(np.ceil(segments))\n    \n    mel_array = []\n    \n    for i in range(0, segments):\n        if (i + 1) * length > len(wav):\n            slice = wav[len(wav) - length:len(wav)]\n        else:\n            slice = wav[i * length:(i + 1) * length]\n        \n        spec = librosa.feature.melspectrogram(y=slice, sr=sr,fmin=fmin,fmax=fmax)\n        spec_db = librosa.power_to_db(spec,top_db=80)\n\n        img = normalize(colorize(resize_image(spec_db)))\n        mel_spec = img\n        mel_array.append(mel_spec)\n    \n    return mel_array","metadata":{"execution":{"iopub.status.busy":"2023-11-28T18:27:14.810520Z","iopub.execute_input":"2023-11-28T18:27:14.810892Z","iopub.status.idle":"2023-11-28T18:27:14.818970Z","shell.execute_reply.started":"2023-11-28T18:27:14.810862Z","shell.execute_reply":"2023-11-28T18:27:14.817996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn.functional as nnf\nimport csv\nimport os\n\nmodel.eval()\nwith open('submission.csv', 'w', newline='') as csvfile:\n    submission_writer = csv.writer(csvfile, delimiter=',')\n    submission_writer.writerow(['recording_id','s0','s1','s2','s3','s4','s5','s6','s7','s8','s9','s10','s11',\n                               's12','s13','s14','s15','s16','s17','s18','s19','s20','s21','s22','s23'])\n    \n    test_files = os.listdir('/kaggle/input/rfcx-species-audio-detection/test/')\n\n    for i in range(0, len(test_files)):\n        data = load_test_file(test_files[i])\n        data = torch.tensor(data)\n        data = data.float()\n        if torch.cuda.is_available():\n            data = data.cuda()\n        with torch.no_grad():\n            output = model(data)\n            output = nnf.softmax(output, dim=1)\n        maxed_output = torch.max(output, dim=0)[0]\n        maxed_output = maxed_output.cpu().detach()\n        file_id = str.split(test_files[i], '.')[0]\n        write_array = [file_id]\n        \n        for out in maxed_output:\n            write_array.append(out.item())\n    \n        submission_writer.writerow(write_array)\n        \n        \n        if i % 100 == 0 and i > 0:\n            print(str(i) + '/' + str(len(test_files)))\n\nprint('Done')","metadata":{"execution":{"iopub.status.busy":"2023-11-28T18:30:26.186731Z","iopub.execute_input":"2023-11-28T18:30:26.187443Z","iopub.status.idle":"2023-11-28T19:07:01.781999Z","shell.execute_reply.started":"2023-11-28T18:30:26.187388Z","shell.execute_reply":"2023-11-28T19:07:01.780936Z"},"trusted":true},"execution_count":null,"outputs":[]}]}