{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":21669,"databundleVersionId":1692278,"sourceType":"competition"},{"sourceId":1314904,"sourceType":"datasetVersion","datasetId":761895}],"dockerImageVersionId":30588,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install  resnest > /dev/null","metadata":{"execution":{"iopub.status.busy":"2024-04-22T14:39:39.129566Z","iopub.execute_input":"2024-04-22T14:39:39.129922Z","iopub.status.idle":"2024-04-22T14:39:53.774505Z","shell.execute_reply.started":"2024-04-22T14:39:39.129894Z","shell.execute_reply":"2024-04-22T14:39:53.773299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport librosa \nimport numpy as np\nfrom skimage.transform import resize\n\nsr = 48000\nlength = 10 * sr\ndata = pd.read_csv(\"../input/rfcx-species-audio-detection/train_tp.csv\")\n\ndef resize_image(spec):\n    return resize(spec, (224, 400))\n\ndef colorize(X, eps=1e-6, mean=None, std=None):\n    mean = mean or X.mean()\n    std = std or X.std()\n    X = (X - mean) / (std + eps)\n    _min, _max = X.min(), X.max()\n    if (_max - _min) > eps:\n        V = np.clip(X, _min, _max)\n        V = 255 * (V - _min) / (_max - _min)\n        V = V.astype(np.uint8)\n    else:\n        V = np.zeros_like(X, dtype=np.uint8)\n    return V\n\ndef normalize(image):\n    image = image.astype(\"float32\", copy=False) / 255.0\n    image = np.stack([image, image, image])\n    return image\n\nfmin = 100000\nfmax = 0\nfor i in range(0, len(data)):\n    if fmin > float(data.iloc[i]['f_min']):\n        fmin = float(data.iloc[i]['f_min'])\n    if fmax < float(data.iloc[i]['f_max']):\n        fmax = float(data.iloc[i]['f_max'])\n        \nlabel_list = []\ndata_list = []\naudio_data = {}\nfor i in range(0, len(data)):\n    if i % 100 == 0:\n        print(str(i) + '/' + str(len(data)))\n    recording_id = data.recording_id.values[i]\n    species_id = int(data.species_id.values[i])\n    data_list.append(recording_id)\n    label_list.append(species_id)\n\n    wav, sr = librosa.load('../input/rfcx-species-audio-detection/train/' + recording_id + '.flac', sr=None)\n    t_min = float(data.t_min.values[i]) * sr\n    t_max = float(data.t_max.values[i]) * sr\n    center = np.round((t_min + t_max) / 2)\n    beginning = center - length / 2\n    if beginning < 0:\n        beginning = 0\n    ending = beginning + length\n    if ending > len(wav):\n        ending = len(wav)\n        beginning = ending - length\n    slice = wav[int(beginning):int(ending)]\n    \n    spec = librosa.feature.melspectrogram(y=slice, sr=sr, fmin=fmin, fmax=fmax)\n    spec_db = librosa.power_to_db(spec, top_db=80)\n    \n    img = normalize(colorize(resize_image(spec_db)))\n    \n    audio_data[recording_id] = img\n ","metadata":{"execution":{"iopub.status.busy":"2024-04-22T14:39:53.776591Z","iopub.execute_input":"2024-04-22T14:39:53.776902Z","iopub.status.idle":"2024-04-22T14:45:03.274933Z","shell.execute_reply.started":"2024-04-22T14:39:53.776874Z","shell.execute_reply":"2024-04-22T14:45:03.273172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nimport random\nfrom torch.utils.data import Dataset, DataLoader\n\nclass CustomDataset(Dataset):\n    def __init__(self, X, y, funcs=[]):\n        self.data = []\n        self.labels = []\n        self.funcs=funcs\n        for i in range(0, len(X)):\n            recording_id = X[i]\n            label = y[i]\n            mel_spec = audio_data[recording_id]\n            self.data.append(mel_spec)\n            self.labels.append(label)\n\n                \n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        if len(self.funcs):\n            func = random.choice(self.funcs)\n            data = func(self.data[idx])\n        else:\n            data = self.data[idx]\n        return data, self.labels[idx]","metadata":{"execution":{"iopub.status.busy":"2024-04-22T14:45:03.277269Z","iopub.execute_input":"2024-04-22T14:45:03.278164Z","iopub.status.idle":"2024-04-22T14:45:03.291397Z","shell.execute_reply.started":"2024-04-22T14:45:03.278124Z","shell.execute_reply":"2024-04-22T14:45:03.289752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Mobilenet ka lfc\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport librosa\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, f1_score, recall_score, precision_score\nfrom torchvision.models import mobilenet_v2\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader\nimport matplotlib.pyplot as plt\n\n# Constants\nsr = 48000\nlength = 10 * sr\nnum_lpc_coeffs = 10  # Number of LPC coefficients to extract\n\ndef extract_lpc_features(wav):\n    lpc_coeffs = librosa.lpc(wav, num_lpc_coeffs)\n    return lpc_coeffs\n\n# Function to preprocess LPC features\ndef preprocess_lpc_features(lpc_features):\n    # You can perform additional preprocessing steps here if needed\n    return lpc_features\n\n# Data loading and preprocessing\ndata = pd.read_csv(\"../input/rfcx-species-audio-detection/train_tp.csv\")\nX_train, X_val, y_train, y_val = train_test_split(data_list, label_list, test_size=0.1)\ntrain_data = CustomDataset(X_train, y_train, funcs=[preprocess_lpc_features])\nvalid_data = CustomDataset(X_val, y_val, funcs=[preprocess_lpc_features])\ntrain_loader = DataLoader(train_data, batch_size=32, shuffle=True)\nvalid_loader = DataLoader(valid_data, batch_size=32, shuffle=True)\n\n# Model definition\nnum_classes = 24  # Number of output classes\nmodel = mobilenet_v2(pretrained=True)\nmodel.classifier = nn.Sequential(\n    nn.Dropout(0.2),\n    nn.Linear(model.last_channel, num_classes)\n)\nmodel = model.to(device)\n\n# Loss function and optimizer\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=0.001)\n\n# Training loop\nnum_epochs = 20\ntrain_losses = []\nval_losses = []\ntrain_accuracies = []\nval_accuracies = []\nf1_scores = []\n\nfor epoch in range(num_epochs):\n    # Training\n    model.train()\n    train_loss = 0.0\n    correct_train = 0\n    total_train = 0\n    \n    for images, labels in train_loader:\n        images = images.to(device)\n        labels = labels.to(device)\n\n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n\n        train_loss += loss.item() * images.size(0)\n        _, predicted = torch.max(outputs.data, 1)\n        total_train += labels.size(0)\n        correct_train += (predicted == labels).sum().item()\n\n    train_accuracy = correct_train / total_train\n    train_loss = train_loss / len(train_loader.dataset)\n    train_accuracies.append(train_accuracy)\n    train_losses.append(train_loss)\n\n    # Validation\n    model.eval()\n    val_loss = 0.0\n    correct_val = 0\n    total_val = 0\n    all_predictions = []\n    all_targets = []\n\n    with torch.no_grad():\n        for images, labels in valid_loader:\n            images = images.to(device)\n            labels = labels.to(device)\n\n            outputs = model(images)\n            loss = criterion(outputs, labels)\n\n            val_loss += loss.item() * images.size(0)\n            _, predicted = torch.max(outputs.data, 1)\n            total_val += labels.size(0)\n            correct_val += (predicted == labels).sum().item()\n\n            all_predictions.extend(predicted.cpu().numpy())\n            all_targets.extend(labels.cpu().numpy())\n\n    val_accuracy = correct_val / total_val\n    val_loss = val_loss / len(valid_loader.dataset)\n    val_accuracies.append(val_accuracy)\n    val_losses.append(val_loss)\n\n    # Calculate F1 score\n    f1 = f1_score(all_targets, all_predictions, average='weighted')\n    f1_scores.append(f1)\n\n    print(f\"Epoch {epoch+1}/{num_epochs} - Train Loss: {train_loss:.4f} - Train Acc: {train_accuracy:.4f} - \"\n          f\"Val Loss: {val_loss:.4f} - Val Acc: {val_accuracy:.4f} - F1 Score: {f1:.4f}\")\n\n# Plot training and validation losses\nplt.figure(figsize=(10, 5))\nplt.plot(train_losses, label='Training Loss')\nplt.plot(val_losses, label='Validation Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.title('Training and Validation Loss')\nplt.legend()\nplt.show()\n\n# Plot F1 score\nplt.figure(figsize=(10, 5))\nplt.plot(f1_scores, label='F1 Score')\nplt.xlabel('Epoch')\nplt.ylabel('F1 Score')\nplt.title('F1 Score')\nplt.legend()\nplt.show()\n\n# Confusion matrix\ncm = confusion_matrix(all_targets, all_predictions)\nplt.figure(figsize=(8, 6))\nplt.imshow(cm, interpolation='nearest', cmap=plt.cm.Blues)\nplt.title('Confusion Matrix')\nplt.colorbar()\ntick_marks = range(num_classes)\nplt.xticks(tick_marks, range(num_classes), rotation=45)\nplt.yticks(tick_marks, range(num_classes))\nplt.ylabel('True label')\nplt.xlabel('Predicted label')\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-22T11:50:15.315952Z","iopub.execute_input":"2024-04-22T11:50:15.316845Z","iopub.status.idle":"2024-04-22T11:52:03.564967Z","shell.execute_reply.started":"2024-04-22T11:50:15.316811Z","shell.execute_reply":"2024-04-22T11:52:03.563913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Efficientnetv2 ka lfc","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport librosa\nimport pandas as pd\nfrom efficientnet_pytorch import EfficientNet\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, f1_score, recall_score, precision_score\nfrom torchvision.models import mobilenet_v2\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader\nimport matplotlib.pyplot as plt\n\n# Constants\nsr = 48000\nlength = 10 * sr\nnum_lpc_coeffs = 10  # Number of LPC coefficients to extract\n\ndef extract_lpc_features(wav):\n    lpc_coeffs = librosa.lpc(wav, num_lpc_coeffs)\n    return lpc_coeffs\n\n# Function to preprocess LPC features\ndef preprocess_lpc_features(lpc_features):\n    # You can perform additional preprocessing steps here if needed\n    return lpc_features\n\n# Data loading and preprocessing\ndata = pd.read_csv(\"../input/rfcx-species-audio-detection/train_tp.csv\")\nX_train, X_val, y_train, y_val = train_test_split(data_list, label_list, test_size=0.1)\ntrain_data = CustomDataset(X_train, y_train, funcs=[preprocess_lpc_features])\nvalid_data = CustomDataset(X_val, y_val, funcs=[preprocess_lpc_features])\ntrain_loader = DataLoader(train_data, batch_size=32, shuffle=True)\nvalid_loader = DataLoader(valid_data, batch_size=32, shuffle=True)\n\n# Model definition\nnum_classes = 24  # Number of output classes\nnum_classes = 24  # Number of output classes\nmodel = EfficientNet.from_pretrained('efficientnet-b0', num_classes=num_classes)\nmodel = model.to(device)\n\n# Loss function and optimizer\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=0.001)\n\n# Training loop\nnum_epochs = 20\ntrain_losses = []\nval_losses = []\ntrain_accuracies = []\nval_accuracies = []\nf1_scores = []\n\nfor epoch in range(num_epochs):\n    # Training\n    model.train()\n    train_loss = 0.0\n    correct_train = 0\n    total_train = 0\n    \n    for images, labels in train_loader:\n        images = images.to(device)\n        labels = labels.to(device)\n\n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n\n        train_loss += loss.item() * images.size(0)\n        _, predicted = torch.max(outputs.data, 1)\n        total_train += labels.size(0)\n        correct_train += (predicted == labels).sum().item()\n\n    train_accuracy = correct_train / total_train\n    train_loss = train_loss / len(train_loader.dataset)\n    train_accuracies.append(train_accuracy)\n    train_losses.append(train_loss)\n\n    # Validation\n    model.eval()\n    val_loss = 0.0\n    correct_val = 0\n    total_val = 0\n    all_predictions = []\n    all_targets = []\n\n    with torch.no_grad():\n        for images, labels in valid_loader:\n            images = images.to(device)\n            labels = labels.to(device)\n\n            outputs = model(images)\n            loss = criterion(outputs, labels)\n\n            val_loss += loss.item() * images.size(0)\n            _, predicted = torch.max(outputs.data, 1)\n            total_val += labels.size(0)\n            correct_val += (predicted == labels).sum().item()\n\n            all_predictions.extend(predicted.cpu().numpy())\n            all_targets.extend(labels.cpu().numpy())\n\n    val_accuracy = correct_val / total_val\n    val_loss = val_loss / len(valid_loader.dataset)\n    val_accuracies.append(val_accuracy)\n    val_losses.append(val_loss)\n\n    # Calculate F1 score\n    f1 = f1_score(all_targets, all_predictions, average='weighted')\n    f1_scores.append(f1)\n\n    print(f\"Epoch {epoch+1}/{num_epochs} - Train Loss: {train_loss:.4f} - Train Acc: {train_accuracy:.4f} - \"\n          f\"Val Loss: {val_loss:.4f} - Val Acc: {val_accuracy:.4f} - F1 Score: {f1:.4f}\")\n\n# Plot training and validation losses\nplt.figure(figsize=(10, 5))\nplt.plot(train_losses, label='Training Loss')\nplt.plot(val_losses, label='Validation Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.title('Training and Validation Loss')\nplt.legend()\nplt.show()\n\n# Plot F1 score\nplt.figure(figsize=(10, 5))\nplt.plot(f1_scores, label='F1 Score')\nplt.xlabel('Epoch')\nplt.ylabel('F1 Score')\nplt.title('F1 Score')\nplt.legend()\nplt.show()\n\n# Confusion matrix\ncm = confusion_matrix(all_targets, all_predictions)\nplt.figure(figsize=(8, 6))\nplt.imshow(cm, interpolation='nearest', cmap=plt.cm.Blues)\nplt.title('Confusion Matrix')\nplt.colorbar()\ntick_marks = range(num_classes)\nplt.xticks(tick_marks, range(num_classes), rotation=45)\nplt.yticks(tick_marks, range(num_classes))\nplt.ylabel('True label')\nplt.xlabel('Predicted label')\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-22T11:58:30.168076Z","iopub.execute_input":"2024-04-22T11:58:30.168797Z","iopub.status.idle":"2024-04-22T12:01:25.234121Z","shell.execute_reply.started":"2024-04-22T11:58:30.168765Z","shell.execute_reply":"2024-04-22T12:01:25.233256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_lpc_features(audio_data, sr):\n    lpc_features = []\n    for recording_id, wav_data in audio_data.items():\n        lpc = librosa.core.lpc(wav_data[0], order=12)  # Adjust the order as needed\n        lpc_features.append(lpc)\n    return lpc_features\n\n# Assuming you have a LPC feature extracted and stored in a variable 'lpc_feature'\ndef visualize_lpc_feature(lpc_feature):\n    plt.figure(figsize=(10, 5))\n    lpc_feature = lpc_feature[0]  # Take the first channel\n    lpc_feature = lpc_feature.T  # Transpose to have time on x-axis\n    librosa.display.specshow(lpc_feature, sr=sr, x_axis='time', cmap='viridis')\n    plt.title('LPC Feature')\n    plt.colorbar(format='%+2.0f dB')\n    plt.tight_layout()\n    plt.show()\n\n\n# Assuming 'audio_data' contains the audio data in the format (num_channels, num_samples)\nlpc_features = extract_lpc_features(audio_data, sr)\n\nvisualize_lpc_feature(lpc_feature)\n# Function to extract LPC features from audio","metadata":{"execution":{"iopub.status.busy":"2024-04-22T11:49:54.082148Z","iopub.execute_input":"2024-04-22T11:49:54.082577Z","iopub.status.idle":"2024-04-22T11:50:05.785648Z","shell.execute_reply.started":"2024-04-22T11:49:54.082546Z","shell.execute_reply":"2024-04-22T11:50:05.784718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom torchvision.models import mobilenet_v2\nfrom torchvision import transforms\nimport torch\nimport torch.nn as nn\nfrom skimage import exposure, util\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix, f1_score, recall_score, precision_score\nimport random\nfrom torchvision.transforms import RandomRotation, RandomHorizontalFlip, RandomVerticalFlip, ColorJitter\n\ndevice = 'cuda' if torch.cuda.is_available() else 'cpu'\n\n# Load MobileNetV2 pre-trained weights\nmodelm1 = mobilenet_v2(pretrained=True)\n\n# Modify the classifier (fully connected layer) to match your problem\nmodelm1.classifier = nn.Sequential(\n    nn.Dropout(0.2),  # Add dropout if needed\n    nn.Linear(modelm1.last_channel, 24)  # Modify output to match your number of classes\n)\n\n# Move the model to the appropriate device\nmodelm1 = model.to(device)\n\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.Adam(modelm1.parameters(), lr=0.001)\n\ndef plot_confusion_matrix(y_true, y_pred, classes):\n    cm = confusion_matrix(y_true, y_pred)\n    plt.figure(figsize=(8, 6))\n    plt.imshow(cm, interpolation='nearest', cmap=plt.cm.Blues)\n    plt.title('Confusion Matrix')\n    plt.colorbar()\n    tick_marks = range(len(classes))\n    plt.xticks(tick_marks, classes, rotation=45)\n    plt.yticks(tick_marks, classes)\n\n    fmt = 'd'\n    thresh = cm.max() / 2.\n    for i in range(cm.shape[0]):\n        for j in range(cm.shape[1]):\n            plt.text(j, i, format(cm[i, j], fmt),\n                     horizontalalignment=\"center\",\n                     color=\"white\" if cm[i, j] > thresh else \"black\")\n\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    plt.tight_layout()\n\n# Function to plot training and validation loss\ndef plot_loss(train_losses, val_losses):\n    plt.figure(figsize=(10, 5))\n    plt.plot(train_losses, label='Training Loss')\n    plt.plot(val_losses, label='Validation Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.title('Training and Validation Loss')\n    plt.legend()\n    plt.show()\n\n# Function to plot F1 score and recall\ndef plot_metrics(f1_scores, recalls, precision):\n    plt.figure(figsize=(10, 5))\n    plt.plot(f1_scores, label='F1 Score')\n    plt.plot(recalls, label='Recall')\n    plt.plot(precision, label='Precision')\n    plt.xlabel('Epoch')\n    plt.ylabel('Score')\n    plt.title('F1 Score, Recall, and Precision')\n    plt.legend()\n    plt.show()\n\ndef change_gamma(img):\n    output = exposure.adjust_gamma(img, (random.randint(3, 7) / 5))\n    return output\n\ndef add_noise(img):\n    output = util.random_noise(img)\n    return output\n\nX_train, X_val, y_train, y_val = train_test_split(data_list, label_list, test_size=0.1)\ntrain_data = CustomDataset(X_train, y_train)\ntrain_data2 = CustomDataset(X_train, y_train, [change_gamma, add_noise])\nvalid_data = CustomDataset(X_val, y_val)\ntrain_loader = DataLoader(\n    torch.utils.data.ConcatDataset([train_data, train_data2]), \n    batch_size=32, shuffle=True\n)\nvalid_loader = DataLoader(valid_data, batch_size=32, shuffle=True)\ntrain_losses = []\nval_losses = []\ntrain_accuracies = []\ntest_accuracies = []\nf1_scores = []\nrecalls = []\nprecisions = []\nnum_epochs = 20\nfor epoch in range(num_epochs):\n    # Training\n    train_loss = 0.0\n    correct_train = 0\n    total_train = 0\n    for images, labels in tqdm(train_loader):\n        images = images.to(device)\n        labels = labels.to(device)\n\n        # Forward pass\n        outputs = modelm1(images.float())\n        loss = criterion(outputs, labels)\n\n        # Backward pass and optimization\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        # Track train loss and accuracy\n        train_loss += loss.item() * images.size(0)\n        _, predicted = torch.max(outputs.data, 1)\n        total_train += labels.size(0)\n        correct_train += (predicted == labels).sum().item()\n\n    # Calculate train accuracy and loss\n    train_accuracy = correct_train / total_train\n    train_loss = train_loss / len(train_loader.dataset)\n    \n    train_accuracies.append(train_accuracy)\n    train_losses.append(train_loss)\n    \n    # Evaluation (Test)\n    test_loss = 0.0\n    correct_test = 0\n    total_test = 0\n    all_predictions = []\n    all_targets = []\n    with torch.no_grad():\n        for images, labels in valid_loader:\n            images = images.to(device)\n            labels = labels.to(device)\n\n            # Forward pass\n            outputs = modelm1(images.float())\n            loss = criterion(outputs, labels)\n\n            # Track test loss and accuracy\n            test_loss += loss.item() * images.size(0)\n            _, predicted = torch.max(outputs.data, 1)\n            total_test += labels.size(0)\n            correct_test += (predicted == labels).sum().item()\n            all_predictions.extend(predicted.cpu().numpy())\n            all_targets.extend(labels.cpu().numpy())\n\n    # Calculate test accuracy and loss\n    test_accuracy = correct_test / total_test\n    test_loss = test_loss / len(valid_loader.dataset)\n    \n    test_accuracies.append(test_accuracy)\n    val_losses.append(test_loss)\n\n    # Print epoch results\n    print(f\"Epoch {epoch+1}/{num_epochs} - Train Loss: {train_loss:.4f} - Train Acc: {train_accuracy:.4f} - Test Loss: {test_loss:.4f} - Test Acc: {test_accuracy:.4f}\")\n    \n    # Calculate and print F1 score, recall, and precision\n    f1 = f1_score(all_targets, all_predictions, average='weighted')\n    recall = recall_score(all_targets, all_predictions, average='weighted', zero_division=1)\n    precision = precision_score(all_targets, all_predictions, average='weighted')\n    print(f\"F1 Score: {f1:.4f}, Recall: {recall:.4f}, Precision: {precision:.4f}\")\n    f1_scores.append(f1)\n    recalls.append(recall)\n    precisions.append(precision)\n    \n# Obtain unique class labels from the dataset\nclasses = sorted(list(set(y_train + y_val)))\n\n# Plot confusion matrix\nplot_confusion_matrix(all_targets, all_predictions, classes)\n\n# Plot training and validation loss\nplot_loss(train_losses, test_loss)\n\n# Plot F1 score, recall, and precision\nplot_metrics(f1_scores, recalls, precisions)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-22T12:29:45.212341Z","iopub.execute_input":"2024-04-22T12:29:45.212668Z","iopub.status.idle":"2024-04-22T12:35:10.435675Z","shell.execute_reply.started":"2024-04-22T12:29:45.212643Z","shell.execute_reply":"2024-04-22T12:35:10.434680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install efficientnet-pytorch\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-22T11:54:41.559369Z","iopub.execute_input":"2024-04-22T11:54:41.560101Z","iopub.status.idle":"2024-04-22T11:54:56.674900Z","shell.execute_reply.started":"2024-04-22T11:54:41.560069Z","shell.execute_reply":"2024-04-22T11:54:56.673699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom torchvision import transforms\nimport torch\nimport torch.nn as nn\nfrom skimage import exposure, util\nfrom efficientnet_pytorch import EfficientNet\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix, f1_score, recall_score, precision_score\nimport random\nfrom torchvision.transforms import RandomRotation, RandomHorizontalFlip, RandomVerticalFlip, ColorJitter\n\ndevice = 'cuda' if torch.cuda.is_available() else 'cpu'\n# Load EfficientNet-B0 pre-trained weights\nmodel_efficientnet = EfficientNet.from_pretrained('efficientnet-b0')\n\n# Modify the classifier (fully connected layer) to match your problem\nnum_ftrs = model_efficientnet._fc.in_features\nmodel_efficientnet._fc = nn.Linear(num_ftrs, 24)  # Modify output to match your number of classes\n\n# Move the model to the appropriate device\nmodele1 = model_efficientnet.to(device)\n\ncriterion = nn.CrossEntropyLoss()\n\n# Adjust learning rate\noptimizer = torch.optim.Adam(modele1.parameters(), lr=0.0001)\n\n# Apply dropout\nmodel._dropout = nn.Dropout(0.5)  # Example dropout rate of 0.5, adjust as needed\n\n# Increase data augmentation\ntrain_transform = transforms.Compose([\n    transforms.RandomResizedCrop(224),\n    RandomRotation(degrees=20),\n    RandomHorizontalFlip(),\n    RandomVerticalFlip(),\n    ColorJitter(brightness=0.2, contrast=0.2, saturation=0.2, hue=0.2),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\ndef plot_confusion_matrix(y_true, y_pred, classes):\n    cm = confusion_matrix(y_true, y_pred)\n    plt.figure(figsize=(8, 6))\n    plt.imshow(cm, interpolation='nearest', cmap=plt.cm.Blues)\n    plt.title('Confusion Matrix')\n    plt.colorbar()\n    tick_marks = range(len(classes))\n    plt.xticks(tick_marks, classes, rotation=45)\n    plt.yticks(tick_marks, classes)\n\n    fmt = 'd'\n    thresh = cm.max() / 2.\n    for i in range(cm.shape[0]):\n        for j in range(cm.shape[1]):\n            plt.text(j, i, format(cm[i, j], fmt),\n                     horizontalalignment=\"center\",\n                     color=\"white\" if cm[i, j] > thresh else \"black\")\n\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    plt.tight_layout()\n\n# Function to plot training and validation loss\ndef plot_loss(train_losses, val_losses):\n    plt.figure(figsize=(10, 5))\n    plt.plot(train_losses, label='Training Loss')\n    plt.plot(val_losses, label='Validation Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.title('Training and Validation Loss')\n    plt.legend()\n    plt.show()\n\n# Function to plot F1 score and recall\ndef plot_metrics(f1_scores, recalls, precision):\n    plt.figure(figsize=(10, 5))\n    plt.plot(f1_scores, label='F1 Score')\n    plt.plot(recalls, label='Recall')\n    plt.plot(precision, label='Precision')\n    plt.xlabel('Epoch')\n    plt.ylabel('Score')\n    plt.title('F1 Score, Recall, and Precision')\n    plt.legend()\n    plt.show()\n\ndef change_gamma(img):\n    output = exposure.adjust_gamma(img, (random.randint(3, 7) / 5))\n    return output\n\ndef add_noise(img):\n    output = util.random_noise(img)\n    return output\n\nX_train, X_val, y_train, y_val = train_test_split(data_list, label_list, test_size=0.1)\ntrain_data = CustomDataset(X_train, y_train)\ntrain_data2 = CustomDataset(X_train, y_train, [change_gamma, add_noise])\nvalid_data = CustomDataset(X_val, y_val)\ntrain_loader = DataLoader(\n    torch.utils.data.ConcatDataset([train_data, train_data2]), \n    batch_size=32, shuffle=True\n)\nvalid_loader = DataLoader(valid_data, batch_size=32, shuffle=True)\ntrain_losses = []\nval_losses = []\ntrain_accuracies = []\ntest_accuracies = []\nf1_scores = []\nrecalls = []\nprecisions = []\nnum_epochs = 20\nfor epoch in range(num_epochs):\n    # Training\n    train_loss = 0.0\n    correct_train = 0\n    total_train = 0\n    for images, labels in tqdm(train_loader):\n        images = images.to(device)\n        labels = labels.to(device)\n\n        # Forward pass\n        outputs = modele1(images.float())\n        loss = criterion(outputs, labels)\n\n        # Backward pass and optimization\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        # Track train loss and accuracy\n        train_loss += loss.item() * images.size(0)\n        _, predicted = torch.max(outputs.data, 1)\n        total_train += labels.size(0)\n        correct_train += (predicted == labels).sum().item()\n\n    # Calculate train accuracy and loss\n    train_accuracy = correct_train / total_train\n    train_loss = train_loss / len(train_loader.dataset)\n    \n    train_accuracies.append(train_accuracy)\n    train_losses.append(train_loss)\n    \n    # Evaluation (Test)\n    test_loss = 0.0\n    correct_test = 0\n    total_test = 0\n    all_predictions = []\n    all_targets = []\n    with torch.no_grad():\n        for images, labels in valid_loader:\n            images = images.to(device)\n            labels = labels.to(device)\n\n            # Forward pass\n            outputs = modele1(images.float())\n            loss = criterion(outputs, labels)\n\n            # Track test loss and accuracy\n            test_loss += loss.item() * images.size(0)\n            _, predicted = torch.max(outputs.data, 1)\n            total_test += labels.size(0)\n            correct_test += (predicted == labels).sum().item()\n            all_predictions.extend(predicted.cpu().numpy())\n            all_targets.extend(labels.cpu().numpy())\n\n    # Calculate test accuracy and loss\n    test_accuracy = correct_test / total_test\n    test_loss = test_loss / len(valid_loader.dataset)\n    \n    test_accuracies.append(test_accuracy)\n    val_losses.append(test_loss)\n\n    # Print epoch results\n    print(f\"Epoch {epoch+1}/{num_epochs} - Train Loss: {train_loss:.4f} - Train Acc: {train_accuracy:.4f} - Test Loss: {test_loss:.4f} - Test Acc: {test_accuracy:.4f}\")\n    \n    # Calculate and print F1 score, recall, and precision\n    f1 = f1_score(all_targets, all_predictions, average='weighted')\n    recall = recall_score(all_targets, all_predictions, average='weighted', zero_division=1)\n    precision = precision_score(all_targets, all_predictions, average='weighted')\n    print(f\"F1 Score: {f1:.4f}, Recall: {recall:.4f}, Precision: {precision:.4f}\")\n    f1_scores.append(f1)\n    recalls.append(recall)\n    precisions.append(precision)\n# Obtain unique class labels from the dataset\nclasses = sorted(list(set(y_train + y_val)))\n# Plot confusion matrix\nplot_confusion_matrix(all_targets, all_predictions, classes)\n\n# Plot training and validation loss\nplot_loss(train_losses, val_losses)\n\n# Plot F1 score, recall, and precision\nplot_metrics(f1_scores, recalls, precisions)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-22T12:44:33.875184Z","iopub.execute_input":"2024-04-22T12:44:33.875562Z","iopub.status.idle":"2024-04-22T12:52:09.771543Z","shell.execute_reply.started":"2024-04-22T12:44:33.875533Z","shell.execute_reply":"2024-04-22T12:52:09.770627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport librosa\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, f1_score\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader\nfrom torchvision import models\nimport matplotlib.pyplot as plt\n\n# Constants\nsr = 48000\nlength = 10 * sr\nnum_lpc_coeffs = 10  # Number of LPC coefficients to extract\n# Define the device\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Function to extract LPC features from audio\ndef extract_lpc_features(audio_data, sr):\n    lpc_features = []\n    for recording_id, wav_data in audio_data.items():\n        lpc = librosa.core.lpc(wav_data[0], order=num_lpc_coeffs)\n        lpc_features.append(lpc)\n    return lpc_features\n\n# Function to preprocess LPC features\ndef preprocess_lpc_features(lpc_features):\n    # You can perform additional preprocessing steps here if needed\n    return lpc_features\n\n# Data loading and preprocessing\ndata = pd.read_csv(\"../input/rfcx-species-audio-detection/train_tp.csv\")\nX_train, X_val, y_train, y_val = train_test_split(data_list, label_list, test_size=0.1)\ntrain_data = CustomDataset(X_train, y_train, funcs=[preprocess_lpc_features])\nvalid_data = CustomDataset(X_val, y_val, funcs=[preprocess_lpc_features])\ntrain_loader = DataLoader(train_data, batch_size=32, shuffle=True)\nvalid_loader = DataLoader(valid_data, batch_size=32, shuffle=True)\n\n# Model definition - ResNet-50\nnum_classes = 24  # Number of output classes\nresnet50 = models.resnet50(pretrained=True)\nnum_ftrs = resnet50.fc.in_features\nresnet50.fc = nn.Linear(num_ftrs, num_classes)\nresnet50 = resnet50.to(device)\n\n# Model definition - ResNet-101\nresnet101 = models.resnet101(pretrained=True)\nnum_ftrs = resnet101.fc.in_features\nresnet101.fc = nn.Linear(num_ftrs, num_classes)\nresnet101 = resnet101.to(device)\n\n# Loss function and optimizer\ncriterion = nn.CrossEntropyLoss()\noptimizer_resnet50 = torch.optim.Adam(resnet50.parameters(), lr=0.001)\noptimizer_resnet101 = torch.optim.Adam(resnet101.parameters(), lr=0.001)\n\n# Training loop\nnum_epochs = 20\ntrain_losses_resnet50 = []\nval_losses_resnet50 = []\ntrain_accuracies_resnet50 = []\nval_accuracies_resnet50 = []\nf1_scores_resnet50 = []\n\ntrain_losses_resnet101 = []\nval_losses_resnet101 = []\ntrain_accuracies_resnet101 = []\nval_accuracies_resnet101 = []\nf1_scores_resnet101 = []\n\nfor epoch in range(num_epochs):\n    # Training - ResNet-50\n    resnet50.train()\n    train_loss_resnet50 = 0.0\n    correct_train_resnet50 = 0\n    total_train_resnet50 = 0\n    \n    for images, labels in train_loader:\n        images = images.to(device)\n        labels = labels.to(device)\n\n        optimizer_resnet50.zero_grad()\n        outputs = resnet50(images)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer_resnet50.step()\n\n        train_loss_resnet50 += loss.item() * images.size(0)\n        _, predicted = torch.max(outputs.data, 1)\n        total_train_resnet50 += labels.size(0)\n        correct_train_resnet50 += (predicted == labels).sum().item()\n\n    train_accuracy_resnet50 = correct_train_resnet50 / total_train_resnet50\n    train_loss_resnet50 = train_loss_resnet50 / len(train_loader.dataset)\n    train_accuracies_resnet50.append(train_accuracy_resnet50)\n    train_losses_resnet50.append(train_loss_resnet50)\n\n    # Validation - ResNet-50\n    resnet50.eval()\n    val_loss_resnet50 = 0.0\n    correct_val_resnet50 = 0\n    total_val_resnet50 = 0\n    all_predictions_resnet50 = []\n    all_targets_resnet50 = []\n\n    with torch.no_grad():\n        for images, labels in valid_loader:\n            images = images.to(device)\n            labels = labels.to(device)\n\n            outputs = resnet50(images)\n            loss = criterion(outputs, labels)\n\n            val_loss_resnet50 += loss.item() * images.size(0)\n            _, predicted = torch.max(outputs.data, 1)\n            total_val_resnet50 += labels.size(0)\n            correct_val_resnet50 += (predicted == labels).sum().item()\n\n            all_predictions_resnet50.extend(predicted.cpu().numpy())\n            all_targets_resnet50.extend(labels.cpu().numpy())\n\n    val_accuracy_resnet50 = correct_val_resnet50 / total_val_resnet50\n    val_loss_resnet50 = val_loss_resnet50 / len(valid_loader.dataset)\n    val_accuracies_resnet50.append(val_accuracy_resnet50)\n    val_losses_resnet50.append(val_loss_resnet50)\n\n    # Calculate F1 score - ResNet-50\n    f1_resnet50 = f1_score(all_targets_resnet50, all_predictions_resnet50, average='weighted')\n    f1_scores_resnet50.append(f1_resnet50)\n\n    # Training - ResNet-101\n    resnet101.train()\n    train_loss_resnet101 = 0.0\n    correct_train_resnet101 = 0\n    total_train_resnet101 = 0\n    \n    for images, labels in train_loader:\n        images = images.to(device)\n        labels = labels.to(device)\n\n        optimizer_resnet101.zero_grad()\n        outputs = resnet101(images)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer_resnet101.step()\n\n        train_loss_resnet101 += loss.item() * images.size(0)\n        _, predicted = torch.max(outputs.data, 1)\n        total_train_resnet101 += labels.size(0)\n        correct_train_resnet101 += (predicted == labels).sum().item()\n\n    train_accuracy_resnet101 = correct_train_resnet101 / total_train_resnet101\n    train_loss_resnet101 = train_loss_resnet101 / len(train_loader.dataset)\n    train_accuracies_resnet101.append(train_accuracy_resnet101)\n    train_losses_resnet101.append(train_loss_resnet101)\n\n    # Validation - ResNet-101\n    resnet101.eval()\n    val_loss_resnet101 = 0.0\n    correct_val_resnet101 = 0\n    total_val_resnet101 = 0\n    all_predictions_resnet101 = []\n    all_targets_resnet101 = []\n\n    with torch.no_grad():\n        for images, labels in valid_loader:\n            images = images.to(device)\n            labels = labels.to(device)\n\n            outputs = resnet101(images)\n            loss = criterion(outputs, labels)\n\n            val_loss_resnet101 += loss.item() * images.size(0)\n            _, predicted = torch.max(outputs.data, 1)\n            total_val_resnet101 += labels.size(0)\n            correct_val_resnet101 += (predicted == labels).sum().item()\n\n            all_predictions_resnet101.extend(predicted.cpu().numpy())\n            all_targets_resnet101.extend(labels.cpu().numpy())\n\n    val_accuracy_resnet101 = correct_val_resnet101 / total_val_resnet101\n    val_loss_resnet101 = val_loss_resnet101 / len(valid_loader.dataset)\n    val_accuracies_resnet101.append(val_accuracy_resnet101)\n    val_losses_resnet101.append(val_loss_resnet101)\n\n    # Calculate F1 score - ResNet-101\n    f1_resnet101 = f1_score(all_targets_resnet101, all_predictions_resnet101, average='weighted')\n    f1_scores_resnet101.append(f1_resnet101)\n\n    print(f\"Epoch {epoch+1}/{num_epochs}\")\n    print(f\"ResNet-50: Train Loss: {train_loss_resnet50:.4f}, Train Acc: {train_accuracy_resnet50:.4f}, \"\n          f\"Val Loss: {val_loss_resnet50:.4f}, Val Acc: {val_accuracy_resnet50:.4f}, F1 Score: {f1_resnet50:.4f}\")\n    print(f\"ResNet-101: Train Loss: {train_loss_resnet101:.4f}, Train Acc: {train_accuracy_resnet101:.4f}, \"\n          f\"Val Loss: {val_loss_resnet101:.4f}, Val Acc: {val_accuracy_resnet101:.4f}, F1 Score: {f1_resnet101:.4f}\")\n\n# Plot training and validation losses\nplt.figure(figsize=(10, 5))\nplt.plot(train_losses_resnet50, label='ResNet-50 Training Loss')\nplt.plot(val_losses_resnet50, label='ResNet-50 Validation Loss')\nplt.plot(train_losses_resnet101, label='ResNet-101 Training Loss')\nplt.plot(val_losses_resnet101, label='ResNet-101 Validation Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.title('Training and Validation Loss')\nplt.legend()\nplt.show()\n\n# Plot F1 score\nplt.figure(figsize=(10, 5))\nplt.plot(f1_scores_resnet50, label='ResNet-50 F1 Score')\nplt.plot(f1_scores_resnet101, label='ResNet-101 F1 Score')\nplt.xlabel('Epoch')\nplt.ylabel('F1 Score')\nplt.title('F1 Score')\nplt.legend()\nplt.show()\n\n# Confusion matrix\ncm_resnet50 = confusion_matrix(all_targets_resnet50, all_predictions_resnet50)\ncm_resnet101 = confusion_matrix(all_targets_resnet101, all_predictions_resnet101)\n\nplt.figure(figsize=(16, 6))\nplt.subplot(1, 2, 1)\nplt.imshow(cm_resnet50, interpolation='nearest', cmap=plt.cm.Blues)\nplt.title('ResNet-50 Confusion Matrix')\nplt.colorbar()\nplt.xlabel('Predicted label')\nplt.ylabel('True label')\nplt.subplot(1, 2, 2)\nplt.imshow(cm_resnet101, interpolation='nearest', cmap=plt.cm.Blues)\nplt.title('ResNet-101 Confusion Matrix')\nplt.colorbar()\nplt.xlabel('Predicted label')\nplt.ylabel('True label')\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-22T14:47:34.781766Z","iopub.execute_input":"2024-04-22T14:47:34.782440Z","iopub.status.idle":"2024-04-22T14:56:43.005815Z","shell.execute_reply.started":"2024-04-22T14:47:34.782406Z","shell.execute_reply":"2024-04-22T14:56:43.004784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom skimage import exposure, util\nfrom resnest.torch import resnest101\nimport torch\nimport torch.nn as nn\nfrom torchvision import transforms\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix, f1_score, recall_score, precision_score\nimport random\nfrom torchvision.transforms import RandomRotation, RandomHorizontalFlip, RandomVerticalFlip, ColorJitter\n\ndevice = 'cuda' if torch.cuda.is_available() else 'cpu'\nweights = torch.load('../input/resnest-package/resnest101-22405ba7.pth', map_location=device)\nmodel = resnest101()\nmodel.load_state_dict(weights)\nmodel.fc = nn.Linear(model.fc.in_features, 24)\nmodelr = model.to(device)\n\ncriterion = nn.CrossEntropyLoss()\n\n# Adjust learning rate\noptimizer = torch.optim.Adam(modelr.parameters(), lr=0.0001)\n\n# Apply dropout\nmodel._dropout = nn.Dropout(0.5)  # Example dropout rate of 0.5, adjust as needed\n\n# Increase data augmentation\ntrain_transform = transforms.Compose([\n    transforms.RandomResizedCrop(224),\n    RandomRotation(degrees=20),\n    RandomHorizontalFlip(),\n    RandomVerticalFlip(),\n    ColorJitter(brightness=0.2, contrast=0.2, saturation=0.2, hue=0.2),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\ndef plot_confusion_matrix(y_true, y_pred, classes):\n    cm = confusion_matrix(y_true, y_pred)\n    plt.figure(figsize=(8, 6))\n    plt.imshow(cm, interpolation='nearest', cmap=plt.cm.Blues)\n    plt.title('Confusion Matrix')\n    plt.colorbar()\n    tick_marks = range(len(classes))\n    plt.xticks(tick_marks, classes, rotation=45)\n    plt.yticks(tick_marks, classes)\n\n    fmt = 'd'\n    thresh = cm.max() / 2.\n    for i in range(cm.shape[0]):\n        for j in range(cm.shape[1]):\n            plt.text(j, i, format(cm[i, j], fmt),\n                     horizontalalignment=\"center\",\n                     color=\"white\" if cm[i, j] > thresh else \"black\")\n\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    plt.tight_layout()\n\n# Function to plot training and validation loss\ndef plot_loss(train_losses, val_losses):\n    plt.figure(figsize=(10, 5))\n    plt.plot(train_losses, label='Training Loss')\n    plt.plot(val_losses, label='Validation Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.title('Training and Validation Loss')\n    plt.legend()\n    plt.show()\n\n# Function to plot F1 score and recall\ndef plot_metrics(f1_scores, recalls, precision):\n    plt.figure(figsize=(10, 5))\n    plt.plot(f1_scores, label='F1 Score')\n    plt.plot(recalls, label='Recall')\n    plt.plot(precision, label='Precision')\n    plt.xlabel('Epoch')\n    plt.ylabel('Score')\n    plt.title('F1 Score, Recall, and Precision')\n    plt.legend()\n    plt.show()\n\ndef change_gamma(img):\n    output = exposure.adjust_gamma(img, (random.randint(3, 7) / 5))\n    return output\n\ndef add_noise(img):\n    output = util.random_noise(img)\n    return output\n\nX_train, X_val, y_train, y_val = train_test_split(data_list, label_list, test_size=0.1)\ntrain_data = CustomDataset(X_train, y_train)\ntrain_data2 = CustomDataset(X_train, y_train, [change_gamma, add_noise])\nvalid_data = CustomDataset(X_val, y_val)\ntrain_loader = DataLoader(\n    torch.utils.data.ConcatDataset([train_data, train_data2]), \n    batch_size=32, shuffle=True\n)\nvalid_loader = DataLoader(valid_data, batch_size=32, shuffle=True)\ntrain_losses = []\nval_losses = []\ntrain_accuracies = []\ntest_accuracies = []\nf1_scores = []\nrecalls = []\nprecisions = []\nnum_epochs = 20\nfor epoch in range(num_epochs):\n    # Training\n    train_loss = 0.0\n    correct_train = 0\n    total_train = 0\n    for images, labels in tqdm(train_loader):\n        images = images.to(device)\n        labels = labels.to(device)\n\n        # Forward pass\n        outputs = modelr(images.float())\n        loss = criterion(outputs, labels)\n\n        # Backward pass and optimization\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        # Track train loss and accuracy\n        train_loss += loss.item() * images.size(0)\n        _, predicted = torch.max(outputs.data, 1)\n        total_train += labels.size(0)\n        correct_train += (predicted == labels).sum().item()\n\n    # Calculate train accuracy and loss\n    train_accuracy = correct_train / total_train\n    train_loss = train_loss / len(train_loader.dataset)\n    \n    train_accuracies.append(train_accuracy)\n    train_losses.append(train_loss)\n    \n    # Evaluation (Test)\n    test_loss = 0.0\n    correct_test = 0\n    total_test = 0\n    all_predictions = []\n    all_targets = []\n    with torch.no_grad():\n        for images, labels in valid_loader:\n            images = images.to(device)\n            labels = labels.to(device)\n\n            # Forward pass\n            outputs = modelr(images.float())\n            loss = criterion(outputs, labels)\n\n            # Track test loss and accuracy\n            test_loss += loss.item() * images.size(0)\n            _, predicted = torch.max(outputs.data, 1)\n            total_test += labels.size(0)\n            correct_test += (predicted == labels).sum().item()\n            all_predictions.extend(predicted.cpu().numpy())\n            all_targets.extend(labels.cpu().numpy())\n\n    # Calculate test accuracy and loss\n    test_accuracy = correct_test / total_test\n    test_loss = test_loss / len(valid_loader.dataset)\n    \n    test_accuracies.append(test_accuracy)\n    val_losses.append(test_loss)\n\n    # Print epoch results\n    print(f\"Epoch {epoch+1}/{num_epochs} - Train Loss: {train_loss:.4f} - Train Acc: {train_accuracy:.4f} - Test Loss: {test_loss:.4f} - Test Acc: {test_accuracy:.4f}\")\n    \n    # Calculate and print F1 score, recall, and precision\n    f1 = f1_score(all_targets, all_predictions, average='weighted')\n    recall = recall_score(all_targets, all_predictions, average='weighted', zero_division=1)\n    precision = precision_score(all_targets, all_predictions, average='weighted')\n    print(f\"F1 Score: {f1:.4f}, Recall: {recall:.4f}, Precision: {precision:.4f}\")\n    f1_scores.append(f1)\n    recalls.append(recall)\n    precisions.append(precision)\n# Obtain unique class labels from the dataset\nclasses = sorted(list(set(y_train + y_val)))\n# Plot confusion matrix\nplot_confusion_matrix(all_targets, all_predictions, classes)\n\n# Plot training and validation loss\nplot_loss(train_losses, val_losses)\n\n# Plot F1 score, recall, and precision\nplot_metrics(f1_scores, recalls, precisions)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-22T15:01:43.063805Z","iopub.execute_input":"2024-04-22T15:01:43.064194Z","iopub.status.idle":"2024-04-22T15:26:27.386978Z","shell.execute_reply.started":"2024-04-22T15:01:43.064164Z","shell.execute_reply":"2024-04-22T15:26:27.386066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\n\n# Assuming you have a trained model object named \"model\"\n# and you want to save it to a file named \"model.pkl\"\n\n# Save the model to a pickle file\nwith open('model.pkl', 'wb') as file:\n    pickle.dump(model, file)","metadata":{"execution":{"iopub.status.busy":"2024-04-22T08:24:00.374867Z","iopub.status.idle":"2024-04-22T08:24:00.375381Z","shell.execute_reply.started":"2024-04-22T08:24:00.375061Z","shell.execute_reply":"2024-04-22T08:24:00.375079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_test_file(f):\n    wav, sr = librosa.load('/kaggle/input/rfcx-species-audio-detection/test/' + f, sr=None)\n    \n    segments = len(wav) / length\n    segments = int(np.ceil(segments))\n    \n    mel_array = []\n    \n    for i in range(0, segments):\n        if (i + 1) * length > len(wav):\n            slice = wav[len(wav) - length:len(wav)]\n        else:\n            slice = wav[i * length:(i + 1) * length]\n        \n        spec = librosa.feature.melspectrogram(y=slice, sr=sr,fmin=fmin,fmax=fmax)\n        spec_db = librosa.power_to_db(spec,top_db=80)\n\n        img = normalize(colorize(resize_image(spec_db)))\n        mel_spec = img\n        mel_array.append(mel_spec)\n    \n    return mel_array","metadata":{"execution":{"iopub.status.busy":"2024-04-22T08:24:00.376713Z","iopub.status.idle":"2024-04-22T08:24:00.377158Z","shell.execute_reply.started":"2024-04-22T08:24:00.376908Z","shell.execute_reply":"2024-04-22T08:24:00.376926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn.functional as nnf\nimport csv\nimport os\n\nmodel.eval()\nwith open('submission.csv', 'w', newline='') as csvfile:\n    submission_writer = csv.writer(csvfile, delimiter=',')\n    submission_writer.writerow(['recording_id','s0','s1','s2','s3','s4','s5','s6','s7','s8','s9','s10','s11',\n                               's12','s13','s14','s15','s16','s17','s18','s19','s20','s21','s22','s23'])\n    \n    test_files = os.listdir('/kaggle/input/rfcx-species-audio-detection/test/')\n\n    for i in range(0, len(test_files)):\n        data = load_test_file(test_files[i])\n        data = torch.tensor(data)\n        data = data.float()\n        if torch.cuda.is_available():\n            data = data.cuda()\n        with torch.no_grad():\n            output = model(data)\n            output = nnf.softmax(output, dim=1)\n        maxed_output = torch.max(output, dim=0)[0]\n        maxed_output = maxed_output.cpu().detach()\n        file_id = str.split(test_files[i], '.')[0]\n        write_array = [file_id]\n        \n        for out in maxed_output:\n            write_array.append(out.item())\n    \n        submission_writer.writerow(write_array)\n        \n        \n        if i % 100 == 0 and i > 0:\n            print(str(i) + '/' + str(len(test_files)))\n\nprint('Done')","metadata":{"execution":{"iopub.status.busy":"2024-04-22T08:24:00.378485Z","iopub.status.idle":"2024-04-22T08:24:00.378881Z","shell.execute_reply.started":"2024-04-22T08:24:00.378706Z","shell.execute_reply":"2024-04-22T08:24:00.378723Z"},"trusted":true},"execution_count":null,"outputs":[]}]}