{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8539162,"sourceType":"datasetVersion","datasetId":4870988}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nimport warnings\nwarnings.filterwarnings(action='ignore')\n\nimport pandas as pd\nimport librosa\nimport numpy as np\n\n\nfrom sklearn.utils import shuffle\nfrom PIL import Image\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2024-05-28T14:03:47.771764Z","iopub.execute_input":"2024-05-28T14:03:47.772149Z","iopub.status.idle":"2024-05-28T14:04:03.455714Z","shell.execute_reply.started":"2024-05-28T14:03:47.772117Z","shell.execute_reply":"2024-05-28T14:04:03.454703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Weights and Biases setup","metadata":{}},{"cell_type":"code","source":"import wandb\nfrom wandb.keras import WandbCallback\n\nwandb.login()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T14:04:03.457812Z","iopub.execute_input":"2024-05-28T14:04:03.458550Z","iopub.status.idle":"2024-05-28T14:06:11.225514Z","shell.execute_reply.started":"2024-05-28T14:04:03.458512Z","shell.execute_reply":"2024-05-28T14:06:11.224634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RANDOM_SEED = 1337\nSAMPLE_RATE = 32000\nSPEC_SHAPE = (48, 128) # height x width\nFMIN = 500\nFMAX = 12500","metadata":{"execution":{"iopub.status.busy":"2024-05-28T14:07:41.875703Z","iopub.execute_input":"2024-05-28T14:07:41.876355Z","iopub.status.idle":"2024-05-28T14:07:41.881767Z","shell.execute_reply.started":"2024-05-28T14:07:41.876322Z","shell.execute_reply":"2024-05-28T14:07:41.880627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data=pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T14:06:17.546281Z","iopub.execute_input":"2024-05-28T14:06:17.546977Z","iopub.status.idle":"2024-05-28T14:06:17.738792Z","shell.execute_reply.started":"2024-05-28T14:06:17.546945Z","shell.execute_reply":"2024-05-28T14:06:17.737875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = data.query('rating>=3')\nbirds_count = {}\nfor bird_species, count in zip(data.primary_label.unique(), \n                               data.groupby('primary_label')['primary_label'].count().values):\n    birds_count[bird_species] = count\nmost_represented_birds = [key for key,value in birds_count.items() if value >= 200] \n\nTRAIN = data.query('primary_label in @most_represented_birds')\nLABELS = sorted(TRAIN.primary_label.unique())\n\n# Let's see how many species and samples we have left\nprint('NUMBER OF SPECIES IN TRAIN DATA:', len(LABELS))\nprint('NUMBER OF SAMPLES IN TRAIN DATA:', len(TRAIN))\nprint('LABELS:', most_represented_birds)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T14:06:20.171046Z","iopub.execute_input":"2024-05-28T14:06:20.171371Z","iopub.status.idle":"2024-05-28T14:06:20.211387Z","shell.execute_reply.started":"2024-05-28T14:06:20.171348Z","shell.execute_reply":"2024-05-28T14:06:20.210452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom tqdm import tqdm\nfrom sklearn.utils import shuffle\n\ndef load_spectrograms(input_dir, TRAIN, LABELS, RANDOM_SEED):\n    samples = []\n\n    with tqdm(total=len(TRAIN)) as pbar:\n        for idx, row in TRAIN.iterrows():\n            pbar.update(1)\n            \n            spectrogram_filename = os.path.splitext(os.path.basename(row.filename))[0]\n                \n            for class_name in os.listdir(input_dir):\n                    \n                if class_name in LABELS:\n                    class_dir = os.path.join(input_dir, class_name)\n                    for filename in os.listdir(class_dir):\n                        sample_base = filename.split('_')[0] \n                        if sample_base == spectrogram_filename:\n                            full_path = os.path.join(class_dir, filename)\n                            samples.append(full_path)\n\n    TRAIN_SPECS = shuffle(samples, random_state=RANDOM_SEED)\n    print('SUCCESSFULLY LOADED {} SPECTROGRAMS'.format(len(TRAIN_SPECS)))\n    \n    return TRAIN_SPECS\n\ndef process_train_data(TRAIN_SPECS,LABELS):\n    train_specs, train_labels = [], []\n    with tqdm(total=len(TRAIN_SPECS)) as pbar:\n        for path in TRAIN_SPECS:\n            pbar.update(1)\n\n            # Open image\n            spec = Image.open(path)\n            spec = spec.resize((128, 48))\n            # Convert to numpy array\n            spec = np.array(spec, dtype='float32')\n        \n            # Normalize between 0.0 and 1.0\n            # and exclude samples with nan \n            spec -= spec.min()\n            spec /= spec.max()\n            if not spec.max() == 1.0 or not spec.min() == 0.0:\n                continue\n\n            # Add channel axis to 2D array\n            spec = np.expand_dims(spec, -1)\n\n            # Add new dimension for batch size\n            spec = np.expand_dims(spec, 0)\n\n            # Add to train data\n            if len(train_specs) == 0:\n                train_specs = spec\n            else:\n                train_specs = np.vstack((train_specs, spec))\n\n            # Add to label data\n            target = np.zeros((len(LABELS)), dtype='float32')\n            bird = path.split(os.sep)[-2]\n            target[LABELS.index(bird)] = 1.0\n            if len(train_labels) == 0:\n                train_labels = target\n            else:\n                train_labels = np.vstack((train_labels, target))\n    \n    return train_specs,train_labels","metadata":{"execution":{"iopub.status.busy":"2024-05-28T09:34:40.924366Z","iopub.execute_input":"2024-05-28T09:34:40.924748Z","iopub.status.idle":"2024-05-28T09:34:40.943967Z","shell.execute_reply.started":"2024-05-28T09:34:40.924720Z","shell.execute_reply":"2024-05-28T09:34:40.942787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Loading features into CSV Files**","metadata":{}},{"cell_type":"code","source":"input_dir='/kaggle/input/birdclef-2024-5s-spectrogram-features/features/percussive_training'\nTRAIN_SPECS=load_spectrograms(input_dir, TRAIN, LABELS, RANDOM_SEED)\nlen(TRAIN_SPECS)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T09:34:43.704942Z","iopub.execute_input":"2024-05-28T09:34:43.705698Z","iopub.status.idle":"2024-05-28T09:42:25.196133Z","shell.execute_reply.started":"2024-05-28T09:34:43.705647Z","shell.execute_reply":"2024-05-28T09:42:25.195222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_specs,train_labels=process_train_data(TRAIN_SPECS,LABELS)\ntrain_specs = train_specs[:, :, :, 0]","metadata":{"execution":{"iopub.status.busy":"2024-05-28T09:42:25.197971Z","iopub.execute_input":"2024-05-28T09:42:25.199037Z","iopub.status.idle":"2024-05-28T10:05:34.928871Z","shell.execute_reply.started":"2024-05-28T09:42:25.199003Z","shell.execute_reply":"2024-05-28T10:05:34.927156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"train_specs shape:\", train_specs.shape)\nprint(\"train_labels shape:\", train_labels.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:05:34.930927Z","iopub.execute_input":"2024-05-28T10:05:34.931610Z","iopub.status.idle":"2024-05-28T10:05:34.939555Z","shell.execute_reply.started":"2024-05-28T10:05:34.931575Z","shell.execute_reply":"2024-05-28T10:05:34.937814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15, 7))\nfor i in range(12):\n    spec = Image.open(TRAIN_SPECS[i])\n    spec = spec.resize((128, 48))\n    plt.subplot(3, 4, i + 1)\n    plt.title(TRAIN_SPECS[i].split(os.sep)[-1])\n    plt.imshow(spec, origin='lower')","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:05:34.942045Z","iopub.execute_input":"2024-05-28T10:05:34.942480Z","iopub.status.idle":"2024-05-28T10:05:37.921091Z","shell.execute_reply.started":"2024-05-28T10:05:34.942445Z","shell.execute_reply":"2024-05-28T10:05:37.920013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert numpy arrays to DataFrame\nspecs_df = pd.DataFrame(train_specs.reshape(train_specs.shape[0], -1))\nlabels_df = pd.DataFrame(train_labels, columns=LABELS)\n\n# Merge DataFrames on 'train_id' (assuming 'train_id' exists in your data)\nmerged_df = pd.concat([specs_df, labels_df], axis=1)\n\n# Save to CSV\noutput_dir = '/kaggle/working/'\noutput_file = os.path.join(output_dir, 'percussion_features_36.csv')\nmerged_df.to_csv(output_file, index=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T10:05:37.922441Z","iopub.execute_input":"2024-05-28T10:05:37.923006Z","iopub.status.idle":"2024-05-28T10:07:18.642123Z","shell.execute_reply.started":"2024-05-28T10:05:37.922972Z","shell.execute_reply":"2024-05-28T10:07:18.640697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Training Basic CNN**","metadata":{}},{"cell_type":"code","source":"wandb.init(\n    # set the wandb project where this run will be logged\n    project=\"perc_cnn_36_30ep_adam\",\n\n    # track hyperparameters and run metadata\n    config={\n    \"learning_rate\": 0.001,\n    \"architecture\": \"CNN\",\n    \"dataset\": \"BIRDCLEF2024\",\n    \"epochs\": 30,\n    }\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T14:19:33.094297Z","iopub.execute_input":"2024-05-28T14:19:33.094561Z","iopub.status.idle":"2024-05-28T14:19:52.851374Z","shell.execute_reply.started":"2024-05-28T14:19:33.094537Z","shell.execute_reply":"2024-05-28T14:19:52.850366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load csv\ninp_dir = \"/kaggle/input/birdclef-2024-5s-spectrogram-features/\"\ndata = pd.read_csv(inp_dir+\"percussion_features_36.csv\")\ndata","metadata":{"execution":{"iopub.status.busy":"2024-05-28T14:20:03.819971Z","iopub.execute_input":"2024-05-28T14:20:03.820344Z","iopub.status.idle":"2024-05-28T14:20:22.070709Z","shell.execute_reply.started":"2024-05-28T14:20:03.820314Z","shell.execute_reply":"2024-05-28T14:20:22.069670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = data.iloc[:, :-36]  # All columns except the last 25 columns\ny = data.iloc[:, -36:]  # Last 25 columns","metadata":{"execution":{"iopub.status.busy":"2024-05-28T14:20:54.993064Z","iopub.execute_input":"2024-05-28T14:20:54.993895Z","iopub.status.idle":"2024-05-28T14:20:54.998545Z","shell.execute_reply.started":"2024-05-28T14:20:54.993852Z","shell.execute_reply":"2024-05-28T14:20:54.997566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = X.values\nX = X.reshape(-1, SPEC_SHAPE[0], SPEC_SHAPE[1], 1)\nprint(\"New shape of X:\", X.shape)\n\ny = y.values\ny = y.reshape(-1, 36)  \nprint(\"New shape of y_train_reshaped:\", y.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T14:20:56.397174Z","iopub.execute_input":"2024-05-28T14:20:56.397512Z","iopub.status.idle":"2024-05-28T14:20:56.404761Z","shell.execute_reply.started":"2024-05-28T14:20:56.397485Z","shell.execute_reply":"2024-05-28T14:20:56.403806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Print the shape of the resulting subsets to verify the split\nprint(\"Shape of X_train:\", X_train.shape)\nprint(\"Shape of X_test:\", X_test.shape)\nprint(\"Shape of y_train:\", y_train.shape)\nprint(\"Shape of y_test:\", y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T14:20:58.490515Z","iopub.execute_input":"2024-05-28T14:20:58.491326Z","iopub.status.idle":"2024-05-28T14:20:59.154210Z","shell.execute_reply.started":"2024-05-28T14:20:58.491292Z","shell.execute_reply":"2024-05-28T14:20:59.153286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.random.set_seed(RANDOM_SEED)\n\n\nmodel = tf.keras.Sequential([\n    \n    # First conv block\n    tf.keras.layers.Conv2D(16, (3, 3), activation='relu', \n                           input_shape=(SPEC_SHAPE[0], SPEC_SHAPE[1], 1)),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n    \n    # Second conv block\n    tf.keras.layers.Conv2D(32, (3, 3), activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)), \n    \n\n    # Third conv block\n    tf.keras.layers.Conv2D(64, (3, 3), activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)), \n    \n    # Fourth conv block\n    tf.keras.layers.Conv2D(128, (3, 3), activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n    \n    # Global pooling instead of flatten()\n    tf.keras.layers.GlobalAveragePooling2D(), \n    \n    # Dense block\n    tf.keras.layers.Dense(256, activation='relu'),   \n    tf.keras.layers.Dropout(0.5),  \n    tf.keras.layers.Dense(256, activation='relu'),   \n    tf.keras.layers.Dropout(0.5),\n    \n    # Classification layer\n    tf.keras.layers.Dense(len(LABELS), activation='softmax')\n])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T14:21:01.303295Z","iopub.execute_input":"2024-05-28T14:21:01.303651Z","iopub.status.idle":"2024-05-28T14:21:01.476255Z","shell.execute_reply.started":"2024-05-28T14:21:01.303623Z","shell.execute_reply":"2024-05-28T14:21:01.475424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n              loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.01),\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-05-28T14:21:16.366319Z","iopub.execute_input":"2024-05-28T14:21:16.367182Z","iopub.status.idle":"2024-05-28T14:21:16.381369Z","shell.execute_reply.started":"2024-05-28T14:21:16.367151Z","shell.execute_reply":"2024-05-28T14:21:16.380623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import Callback\n\nclass WandbCallback(Callback):\n    def on_epoch_end(self, epoch, logs=None):\n        wandb.log(logs)\n        \ncallbacks = [tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', \n                                                  patience=2, \n                                                  verbose=1, \n                                                  factor=0.5),\n             tf.keras.callbacks.EarlyStopping(monitor='val_loss', \n                                              verbose=1,\n                                              patience=5),\n             tf.keras.callbacks.ModelCheckpoint(filepath='perc_36_30ep_adam.keras',  # Change filepath here\n                                                monitor='val_loss',\n                                                verbose=0,\n                                                save_best_only=True),\n            WandbCallback()],\n            ","metadata":{"execution":{"iopub.status.busy":"2024-05-28T14:21:18.183205Z","iopub.execute_input":"2024-05-28T14:21:18.183857Z","iopub.status.idle":"2024-05-28T14:21:18.190692Z","shell.execute_reply.started":"2024-05-28T14:21:18.183823Z","shell.execute_reply":"2024-05-28T14:21:18.189634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(X_train, y_train,  \n                    batch_size=32,\n                    validation_split=0.2,\n                    callbacks=callbacks,\n                    epochs=30)\nwandb.save(\"perc_36_30ep_adam.keras\")","metadata":{"execution":{"iopub.status.busy":"2024-05-28T14:21:19.749398Z","iopub.execute_input":"2024-05-28T14:21:19.749740Z","iopub.status.idle":"2024-05-28T14:22:03.489089Z","shell.execute_reply.started":"2024-05-28T14:21:19.749711Z","shell.execute_reply":"2024-05-28T14:22:03.487921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions on the test data\ny_pred = model.predict(X_test)\n\n# Print the shape of the predictions\nprint(\"Shape of y_pred:\", y_pred.shape)\n\n# If you want to convert the predicted probabilities to class labels (assuming y_pred is one-hot encoded):\npredicted_labels = np.argmax(y_pred, axis=1)\n\n# Print the first few predicted labels\nprint(\"Predicted labels:\", predicted_labels)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T14:22:03.493074Z","iopub.execute_input":"2024-05-28T14:22:03.493373Z","iopub.status.idle":"2024-05-28T14:22:04.830076Z","shell.execute_reply.started":"2024-05-28T14:22:03.493348Z","shell.execute_reply":"2024-05-28T14:22:04.829065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, classification_report\ntrue_labels = np.argmax(y_test, axis=1)\n\n# Compute accuracy\naccuracy = accuracy_score(true_labels, predicted_labels)\nprint(\"Accuracy:\", accuracy)\n\n# Print classification report\nprint(\"Classification Report:\")\nprint(classification_report(true_labels, predicted_labels))","metadata":{"execution":{"iopub.status.busy":"2024-05-28T14:22:04.831457Z","iopub.execute_input":"2024-05-28T14:22:04.832284Z","iopub.status.idle":"2024-05-28T14:22:04.852536Z","shell.execute_reply.started":"2024-05-28T14:22:04.832247Z","shell.execute_reply":"2024-05-28T14:22:04.851668Z"},"trusted":true},"execution_count":null,"outputs":[]}]}