{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8089287,"sourceType":"datasetVersion","datasetId":4775532}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"https://www.kaggle.com/code/adhok93/saving-data-in-numpy-format/notebook?scriptVersionId=171476154","metadata":{}},{"cell_type":"code","source":"import librosa\nimport numpy as np\nimport os\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.applications import MobileNet\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Dense\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.utils import to_categorical\nfrom tqdm import tqdm\nimport json\n\n# Constants\nSAMPLE_RATE = 22050\nN_MELS = 128\nHOP_LENGTH = 512\nN_FFT = 2048\n#INPUT_SHAPE = (128, 128, 3)  # Expected by MobileNet\nCLASSES = 182  # Adjust this to your number of classes\n\nfrom tensorflow.keras.callbacks import ModelCheckpoint\n\n\ncheckpoint = ModelCheckpoint(\n    'best_model.keras',  # Model filename\n    monitor='val_accuracy',  # Metric to monitor\n    verbose=1,  # Log level\n    save_best_only=True,  # Save only the best model\n    mode='max'  # Mode for the monitored quantity (maximize accuracy)\n)\n\ndef lr_schedule(epoch, lr):\n    # Reduce the learning rate by a factor of 0.1 every 10 epochs\n    if epoch % 10 == 0 and epoch > 0:\n        lr *= 0.1\n    return lr\n\n\nfrom tensorflow.keras.callbacks import LearningRateScheduler\n\nlr_scheduler = LearningRateScheduler(lr_schedule, verbose=1)\n\n# def extract_features(audio_path):\n#     # Load only the first 5 seconds of the audio file\n#     y, sr = librosa.load(audio_path, sr=SAMPLE_RATE, duration=10)\n#     mel_spec = librosa.feature.melspectrogram(y=y, sr=sr, n_mels=N_MELS, n_fft=N_FFT, hop_length=HOP_LENGTH)\n#     mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n    \n\n#     # Ensure mel_spec_db has a uniform shape, e.g., (128, 216)\n#     expected_width = 431  # This should match the width after resizing for 5 seconds\n#     actual_width = mel_spec_db.shape[1]\n\n#     if actual_width < expected_width:\n#         # Pad the spectrogram with zeros if it's shorter than expected\n#         padding_width = expected_width - actual_width\n#         mel_spec_db = np.pad(mel_spec_db, ((0, 0), (0, padding_width)), mode='constant')\n        \n#     print(mel_spec_db.shape)\n        \n#     mel_spec_db_resized = mel_spec_db\n\n#     # Resize the spectrogram to the target shape for the CNN, if necessary\n# #     mel_spec_db_resized = tf.image.resize(mel_spec_db[..., np.newaxis], (128, 128)).numpy()\n#     mel_spec_db_expanded = np.stack([mel_spec_db_resized]*3, axis=-1)  # Expand to 3 channels\n\n#     return mel_spec_db_expanded\n\n\n\n\n\n\n# import random\n\n# def load_data_and_labels(audio_folder):\n#     features = []\n#     labels = []\n#     label_to_class_name = {}\n#     for label, (dirpath, dirnames, filenames) in enumerate(tqdm(os.walk(audio_folder))):\n#         # Filter to include only .ogg files\n#         ogg_filenames = [f for f in filenames if f.endswith('.ogg')]\n#         # Shuffle to ensure randomness\n#         random.shuffle(ogg_filenames)\n#         # Limit to the first 100 files\n#         for filename in ogg_filenames:\n#             audio_path = os.path.join(dirpath, filename)\n#             try:\n#                 feature = extract_features(audio_path)\n#                 features.append(feature)\n#                 labels.append(label-1)  # Adjust label to start from 0\n                \n#                 class_name = dirpath.split(\"/\")[-1]\n#                 label_to_class_name[class_name] = label-1\n#             except Exception as e:\n#                 print(f\"Error processing {filename}: {e}\")\n#     return np.array(features), np.array(labels), label_to_class_name\n\n\n# Load dataset\n# features, labels,label_dict = load_data_and_labels('/kaggle/input/birdclef-2024/train_audio/')\n\n# file_path = 'label_to_class_name.json'\n\n# # Writing the dictionary to a file as JSON\n# with open(file_path, 'w') as json_file:\n#     json.dump(label_dict, json_file)\n\n# labels = to_categorical(labels, num_classes=CLASSES)\n\n# Save features and labels\n# np.save('features_10_sec.npy', features)\n# np.save('labels_10_sec.npy', labels)\n\nfeatures = np.load('/kaggle/input/numpy-data-5/features.npy')\n\nlabels = np.load('/kaggle/input/numpy-data-5/labels.npy')\n\nfeatures = features/255.\n\n\n\n\n\n# Split dataset\nX_train, X_test, y_train, y_test = train_test_split(features, labels, test_size=0.1, random_state=42)\n\n\n\n\n\n\nfrom tensorflow.keras.layers import Input, Conv2D, MaxPooling2D, Lambda, GlobalAveragePooling2D, Dense, Layer, Dropout\nfrom tensorflow.keras.models import Model\nimport tensorflow as tf\nfrom tensorflow.keras.applications import MobileNet\n\ninput_tensor = Input(shape=(128, 216, 3))\n\nx = Conv2D(16, (3, 3), activation='relu', padding='same')(input_tensor)\nx = MaxPooling2D((2, 2))(x)  # This reduces the size to (64, 108, 16)\n#x = Dropout(0.5)(x)  # Increase dropout rate before final predictions\n\nx = Conv2D(32, (3, 3), activation='relu', padding='same')(x)\nx = MaxPooling2D((2, 2))(x)  # Further reduction to (32, 54, 32)\n#x = Dropout(0.5)(x)  # Increase dropout rate before final predictions\n\nx = Conv2D(3, (3, 3), activation='relu', padding='same')(x)\n\n# Use a Lambda layer for resizing\nx = Lambda(lambda image: tf.image.resize(image, (128, 128)))(x)\n\n# Continue with the MobileNet and the rest of the model\nbase_model = MobileNet(input_shape=(128, 128, 3), include_top=False, weights='imagenet')\n\n# Set the base model layers to not be trainable\nfor layer in base_model.layers:\n    layer.trainable = False\n\nx = base_model(x)\nx = GlobalAveragePooling2D()(x)\nx = Dense(128, activation='relu')(x)\n#x = Dropout(0.5)(x)  # Increase dropout rate before final predictions\n\npredictions = Dense(182, activation='softmax')(x)\n\nmodel = Model(inputs=input_tensor, outputs=predictions)\nmodel.summary()\n\n\n# Create the final model with the appropriate inputs and outputs\nmodel = Model(inputs=input_tensor, outputs=predictions)\n\n\n\n# Print the model summary to verify the structure and parameter count\n\n\n\n\nfrom tensorflow.keras.optimizers import Adam\n\n# Compile the model with a specific learning rate\nmodel.compile(optimizer=Adam(learning_rate=0.0003), \n              loss='categorical_crossentropy', \n              metrics=['accuracy'])\n\n\n# Train the model\nhistory = model.fit(X_train, y_train, epochs=50, validation_data=(X_test, y_test), batch_size=256,callbacks=[checkpoint]  # Include the scheduler\n)\n\n# # # Function to predict the class of an unseen .ogg file\n# # def predict_audio_class(file_path, model):\n# #     features = extract_features(file_path)\n# #     features = np.expand_dims(features, axis=0)  # Match the model's expected input\n# #     predictions = model.predict(features)\n# #     predicted_class = np.argmax(predictions, axis=1)\n# #     return predicted_class\n\n# # # Example usage for prediction\n# # test_file_path = \"test.ogg\"\n# # predicted_class = predict_audio_class(test_file_path, model)\n# # print(f\"Predicted class: {predicted_class}\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-11T08:37:27.403787Z","iopub.execute_input":"2024-04-11T08:37:27.404166Z","iopub.status.idle":"2024-04-11T08:37:38.585259Z","shell.execute_reply.started":"2024-04-11T08:37:27.404136Z","shell.execute_reply":"2024-04-11T08:37:38.583968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport matplotlib.pyplot as plt\n\n# Plotting training and validation loss\nplt.figure(figsize=(12, 5))\n\nplt.subplot(1, 2, 1)\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Training and Validation Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\n\n# Plotting training and validation accuracy\nplt.subplot(1, 2, 2)\nplt.plot(history.history['accuracy'], label='Training Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.title('Training and Validation Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}