{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8045887,"sourceType":"datasetVersion","datasetId":4744304},{"sourceId":26431,"sourceType":"modelInstanceVersion","modelInstanceId":22241,"modelId":32660}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Table of Contents**\n\n* **Introduction**\n* **Import Libraries**\n* **Data Collection and Processing**\n* **Exploratory Data Analysis (EDA)**\n* **Feature Engineering**\n    1. Feature Extraction\n    2. Label Encoding\n    3. Random sampling Data\n* **Model Training**\n* **Model Evaluation**\n* **Model Testing**","metadata":{}},{"cell_type":"markdown","source":"### Import the required Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport glob\nimport shutil\nimport zipfile\nimport matplotlib.pyplot as plt\nplt.style.use('dark_background')\nimport seaborn as sns\nimport numpy as np\n!pip install plotly\nimport plotly.express as px\nimport librosa\nfrom IPython.display import Audio\nimport pandas as pd\nimport pickle\nfrom joblib import dump, load\nfrom pathlib import Path\n!pip install -U imbalanced-learn           # For Any Imbalnce in the Dataset\nfrom imblearn.over_sampling import RandomOverSampler\nimport sklearn\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2024-12-05T07:48:24.275821Z","iopub.execute_input":"2024-12-05T07:48:24.276643Z","iopub.status.idle":"2024-12-05T07:48:46.128029Z","shell.execute_reply.started":"2024-12-05T07:48:24.276611Z","shell.execute_reply":"2024-12-05T07:48:46.127232Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Data Collection and Processing**","metadata":{}},{"cell_type":"code","source":"meta_data = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\nmeta_data.head(4)","metadata":{"execution":{"iopub.status.busy":"2024-12-05T07:49:07.968431Z","iopub.execute_input":"2024-12-05T07:49:07.969511Z","iopub.status.idle":"2024-12-05T07:49:08.139446Z","shell.execute_reply.started":"2024-12-05T07:49:07.969482Z","shell.execute_reply":"2024-12-05T07:49:08.138489Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"1. primary_label: The main label categorizing the bird sound \n2. secondary_labels: Additional labels related to the sound \n3. type: Type of sound recorded (e.g., \"call\", \"song\").\n4. latitude: The geographic latitude of the recording location.\n5. longitude: The geographic longitude of the recording location.\n6. scientific_name: The scientific name of the bird species.\n7. common_name: The common name of the bird species.\n8. author: The person who recorded the bird sound.\n9. license: The licensing terms for the recording (e.g., Creative Commons).\n10. rating: The user rating of the recording (e.g., 5.0).\n11. url: The link to the online recording.\n12. filename: The file path or name for the recording.","metadata":{}},{"cell_type":"markdown","source":"### Link to the Corresponding Birds","metadata":{}},{"cell_type":"code","source":"meta_data.url\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:51:58.700782Z","iopub.execute_input":"2024-12-05T07:51:58.701436Z","iopub.status.idle":"2024-12-05T07:51:58.708878Z","shell.execute_reply.started":"2024-12-05T07:51:58.701408Z","shell.execute_reply":"2024-12-05T07:51:58.707932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta_data.info()","metadata":{"execution":{"iopub.status.busy":"2024-12-05T07:49:22.099121Z","iopub.execute_input":"2024-12-05T07:49:22.099788Z","iopub.status.idle":"2024-12-05T07:49:22.130755Z","shell.execute_reply.started":"2024-12-05T07:49:22.099760Z","shell.execute_reply":"2024-12-05T07:49:22.129854Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### There are few NUll Values in our Dataset","metadata":{}},{"cell_type":"markdown","source":"# **Exploratory Data Analysis (EDA)**","metadata":{}},{"cell_type":"code","source":"# Distribution of recordings by authors\nplt.figure(figsize=(12, 6))\nsns.countplot(x='author', data=meta_data, order=meta_data['author'].value_counts().index[:10])\nplt.xticks(rotation=45)\nplt.title('Top 10 Authors by Number of Recordings')\nplt.xlabel('Author')\nplt.ylabel('Count')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:56:37.660503Z","iopub.execute_input":"2024-12-05T07:56:37.661184Z","iopub.status.idle":"2024-12-05T07:56:37.930520Z","shell.execute_reply.started":"2024-12-05T07:56:37.661156Z","shell.execute_reply":"2024-12-05T07:56:37.929634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import geopandas as gpd\nimport matplotlib.pyplot as plt\n\ngeo_df = gpd.GeoDataFrame(meta_data, geometry=gpd.points_from_xy(meta_data.longitude, meta_data.latitude))\n\nworld = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres'))\nfig, ax = plt.subplots(figsize=(10, 8))\n\n\nworld.plot(ax=ax, color='lightgrey')\n\n# Plot bird species locations\ngeo_df.plot(ax=ax, marker='o', color='red', markersize=5, label='Bird Species')\nax.set_title('Origin of Bird Species')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-12-05T08:01:30.979785Z","iopub.execute_input":"2024-12-05T08:01:30.980541Z","iopub.status.idle":"2024-12-05T08:01:33.256734Z","shell.execute_reply.started":"2024-12-05T08:01:30.980497Z","shell.execute_reply":"2024-12-05T08:01:33.255926Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def audio_waveframe(file_path):\n    audio_data, sampling_rate = librosa.load(file_path)\n    duration = len(audio_data) / sampling_rate\n    time = np.arange(0, duration, 1/sampling_rate)\n    plt.figure(figsize=(30, 4))\n    plt.plot(time, audio_data, color='blue')\n    plt.title('Audio Waveform')\n    plt.xlabel('Time (s)')\n    plt.ylabel('Amplitude')\n    plot = plt.show()\n    return plot\n\ndef spectrogram(file_path):\n    n_fft = 500 \n    hop_length = 50  \n    audio_data, sampling_rate = librosa.load(file_path)\n    stft = librosa.stft(audio_data, n_fft=n_fft, hop_length=hop_length)\n    spectrogram = librosa.amplitude_to_db(np.abs(stft))\n    plt.figure(figsize=(30, 6))\n    librosa.display.specshow(spectrogram, sr=sampling_rate, hop_length=hop_length, x_axis='time', y_axis='linear')\n    plt.colorbar(format='%+2.0f dB')\n    plt.title('Spectrogram')\n    plt.xlabel('Time (s)')\n    plt.ylabel('Frequency (Hz)')\n    plt.tight_layout()\n    plot = plt.show()\n    return plot\n\ndef audio_analysis(file_path):\n    aw = audio_waveframe(file_path)\n    spg = spectrogram(file_path)\n    return aw, spg","metadata":{"execution":{"iopub.status.busy":"2024-12-05T08:03:54.567614Z","iopub.execute_input":"2024-12-05T08:03:54.568534Z","iopub.status.idle":"2024-12-05T08:03:54.575753Z","shell.execute_reply.started":"2024-12-05T08:03:54.568502Z","shell.execute_reply":"2024-12-05T08:03:54.574716Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Functions for audio analysis:\n\n1. `audio_waveframe(file_path)`: Loads an audio file, plots its waveform showing amplitude variation over time, and returns the plot.\n2. `spectrogram(file_path)`: Computes the spectrogram of the audio file, which visualizes its frequency content over time, and returns the plot.\n3. `audio_analysis(file_path)`: Calls the `audio_waveframe()` and `spectrogram()` functions for the given audio file, then returns the plots generated by both functions.","metadata":{}},{"cell_type":"markdown","source":"### Sample Audio File Analysis","metadata":{}},{"cell_type":"code","source":"audio_analysis('/kaggle/input/birdclef-2024/train_audio/asbfly/XC134896.ogg')\nAudio('/kaggle/input/birdclef-2024/train_audio/asbfly/XC134896.ogg')","metadata":{"execution":{"iopub.status.busy":"2024-12-05T08:03:55.221006Z","iopub.execute_input":"2024-12-05T08:03:55.221653Z","iopub.status.idle":"2024-12-05T08:03:57.826822Z","shell.execute_reply.started":"2024-12-05T08:03:55.221626Z","shell.execute_reply":"2024-12-05T08:03:57.825948Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Feature Engineering**","metadata":{}},{"cell_type":"markdown","source":"### Creating Mapping between the File Paths and their Corresponing Class Labels (Store All into the Label_Mapping)","metadata":{}},{"cell_type":"code","source":"dataset_dir = '/kaggle/input/birdclef-2024/train_audio'\nlabel_mapping = {}\nfor label in os.listdir(dataset_dir):\n    label_dir = os.path.join(dataset_dir, label)\n    if os.path.isdir(label_dir):\n        for audio_file in os.listdir(label_dir):\n            audio_file_path = os.path.join(label_dir, audio_file)\n            label_mapping[audio_file_path] = label","metadata":{"execution":{"iopub.status.busy":"2024-12-05T08:04:37.566460Z","iopub.execute_input":"2024-12-05T08:04:37.567332Z","iopub.status.idle":"2024-12-05T08:04:40.421868Z","shell.execute_reply.started":"2024-12-05T08:04:37.567300Z","shell.execute_reply":"2024-12-05T08:04:40.420884Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Stored Dictionary of the Audio Files with their Labels","metadata":{}},{"cell_type":"code","source":"data = [(audio_file_path, label) for audio_file_path, label in label_mapping.items()]\nannotated_data = pd.DataFrame(data, columns=['audio_file_path', 'label'])\nannotated_data","metadata":{"execution":{"iopub.status.busy":"2024-12-05T08:05:45.061761Z","iopub.execute_input":"2024-12-05T08:05:45.062679Z","iopub.status.idle":"2024-12-05T08:05:45.079623Z","shell.execute_reply.started":"2024-12-05T08:05:45.062647Z","shell.execute_reply":"2024-12-05T08:05:45.078786Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Feature Extraction","metadata":{}},{"cell_type":"markdown","source":"#### Load the Features through MFCC & Extract the Feature with their Lable from the Audio File","metadata":{}},{"cell_type":"code","source":"def extract_features(file_path):\n    audio, sample_rate = librosa.load(file_path)\n    # Extract features using Mel-Frequency Cepstral Coefficients (MFCC)\n    mfccs = librosa.feature.mfcc(y=audio, sr=sample_rate, n_mfcc=40)\n    flattened_features = np.mean(mfccs.T, axis=0)\n    return flattened_features\n\n\ndef load_data_and_extract_features(data_dir):\n    labels = []\n    features = []\n\n    for filename in os.listdir(data_dir):\n        if filename.endswith('.ogg'):\n            file_path = os.path.join(data_dir, filename)\n            label = filename.split('-')[0]\n            labels.append(label)\n            feature = extract_features(file_path)\n            features.append(feature)\n    return np.array(features), np.array(labels)","metadata":{"execution":{"iopub.status.busy":"2024-12-05T08:07:51.876429Z","iopub.execute_input":"2024-12-05T08:07:51.877116Z","iopub.status.idle":"2024-12-05T08:07:51.882758Z","shell.execute_reply.started":"2024-12-05T08:07:51.877087Z","shell.execute_reply":"2024-12-05T08:07:51.881894Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"1. `extract_features(file_path)`: \n   - Loads an audio file using librosa library.\n   - Calculates Mel-Frequency Cepstral Coefficients (MFCC) features from the audio.\n   - Averages the MFCC features over time to create a flattened feature vector.\n   - Returns the flattened feature vector.\n\n2. `load_data_and_extract_features(data_dir)`:\n   - Iterates through each audio file in the specified directory.\n   - Extracts the label from the filename by splitting it at '-'.\n   - Calls `extract_features()` to extract features from each audio file.\n   - Returns numpy arrays containing the extracted features and corresponding labels.","metadata":{}},{"cell_type":"markdown","source":"### Function to Extract the Features, Time with GPU P100 ()","metadata":{}},{"cell_type":"code","source":"# from tqdm import tqdm\n\n# extracted_features = []\n\n# for i in tqdm(annotated_data['audio_file_path']):\n#     features = extract_features(file_path=i)\n#     # print(features)\n#     extracted_features.append(features)","metadata":{"execution":{"iopub.status.busy":"2024-12-05T08:43:03.216082Z","iopub.execute_input":"2024-12-05T08:43:03.216320Z","iopub.status.idle":"2024-12-05T09:20:30.978217Z","shell.execute_reply.started":"2024-12-05T08:43:03.216299Z","shell.execute_reply":"2024-12-05T09:20:30.977120Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This Python code snippet uses the `tqdm` library to create a progress bar for iterating over a list of audio file paths (`annotated_data['audio_file_path']`). Inside the loop, it calls a function `extract_features()` to extract features from each audio file. The extracted features are then appended to the list `extracted_features`. The `tqdm()` function provides a visual progress indicator, making it easier to track the progress of the loop.","metadata":{}},{"cell_type":"markdown","source":"### Save the Results","metadata":{}},{"cell_type":"code","source":"# with open(\"extracted_features\", \"wb\") as file:   #Pickling\n# \tpickle.dump(extracted_features, file)","metadata":{"execution":{"iopub.status.busy":"2024-04-06T03:53:28.322561Z","iopub.execute_input":"2024-04-06T03:53:28.323616Z","iopub.status.idle":"2024-04-06T03:53:28.44279Z","shell.execute_reply.started":"2024-04-06T03:53:28.32357Z","shell.execute_reply":"2024-04-06T03:53:28.441732Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(\"/kaggle/input/extracted-features-pickle/extracted_features\", \"rb\") as file:   # Unpickling\n\tpickled_extracted_features = pickle.load(file)","metadata":{"execution":{"iopub.status.busy":"2024-12-05T09:20:45.885663Z","iopub.execute_input":"2024-12-05T09:20:45.886007Z","iopub.status.idle":"2024-12-05T09:20:45.959944Z","shell.execute_reply.started":"2024-12-05T09:20:45.885979Z","shell.execute_reply":"2024-12-05T09:20:45.959302Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Label Encoding","metadata":{}},{"cell_type":"code","source":"label_encoder = LabelEncoder()\nannotated_data['encoded_label'] = label_encoder.fit_transform(annotated_data['label'])\nannotated_data","metadata":{"execution":{"iopub.status.busy":"2024-12-05T09:20:48.388457Z","iopub.execute_input":"2024-12-05T09:20:48.389332Z","iopub.status.idle":"2024-12-05T09:20:48.403773Z","shell.execute_reply.started":"2024-12-05T09:20:48.389294Z","shell.execute_reply":"2024-12-05T09:20:48.403019Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"annotated_data.info()","metadata":{"execution":{"iopub.status.busy":"2024-12-05T09:20:51.538619Z","iopub.execute_input":"2024-12-05T09:20:51.539368Z","iopub.status.idle":"2024-12-05T09:20:51.551269Z","shell.execute_reply.started":"2024-12-05T09:20:51.539341Z","shell.execute_reply":"2024-12-05T09:20:51.550365Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This code uses the `LabelEncoder` from scikit-learn to convert categorical labels into numerical values. It fits the encoder to the 'label' column of the annotated dataset, assigning a unique numeric code to each unique label. The encoded labels are then stored in a new column called 'encoded_label' in the annotated dataset. This transformation allows machine learning algorithms to work with categorical data, improving model training and performance.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(24, 12))\nsns.countplot(x='primary_label', data=meta_data, order=meta_data['primary_label'].value_counts().index)\nplt.xticks(rotation=45)\nplt.rc('font', size=6)\nplt.title('Count of Bird Species Classes')\nplt.xlabel('Bird Species')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-05T09:20:52.124535Z","iopub.execute_input":"2024-12-05T09:20:52.125198Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Random Sampling Data","metadata":{}},{"cell_type":"code","source":"x = np.vstack(pickled_extracted_features)\ny = annotated_data['encoded_label']\n\nprint(x.shape)\nprint(y.shape)","metadata":{"execution":{"iopub.status.busy":"2024-12-05T09:21:00.208559Z","iopub.execute_input":"2024-12-05T09:21:00.209163Z","iopub.status.idle":"2024-12-05T09:21:00.237272Z","shell.execute_reply.started":"2024-12-05T09:21:00.209133Z","shell.execute_reply":"2024-12-05T09:21:00.236295Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ros = RandomOverSampler(random_state=42)\nfeatures_resampled, labels_reshampled = ros.fit_resample(x, y)\n\nprint(features_resampled.shape)\nprint(labels_reshampled.shape)","metadata":{"execution":{"iopub.status.busy":"2024-12-05T09:21:00.530314Z","iopub.execute_input":"2024-12-05T09:21:00.530831Z","iopub.status.idle":"2024-12-05T09:21:00.566260Z","shell.execute_reply.started":"2024-12-05T09:21:00.530805Z","shell.execute_reply":"2024-12-05T09:21:00.565389Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Model Training**","metadata":{}},{"cell_type":"code","source":"# Split data into training and testing sets\nx_train, x_test, y_train, y_test = train_test_split(features_resampled, labels_reshampled, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-12-05T09:21:04.601886Z","iopub.execute_input":"2024-12-05T09:21:04.602248Z","iopub.status.idle":"2024-12-05T09:21:04.619246Z","shell.execute_reply.started":"2024-12-05T09:21:04.602220Z","shell.execute_reply":"2024-12-05T09:21:04.618550Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random_forest_classifier = RandomForestClassifier(n_estimators=100, random_state=42)\nrandom_forest_model = random_forest_classifier.fit(x_train, y_train)\ny_predict = random_forest_model.predict(x_test)\naccuracy = accuracy_score(y_test, y_predict)\nprint(\"Accuracy:\", accuracy)","metadata":{"execution":{"iopub.status.busy":"2024-12-05T09:21:12.947582Z","iopub.execute_input":"2024-12-05T09:21:12.947927Z","iopub.status.idle":"2024-12-05T09:24:15.540633Z","shell.execute_reply.started":"2024-12-05T09:21:12.947901Z","shell.execute_reply":"2024-12-05T09:24:15.539745Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This code splits the data into training and testing sets, with 80% for training and 20% for testing. Then, it builds a Random Forest classifier with 100 trees and trains it on the training data. After training, it predicts labels for the testing set and computes the accuracy of the predictions compared to the actual labels. Finally, it prints the accuracy score, indicating how well the classifier performed on the test data.","metadata":{}},{"cell_type":"markdown","source":"# **Model Evaluation**","metadata":{}},{"cell_type":"code","source":"def evaluate_model(y_true, y_pred):\n    # Calculate accuracy\n    accuracy = accuracy_score(y_true, y_pred)\n    # Calculate precision\n    precision = precision_score(y_true, y_pred, average='weighted')\n    # Calculate recall\n    recall = recall_score(y_true, y_pred, average='weighted')\n    # Calculate F1 score\n    f1 = f1_score(y_true, y_pred, average='weighted')\n    \n    return accuracy, precision, recall, f1\n\n# Evaluate the model\naccuracy, precision, recall, f1 = evaluate_model(y_test, y_predict)\n# Print evaluation metrics\nprint(\"Accuracy:\", accuracy)\nprint(\"Precision:\", precision)\nprint(\"Recall:\", recall)\nprint(\"F1 Score:\", f1)\n","metadata":{"execution":{"iopub.status.busy":"2024-12-05T09:24:15.542591Z","iopub.execute_input":"2024-12-05T09:24:15.543291Z","iopub.status.idle":"2024-12-05T09:24:15.571917Z","shell.execute_reply.started":"2024-12-05T09:24:15.543255Z","shell.execute_reply":"2024-12-05T09:24:15.571226Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This code defines a function `evaluate_model` to assess the performance of a classification model using various metrics such as accuracy, precision, recall, and F1 score. It takes the true labels (`y_true`) and predicted labels (`y_pred`) as inputs. The function calculates each metric using scikit-learn's built-in functions (`accuracy_score`, `precision_score`, `recall_score`, `f1_score`) and returns these metrics. Finally, it evaluates the model on a test set and prints out the computed evaluation metrics.","metadata":{}},{"cell_type":"markdown","source":"# **Model Testing and Deployment**","metadata":{}},{"cell_type":"code","source":"# dump(random_forest_model, 'audio_classifier_model.joblib')","metadata":{"execution":{"iopub.status.busy":"2024-04-07T13:15:15.592409Z","iopub.execute_input":"2024-04-07T13:15:15.593543Z","iopub.status.idle":"2024-04-07T13:15:20.14885Z","shell.execute_reply.started":"2024-04-07T13:15:15.593505Z","shell.execute_reply":"2024-04-07T13:15:20.147513Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# model = load('/kaggle/working/audio_classifier_model.joblib')\nmodel = random_forest_classifier","metadata":{"execution":{"iopub.status.busy":"2024-12-05T09:26:57.856373Z","iopub.execute_input":"2024-12-05T09:26:57.856717Z","iopub.status.idle":"2024-12-05T09:26:57.860836Z","shell.execute_reply.started":"2024-12-05T09:26:57.856691Z","shell.execute_reply":"2024-12-05T09:26:57.859899Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def audio_classification(file_path):\n    audio = file_path\n    print(audio)\n    extracted_features = extract_features(audio).reshape(1, -1)\n    # extracted_features = x_test[112].reshape(1, -1)\n    y_predict = model.predict(extracted_features)\n    labels_list = annotated_data['label'].unique()\n    encoded_label = annotated_data['encoded_label'].unique()\n\n    labels = {}\n    for label, prediction in zip(encoded_label, labels_list):\n        labels[label] = prediction\n    if y_predict[0] in labels.keys():\n        predicted = ('Predicted Class:', labels[y_predict[0]])\n    return predicted","metadata":{"execution":{"iopub.status.busy":"2024-12-05T09:27:01.704331Z","iopub.execute_input":"2024-12-05T09:27:01.705038Z","iopub.status.idle":"2024-12-05T09:27:01.709813Z","shell.execute_reply.started":"2024-12-05T09:27:01.705011Z","shell.execute_reply":"2024-12-05T09:27:01.708992Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_path = '/kaggle/input/birdclef-2024/unlabeled_soundscapes/1001358022.ogg'\naudio_analysis(file_path)\nAudio(file_path)","metadata":{"execution":{"iopub.status.busy":"2024-12-05T09:27:02.212949Z","iopub.execute_input":"2024-12-05T09:27:02.213741Z","iopub.status.idle":"2024-12-05T09:27:19.733545Z","shell.execute_reply.started":"2024-12-05T09:27:02.213712Z","shell.execute_reply":"2024-12-05T09:27:19.732450Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"audio_classification(file_path)","metadata":{"execution":{"iopub.status.busy":"2024-12-05T09:27:19.735735Z","iopub.execute_input":"2024-12-05T09:27:19.736022Z","iopub.status.idle":"2024-12-05T09:27:20.401262Z","shell.execute_reply.started":"2024-12-05T09:27:19.735994Z","shell.execute_reply":"2024-12-05T09:27:20.399129Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Project Submission**","metadata":{}},{"cell_type":"code","source":"test_soundscapes = '/kaggle/input/birdclef-2024/test_soundscapes'\n\nfor path in Path(test_soundscapes).glob(\"*.ogg\"):\n    print(path)\n    print(path.stem)\n    print(path.stem.split(\"_\"))","metadata":{"execution":{"iopub.status.busy":"2024-12-05T09:29:13.202227Z","iopub.execute_input":"2024-12-05T09:29:13.202574Z","iopub.status.idle":"2024-12-05T09:29:13.213104Z","shell.execute_reply.started":"2024-12-05T09:29:13.202547Z","shell.execute_reply":"2024-12-05T09:29:13.212250Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pd.DataFrame(\n     [(path.stem, *path.stem.split(\"_\"), path) for path in Path(test_soundscapes).glob(\"*.ogg\")],\n    columns = [\"filename\", \"name\" ,\"id\", \"path\"]\n)\nprint(test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-12-05T09:29:23.810508Z","iopub.execute_input":"2024-12-05T09:29:23.811091Z","iopub.status.idle":"2024-12-05T09:29:23.823543Z","shell.execute_reply.started":"2024-12-05T09:29:23.811063Z","shell.execute_reply":"2024-12-05T09:29:23.822499Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filenames = test.filename.values.tolist()\n\nbird_cols = list(pd.get_dummies(meta_data['primary_label']).columns)\nsubmission_df = pd.DataFrame(columns=['row_id']+bird_cols)\nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2024-12-05T09:29:24.741113Z","iopub.execute_input":"2024-12-05T09:29:24.741724Z","iopub.status.idle":"2024-12-05T09:29:24.761752Z","shell.execute_reply.started":"2024-12-05T09:29:24.741695Z","shell.execute_reply":"2024-12-05T09:29:24.761010Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i, file in enumerate(filenames):\n    predicted = random_forest_model.predict[i]\n    num_rows = len(predicted)\n    row_ids = [f'{file}_{(i+1)*5}' for i in range(num_rows)]\n    df = pd.DataFrame(columns=['row_id']+bird_cols)\n    \n    df['row_id'] = row_ids\n    df[bird_cols] = predicted\n    \n    submission_df = pd.concat([submission_df,df]).reset_index(drop=True)\n    submission_df","metadata":{"execution":{"iopub.status.busy":"2024-12-05T09:29:33.205180Z","iopub.execute_input":"2024-12-05T09:29:33.205815Z","iopub.status.idle":"2024-12-05T09:29:33.210891Z","shell.execute_reply.started":"2024-12-05T09:29:33.205789Z","shell.execute_reply":"2024-12-05T09:29:33.209984Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-12-05T09:29:34.761771Z","iopub.execute_input":"2024-12-05T09:29:34.762197Z","iopub.status.idle":"2024-12-05T09:29:34.768693Z","shell.execute_reply.started":"2024-12-05T09:29:34.762170Z","shell.execute_reply":"2024-12-05T09:29:34.768016Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}