{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport plotly.express as px\nimport librosa\nimport IPython.display as ipd\nimport librosa.display as lid\nimport soundfile as sf\nimport IPython.display as ipd\nimport librosa.display as lid\nimport matplotlib.pyplot as plt\nimport noisereduce as nr","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-28T16:51:43.050418Z","iopub.execute_input":"2024-05-28T16:51:43.050902Z","iopub.status.idle":"2024-05-28T16:51:43.065056Z","shell.execute_reply.started":"2024-05-28T16:51:43.050867Z","shell.execute_reply":"2024-05-28T16:51:43.063880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = '/kaggle/input/birdclef-2024'","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:19.165783Z","iopub.execute_input":"2024-05-28T16:46:19.166331Z","iopub.status.idle":"2024-05-28T16:46:19.177449Z","shell.execute_reply.started":"2024-05-28T16:46:19.166285Z","shell.execute_reply":"2024-05-28T16:46:19.176263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" # Data Exploration","metadata":{}},{"cell_type":"code","source":"# Load the training metadata\ntrain_metadata = pd.read_csv(f'{BASE_PATH}/train_metadata.csv')\ntrain_metadata['filepath'] = BASE_PATH + '/train_audio/' + train_metadata.filename\nprint(train_metadata.head())","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:19.179538Z","iopub.execute_input":"2024-05-28T16:46:19.180862Z","iopub.status.idle":"2024-05-28T16:46:19.332370Z","shell.execute_reply.started":"2024-05-28T16:46:19.180801Z","shell.execute_reply":"2024-05-28T16:46:19.331156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check the shape of dataset\ntrain_metadata.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:19.334106Z","iopub.execute_input":"2024-05-28T16:46:19.334508Z","iopub.status.idle":"2024-05-28T16:46:19.343895Z","shell.execute_reply.started":"2024-05-28T16:46:19.334475Z","shell.execute_reply":"2024-05-28T16:46:19.342090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_metadata.info()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:19.348999Z","iopub.execute_input":"2024-05-28T16:46:19.349644Z","iopub.status.idle":"2024-05-28T16:46:19.391313Z","shell.execute_reply.started":"2024-05-28T16:46:19.349599Z","shell.execute_reply":"2024-05-28T16:46:19.390418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for missing values\nprint(train_metadata.isnull().sum())\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:19.392714Z","iopub.execute_input":"2024-05-28T16:46:19.393052Z","iopub.status.idle":"2024-05-28T16:46:19.428240Z","shell.execute_reply.started":"2024-05-28T16:46:19.393022Z","shell.execute_reply":"2024-05-28T16:46:19.427122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Basic statistics\nprint(train_metadata.describe())","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:19.430518Z","iopub.execute_input":"2024-05-28T16:46:19.430977Z","iopub.status.idle":"2024-05-28T16:46:19.456042Z","shell.execute_reply.started":"2024-05-28T16:46:19.430938Z","shell.execute_reply":"2024-05-28T16:46:19.454493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of birds in each species\ntrain_metadata.primary_label.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:19.458208Z","iopub.execute_input":"2024-05-28T16:46:19.458790Z","iopub.status.idle":"2024-05-28T16:46:19.475336Z","shell.execute_reply.started":"2024-05-28T16:46:19.458749Z","shell.execute_reply":"2024-05-28T16:46:19.473750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualize the relationship between primary and secondary bird labels \ntmp = train_metadata[train_metadata['secondary_labels'] != '[]'].iloc[:40]\nfig = px.sunburst(tmp, path=['primary_label', 'secondary_labels'] )\nfig.update_layout(\n         title=\"Primary & Secondary Labels \",\n    )\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:19.477381Z","iopub.execute_input":"2024-05-28T16:46:19.477872Z","iopub.status.idle":"2024-05-28T16:46:19.592217Z","shell.execute_reply.started":"2024-05-28T16:46:19.477832Z","shell.execute_reply":"2024-05-28T16:46:19.590513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a copy of the df_train DataFrame\ntmp = train_metadata.copy()\n\n# Create a scatter mapbox plot\nfig = px.scatter_mapbox(\n    tmp,  # Use the tmp DataFrame as the data source\n    lat=\"latitude\",  # Latitude column for plotting points\n    lon=\"longitude\",  # Longitude column for plotting points\n    color=\"primary_label\",  # Color points based on the primary_label column\n    zoom=0.1,  # Initial zoom level of the map\n    title='Bird Recordings Loaction'  # Title of the plot\n)\n\n# Update the layout of the plot to use the \"open-street-map\" style for the map background\nfig.update_layout(mapbox_style=\"open-street-map\")\n\n# Update the layout of the plot to set the margin around the map\nfig.update_layout(margin={\"r\":0,\"t\":30,\"l\":0,\"b\":0})\n\n# Display the scatter mapbox plot\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:19.594231Z","iopub.execute_input":"2024-05-28T16:46:19.594674Z","iopub.status.idle":"2024-05-28T16:46:20.292268Z","shell.execute_reply.started":"2024-05-28T16:46:19.594636Z","shell.execute_reply":"2024-05-28T16:46:20.290696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Cleaning","metadata":{}},{"cell_type":"code","source":"sns.displot(data=train_metadata,x='latitude',bins=30,kde=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:20.294189Z","iopub.execute_input":"2024-05-28T16:46:20.294556Z","iopub.status.idle":"2024-05-28T16:46:21.014776Z","shell.execute_reply.started":"2024-05-28T16:46:20.294525Z","shell.execute_reply":"2024-05-28T16:46:21.013216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.displot(data=train_metadata,x='longitude',bins=30,kde=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:21.016985Z","iopub.execute_input":"2024-05-28T16:46:21.017388Z","iopub.status.idle":"2024-05-28T16:46:21.722075Z","shell.execute_reply.started":"2024-05-28T16:46:21.017346Z","shell.execute_reply":"2024-05-28T16:46:21.720549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For latitude and longitude in the train_metadata dataset, it might be more appropriate to fill missing values with the median to avoid potential distortions from outliers.","metadata":{}},{"cell_type":"code","source":"# Calculate mean values\nmedian_latitude = train_metadata['latitude'].median()\nmedian_longitude = train_metadata['longitude'].median()\n#Fill missing latitude and longitude with the mean values\ntrain_metadata['latitude'] = train_metadata['latitude'].fillna(median_latitude)\ntrain_metadata['longitude'] = train_metadata['longitude'].fillna(median_longitude)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:21.723616Z","iopub.execute_input":"2024-05-28T16:46:21.723976Z","iopub.status.idle":"2024-05-28T16:46:21.735444Z","shell.execute_reply.started":"2024-05-28T16:46:21.723946Z","shell.execute_reply":"2024-05-28T16:46:21.733698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check if there are any remaining missing values\nprint(train_metadata.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:21.740881Z","iopub.execute_input":"2024-05-28T16:46:21.741282Z","iopub.status.idle":"2024-05-28T16:46:21.779681Z","shell.execute_reply.started":"2024-05-28T16:46:21.741254Z","shell.execute_reply":"2024-05-28T16:46:21.777945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Audio file path\naudio_file_path = f'{BASE_PATH}/train_audio'\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:21.781830Z","iopub.execute_input":"2024-05-28T16:46:21.782186Z","iopub.status.idle":"2024-05-28T16:46:21.788816Z","shell.execute_reply.started":"2024-05-28T16:46:21.782157Z","shell.execute_reply":"2024-05-28T16:46:21.786967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    seed = 42\n    \n    # Input image size and batch size\n    img_size = [128, 384]\n    batch_size = 64\n    \n    # Audio duration, sample rate, and length\n    duration = 15 # second\n    sample_rate = 32000\n    audio_len = duration*sample_rate\n    # STFT parameters\n    nfft = 2028\n    window = 2048\n    hop_length = audio_len // (img_size[1] - 1)\n    fmin = 20\n    fmax = 16000\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:21.790654Z","iopub.execute_input":"2024-05-28T16:46:21.791239Z","iopub.status.idle":"2024-05-28T16:46:21.801158Z","shell.execute_reply.started":"2024-05-28T16:46:21.791193Z","shell.execute_reply":"2024-05-28T16:46:21.799420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Load Audio Data\ndef load_audio(filepath):\n    audio, sr = librosa.load(filepath)\n    return audio, sr","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:21.803509Z","iopub.execute_input":"2024-05-28T16:46:21.803985Z","iopub.status.idle":"2024-05-28T16:46:21.817379Z","shell.execute_reply.started":"2024-05-28T16:46:21.803941Z","shell.execute_reply":"2024-05-28T16:46:21.815975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply Spectral Subtraction\ndef normalize_and_denoise(audio,sample_rate):\n    # Normalize audio data\n    audio = audio / np.max(np.abs(audio))\n    \n    # Apply Spectral Subtraction Filter\n    filtered_audio = nr.reduce_noise(y=audio, sr=sample_rate)\n    \n    return filtered_audio","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:21.819447Z","iopub.execute_input":"2024-05-28T16:46:21.819894Z","iopub.status.idle":"2024-05-28T16:46:21.830302Z","shell.execute_reply.started":"2024-05-28T16:46:21.819856Z","shell.execute_reply":"2024-05-28T16:46:21.828578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Extraction","metadata":{}},{"cell_type":"code","source":"def display_audio(original_audio, denoised_audio, sample_rate):\n    print(\"Original Audio:\")\n    display(ipd.Audio(original_audio, rate=sample_rate))\n\n    print(\"Denoised Audio:\")\n    display(ipd.Audio(denoised_audio, rate=sample_rate))","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:21.832777Z","iopub.execute_input":"2024-05-28T16:46:21.833320Z","iopub.status.idle":"2024-05-28T16:46:21.843602Z","shell.execute_reply.started":"2024-05-28T16:46:21.833282Z","shell.execute_reply":"2024-05-28T16:46:21.842267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_mel_spectrogram(audio, sample_rate):\n    # Compute mel spectrogram\n    mel_spectrogram = librosa.feature.melspectrogram(y=audio, sr=sample_rate)\n    # Convert to decibels\n    mel_spectrogram_db = librosa.power_to_db(mel_spectrogram, ref=np.max)\n    return mel_spectrogram_db","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:21.844931Z","iopub.execute_input":"2024-05-28T16:46:21.845365Z","iopub.status.idle":"2024-05-28T16:46:21.856977Z","shell.execute_reply.started":"2024-05-28T16:46:21.845330Z","shell.execute_reply":"2024-05-28T16:46:21.855545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def visualize_mel_spectrogram(original_audio,denoised_audio, sr, title=\"Mel Spectrogram\"):\n    # Plot the Mel spectrogram\n    plt.figure(figsize=(14, 5))\n    plt.subplot(2, 1, 1) \n    librosa.display.specshow(extract_mel_spectrogram(original_audio,sr), sr=sr, x_axis='time', y_axis='mel')\n    plt.colorbar(format='%+2.0f dB')\n    plt.title('Original Audio')\n    plt.subplot(2, 1, 2)\n    librosa.display.specshow(extract_mel_spectrogram(denoised_audio,sr), sr=sr, x_axis='time', y_axis='mel')\n    plt.colorbar(format='%+2.0f dB')\n    plt.title('Denoised Audio')\n    plt.suptitle(title)\n    plt.tight_layout()\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:21.858343Z","iopub.execute_input":"2024-05-28T16:46:21.858834Z","iopub.status.idle":"2024-05-28T16:46:21.872321Z","shell.execute_reply.started":"2024-05-28T16:46:21.858789Z","shell.execute_reply":"2024-05-28T16:46:21.871091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"original_audio ,sr= load_audio(BASE_PATH + '/train_audio/asbfly/XC267681.ogg')\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:21.873621Z","iopub.execute_input":"2024-05-28T16:46:21.874036Z","iopub.status.idle":"2024-05-28T16:46:22.097299Z","shell.execute_reply.started":"2024-05-28T16:46:21.873998Z","shell.execute_reply":"2024-05-28T16:46:22.095788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"denoised_audio = normalize_and_denoise(audio,sr)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:22.099339Z","iopub.execute_input":"2024-05-28T16:46:22.099830Z","iopub.status.idle":"2024-05-28T16:46:23.279378Z","shell.execute_reply.started":"2024-05-28T16:46:22.099789Z","shell.execute_reply":"2024-05-28T16:46:23.278096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_audio(original_audio, denoised_audio, sr)","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:23.281251Z","iopub.execute_input":"2024-05-28T16:46:23.282051Z","iopub.status.idle":"2024-05-28T16:46:23.544849Z","shell.execute_reply.started":"2024-05-28T16:46:23.282010Z","shell.execute_reply":"2024-05-28T16:46:23.542708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize_mel_spectrogram(original_audio,denoised_audio, sr, title=\"Mel Spectrogram\")","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:23.547722Z","iopub.execute_input":"2024-05-28T16:46:23.549506Z","iopub.status.idle":"2024-05-28T16:46:25.923945Z","shell.execute_reply.started":"2024-05-28T16:46:23.549412Z","shell.execute_reply":"2024-05-28T16:46:25.922316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Extract Features for All Training Audio Files\n","metadata":{"execution":{"iopub.status.busy":"2024-05-28T16:46:25.925617Z","iopub.execute_input":"2024-05-28T16:46:25.926645Z","iopub.status.idle":"2024-05-28T16:46:25.931899Z","shell.execute_reply.started":"2024-05-28T16:46:25.926607Z","shell.execute_reply":"2024-05-28T16:46:25.930087Z"},"trusted":true},"execution_count":null,"outputs":[]}]}