{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Let's examine the dataset**","metadata":{}},{"cell_type":"code","source":"# Loading train_metadata.csv\ntrain_metadata = pd.read_csv(\"/kaggle/input/birdclef-2023/train_metadata.csv\")\ntrain_metadata","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:02:31.877503Z","iopub.execute_input":"2023-05-16T10:02:31.878376Z","iopub.status.idle":"2023-05-16T10:02:32.066281Z","shell.execute_reply.started":"2023-05-16T10:02:31.878343Z","shell.execute_reply":"2023-05-16T10:02:32.065143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading eBird_Taxonomy_v2023.csv\ntaxonomy = pd.read_csv(\"/kaggle/input/birdclef-2023/eBird_Taxonomy_v2021.csv\")\ntaxonomy","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:04:00.892444Z","iopub.execute_input":"2023-05-16T10:04:00.892837Z","iopub.status.idle":"2023-05-16T10:04:01.007547Z","shell.execute_reply.started":"2023-05-16T10:04:00.89281Z","shell.execute_reply":"2023-05-16T10:04:01.006243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the task description, we know that we have 264 bird species, which is too many to label on the graph. Therefore, instead of labeling each species, let's just look at the distribution of the number of audio recordings for each bird species.","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom matplotlib.colors import ListedColormap\n\nsns.set_style('whitegrid')\n\n# Getting a list of unique bird species\nbird_species = train_metadata['primary_label'].unique()\nbird_colors = ListedColormap(sns.color_palette('bright', n_colors=len(bird_species)).as_hex())\n\n# Histogram plot\nfig, ax = plt.subplots(figsize=(12, 6))\nsns.histplot(data=train_metadata, x='primary_label', ax=ax, color=bird_colors(0), alpha=0.5)\nax.set_xlabel('Bird Species')\nax.set_ylabel('Number of Audio Recordings')\nax.set_title('Distribution of Audio Recordings by Bird Species')\nax.set_xticklabels([])\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:06:22.978473Z","iopub.execute_input":"2023-05-16T10:06:22.978856Z","iopub.status.idle":"2023-05-16T10:06:25.374345Z","shell.execute_reply.started":"2023-05-16T10:06:22.978828Z","shell.execute_reply":"2023-05-16T10:06:25.37302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The graph shows that our dataset is imbalanced.","metadata":{}},{"cell_type":"markdown","source":"**Let's check if there are any duplicate file names.**","metadata":{}},{"cell_type":"code","source":"train_metadata[train_metadata['filename'].str.split('/').str[1].duplicated(keep=False) & ~train_metadata['filename'].duplicated(keep=False)]","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:08:10.394325Z","iopub.execute_input":"2023-05-16T10:08:10.394737Z","iopub.status.idle":"2023-05-16T10:08:10.456597Z","shell.execute_reply.started":"2023-05-16T10:08:10.394709Z","shell.execute_reply":"2023-05-16T10:08:10.455394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking them audibly\nfrom IPython.display import Audio\nAudio(\"/kaggle/input/birdclef-2023/train_audio/gnbcam2/XC316684.ogg\")","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:10:17.25653Z","iopub.execute_input":"2023-05-16T10:10:17.256919Z","iopub.status.idle":"2023-05-16T10:10:17.268136Z","shell.execute_reply.started":"2023-05-16T10:10:17.256891Z","shell.execute_reply":"2023-05-16T10:10:17.266984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Audio(\"/kaggle/input/birdclef-2023/train_audio/cibwar1/XC316684.ogg\")","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:10:22.136326Z","iopub.execute_input":"2023-05-16T10:10:22.136716Z","iopub.status.idle":"2023-05-16T10:10:22.14782Z","shell.execute_reply.started":"2023-05-16T10:10:22.136687Z","shell.execute_reply":"2023-05-16T10:10:22.146543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Both audios are similar. It's possible that both bird species are present in the recordings, or the data might have been misclassified or mistakenly entered. In any case, both recordings will be removed from the dataset in this notebook. Additionally, the cibwar1 and gnbcam2 folders contain a sufficient number of audio files for future training purposes.","metadata":{}},{"cell_type":"code","source":"# removed\ntrain_metadata = train_metadata.drop([3558, 7420])","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:11:39.665837Z","iopub.execute_input":"2023-05-16T10:11:39.667003Z","iopub.status.idle":"2023-05-16T10:11:39.67614Z","shell.execute_reply.started":"2023-05-16T10:11:39.666963Z","shell.execute_reply":"2023-05-16T10:11:39.674924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check if the bird names in train_metadata exist in the taxonomy.","metadata":{}},{"cell_type":"code","source":"is_present = 'African Bare-eyed Thrush' in taxonomy['PRIMARY_COM_NAME'].unique()\n\nif is_present:\n    print('The value \"African Bare-eyed Thrush\" is present in both train_metadata and taxonomy tables.')\nelse:\n    print('The value \"African Bare-eyed Thrush\" is not present in both train_metadata and taxonomy tables.')","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:13:03.036863Z","iopub.execute_input":"2023-05-16T10:13:03.037273Z","iopub.status.idle":"2023-05-16T10:13:03.048835Z","shell.execute_reply.started":"2023-05-16T10:13:03.037241Z","shell.execute_reply":"2023-05-16T10:13:03.047504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The names match, which means we can merge the datasets.","metadata":{}},{"cell_type":"code","source":"merged_df = pd.merge(train_metadata, taxonomy, left_on='scientific_name', right_on='SCI_NAME')\nmerged_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:13:41.832604Z","iopub.execute_input":"2023-05-16T10:13:41.833024Z","iopub.status.idle":"2023-05-16T10:13:41.970058Z","shell.execute_reply.started":"2023-05-16T10:13:41.832995Z","shell.execute_reply":"2023-05-16T10:13:41.968791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check how we merged the datasets on the unique values in 'primary_label' in train_metadata.","metadata":{}},{"cell_type":"code","source":"unique_labels = train_metadata['primary_label'].nunique()\nprint(\"Number of unique values (train_metadata):\", unique_labels)\n\nunique_labels = merged_df['primary_label'].nunique()\nprint(\"Number of unique values (merged_df):\", unique_labels)","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:15:15.073969Z","iopub.execute_input":"2023-05-16T10:15:15.074489Z","iopub.status.idle":"2023-05-16T10:15:15.086198Z","shell.execute_reply.started":"2023-05-16T10:15:15.074446Z","shell.execute_reply":"2023-05-16T10:15:15.084934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nLet's try to find the missing bird species.","metadata":{}},{"cell_type":"code","source":"train_labels = set(train_metadata['primary_label'])\nmerged_labels = set(merged_df['primary_label'])\nmissing_labels = train_labels - merged_labels\nprint(\"Missing bird :\", missing_labels)","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:16:09.801598Z","iopub.execute_input":"2023-05-16T10:16:09.802024Z","iopub.status.idle":"2023-05-16T10:16:09.813545Z","shell.execute_reply.started":"2023-05-16T10:16:09.801992Z","shell.execute_reply":"2023-05-16T10:16:09.812297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_metadata[train_metadata['primary_label'] == 'gnbcam2']","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:16:29.339879Z","iopub.execute_input":"2023-05-16T10:16:29.340305Z","iopub.status.idle":"2023-05-16T10:16:29.373069Z","shell.execute_reply.started":"2023-05-16T10:16:29.340271Z","shell.execute_reply":"2023-05-16T10:16:29.371843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check if the bird species name exists in either the common_name or scientific_name column of taxonomy.","metadata":{}},{"cell_type":"code","source":"# Checking for the presence of 'Camaroptera brevicaudata' or 'Gray-backed Camaroptera' in the PRIMARY_COM_NAME column\nif 'Camaroptera brevicaudata' in taxonomy['PRIMARY_COM_NAME'].values or 'Gray-backed Camaroptera' in taxonomy['PRIMARY_COM_NAME'].values:\n    print('At least one species found in the PRIMARY_COM_NAME column')\nelse:\n    print('No species found in the PRIMARY_COM_NAME column')\n\n# Checking for the presence of 'Camaroptera brevicaudata' or 'Gray-backed Camaroptera' in the SCI_NAME column\nif 'Camaroptera brevicaudata' in taxonomy['SCI_NAME'].values or 'Gray-backed Camaroptera' in taxonomy['SCI_NAME'].values:\n    print('At least one species found in the SCI_NAME column')\nelse:\n    print('No species found in the SCI_NAME column')","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:18:03.549338Z","iopub.execute_input":"2023-05-16T10:18:03.550222Z","iopub.status.idle":"2023-05-16T10:18:03.561392Z","shell.execute_reply.started":"2023-05-16T10:18:03.550185Z","shell.execute_reply":"2023-05-16T10:18:03.560219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"No worries, let's create a separate dataset.","metadata":{}},{"cell_type":"code","source":"df_gnbcam2 = train_metadata.loc[train_metadata['primary_label'] == 'gnbcam2', ['primary_label', 'latitude', 'longitude', 'scientific_name', 'filename']]\ndf_gnbcam2 = df_gnbcam2.assign(FAMILY=\"Other\")\ndf_gnbcam2","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:19:10.017516Z","iopub.execute_input":"2023-05-16T10:19:10.017899Z","iopub.status.idle":"2023-05-16T10:19:10.047313Z","shell.execute_reply.started":"2023-05-16T10:19:10.01787Z","shell.execute_reply":"2023-05-16T10:19:10.045716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nLet's create the final dataset with the selected features. We will remove the unnecessary columns below, as we won't be working with bird databases and their behaviors. Additionally, many columns have missing values that can only be filled using additional databases from ornithologists' websites.","metadata":{}},{"cell_type":"code","source":"df_selected = merged_df.drop(['TAXON_ORDER', 'CATEGORY', 'SPECIES_CODE', 'PRIMARY_COM_NAME', 'ORDER1','common_name' ,'SCI_NAME','secondary_labels','SPECIES_GROUP', 'REPORT_AS', 'type', 'author', 'url','rating', 'license'], axis=1)\ndf_selected.to_csv('/kaggle/working/df_selected.csv', index=False)\ndf_selected.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:21:42.673787Z","iopub.execute_input":"2023-05-16T10:21:42.674231Z","iopub.status.idle":"2023-05-16T10:21:42.863337Z","shell.execute_reply.started":"2023-05-16T10:21:42.674196Z","shell.execute_reply":"2023-05-16T10:21:42.862413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_selected","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:21:54.514891Z","iopub.execute_input":"2023-05-16T10:21:54.515421Z","iopub.status.idle":"2023-05-16T10:21:54.537806Z","shell.execute_reply.started":"2023-05-16T10:21:54.51538Z","shell.execute_reply":"2023-05-16T10:21:54.536547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now let's merge it with the missing bird species.","metadata":{}},{"cell_type":"code","source":"df_selected = pd.concat([df_selected, df_gnbcam2], axis=0).reset_index(drop=True)\ndf_selected","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:22:58.922974Z","iopub.execute_input":"2023-05-16T10:22:58.923377Z","iopub.status.idle":"2023-05-16T10:22:58.947349Z","shell.execute_reply.started":"2023-05-16T10:22:58.923347Z","shell.execute_reply":"2023-05-16T10:22:58.946275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's see where they are located on the map.","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nfig = px.scatter_mapbox(df_selected.dropna(),\n                        lat=\"latitude\",\n                        lon=\"longitude\",\n                        color=\"primary_label\",\n                        zoom=2,\n                        height=600)\n\nfig.update_layout(mapbox_style=\"open-street-map\", legend_title_text=\"Bird Species Code\")\n\nfrom PIL import Image\nimport IPython.display as display\n\n# Path to the PNG image file\nimage_path = \"/kaggle/input/mappng/newplot.png\"\n\n# Open the image\nimage = Image.open(image_path)\n\n# Display the image in the notebook\ndisplay.display(image)","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:55:43.289184Z","iopub.execute_input":"2023-05-16T10:55:43.289619Z","iopub.status.idle":"2023-05-16T10:55:44.323834Z","shell.execute_reply.started":"2023-05-16T10:55:43.289585Z","shell.execute_reply":"2023-05-16T10:55:44.322558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Filtering the data to show only birds within Kenya","metadata":{}},{"cell_type":"code","source":"df = df_selected[(df_selected['latitude'] >= -4.65) & (df_selected['latitude'] <= 4.75) & (df_selected['longitude'] >= 33.85) & (df_selected['longitude'] <= 41.5)]\n\n# Creating a scatter map plot\nfig = px.scatter_mapbox(df, lat=\"latitude\", lon=\"longitude\", zoom=5, height=500,color=\"primary_label\", hover_name=\"primary_label\",hover_data=[\"primary_label\"])\n\n# Setting the map style and centering it on Kenya\nfig.update_layout(mapbox_style=\"carto-positron\", mapbox_center_lon=37.9062,mapbox_center_lat=-0.0236, mapbox_zoom=6)\n\n# Setting the hoverlabel_align parameter to display hover labels to the right of the points\nfig.update_traces(hoverlabel_align='right')\n\nfig.update_layout(legend_title_text='Bird Species Code')\n\n# Path to the PNG image file\nimage_path = \"/kaggle/input/map-kenya/newplot (1).png\"\n\n# Open the image\nimage = Image.open(image_path)\n\n# Display the image in the notebook\ndisplay.display(image)","metadata":{"execution":{"iopub.status.busy":"2023-05-16T10:59:29.986126Z","iopub.execute_input":"2023-05-16T10:59:29.986584Z","iopub.status.idle":"2023-05-16T10:59:31.460009Z","shell.execute_reply.started":"2023-05-16T10:59:29.986548Z","shell.execute_reply":"2023-05-16T10:59:31.458789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nLet's extract them into a separate dataset and check the number of unique primary_labels.","metadata":{}},{"cell_type":"code","source":"df_in_Kenya = df.copy()\ndf_in_Kenya.to_csv('/kaggle/working/df_in_Kenya.csv', index=False)\ndf_in_Kenya","metadata":{"execution":{"iopub.status.busy":"2023-05-16T11:01:36.477153Z","iopub.execute_input":"2023-05-16T11:01:36.477596Z","iopub.status.idle":"2023-05-16T11:01:36.527612Z","shell.execute_reply.started":"2023-05-16T11:01:36.47756Z","shell.execute_reply":"2023-05-16T11:01:36.526483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_labels = df['primary_label'].nunique()\nprint(\"Unique df:\", unique_labels)\n\n# Unique in the dataset with features\nunique_labels_df_selected = df_selected['primary_label'].nunique()\nprint(\"Unique df_selected:\", unique_labels_df_selected)","metadata":{"execution":{"iopub.status.busy":"2023-05-16T11:03:36.478609Z","iopub.execute_input":"2023-05-16T11:03:36.47903Z","iopub.status.idle":"2023-05-16T11:03:36.488102Z","shell.execute_reply.started":"2023-05-16T11:03:36.478999Z","shell.execute_reply":"2023-05-16T11:03:36.486976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Finding missing labels\nunique_labels = df['primary_label'].unique()\nunique_labels_selected = df_selected['primary_label'].unique()\n\nmissing_labels = []\nfor label in unique_labels_selected:\n    if label not in unique_labels:\n        missing_labels.append(label)\n\nprint(\"Number of missing labels:\", len(missing_labels))\nprint(\"Labels:\", missing_labels)","metadata":{"execution":{"iopub.status.busy":"2023-05-16T11:04:17.180221Z","iopub.execute_input":"2023-05-16T11:04:17.180622Z","iopub.status.idle":"2023-05-16T11:04:17.198028Z","shell.execute_reply.started":"2023-05-16T11:04:17.180592Z","shell.execute_reply":"2023-05-16T11:04:17.196516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Found and save to a separate dataset\ndf_not_in_Kenya = df_selected[df_selected['primary_label'].isin(missing_labels)]\ndf_not_in_Kenya.to_csv('/kaggle/working/not_in_Kenya.csv', index=False)\ndf_not_in_Kenya","metadata":{"execution":{"iopub.status.busy":"2023-05-16T11:05:04.613977Z","iopub.execute_input":"2023-05-16T11:05:04.615328Z","iopub.status.idle":"2023-05-16T11:05:04.671705Z","shell.execute_reply.started":"2023-05-16T11:05:04.615273Z","shell.execute_reply":"2023-05-16T11:05:04.670265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove duplicates based on coordinates\nunique_coords_not_in_Kenya = df_not_in_Kenya.drop_duplicates(subset=['latitude', 'primary_label', 'longitude'], keep='first')\nunique_coords_not_in_Kenya.to_csv('/kaggle/working/unique_coords_not_in_Kenya.csv', index=False)\n\nunique_coords_not_in_Kenya","metadata":{"execution":{"iopub.status.busy":"2023-05-16T11:06:29.98475Z","iopub.execute_input":"2023-05-16T11:06:29.98514Z","iopub.status.idle":"2023-05-16T11:06:30.027203Z","shell.execute_reply.started":"2023-05-16T11:06:29.985111Z","shell.execute_reply":"2023-05-16T11:06:30.026019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_coords_in_Kenya = df_in_Kenya.drop_duplicates(subset=['latitude', 'primary_label', 'longitude'], keep='first')\nunique_coords_in_Kenya.to_csv('/kaggle/working/unique_coords_in_Kenya.csv', index=False)\nunique_coords_in_Kenya","metadata":{"execution":{"iopub.status.busy":"2023-05-16T11:06:49.412782Z","iopub.execute_input":"2023-05-16T11:06:49.413197Z","iopub.status.idle":"2023-05-16T11:06:49.458939Z","shell.execute_reply.started":"2023-05-16T11:06:49.413169Z","shell.execute_reply":"2023-05-16T11:06:49.457808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_labels = unique_coords_not_in_Kenya['primary_label'].nunique()\nprint(\"Number of unique labels:\", unique_labels)","metadata":{"execution":{"iopub.status.busy":"2023-05-16T11:07:15.069712Z","iopub.execute_input":"2023-05-16T11:07:15.070154Z","iopub.status.idle":"2023-05-16T11:07:15.077121Z","shell.execute_reply.started":"2023-05-16T11:07:15.070122Z","shell.execute_reply":"2023-05-16T11:07:15.075759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_labels = unique_coords_in_Kenya['primary_label'].nunique()\nprint(\"Number of unique labels:\", unique_labels)","metadata":{"execution":{"iopub.status.busy":"2023-05-16T11:07:36.02729Z","iopub.execute_input":"2023-05-16T11:07:36.027733Z","iopub.status.idle":"2023-05-16T11:07:36.035616Z","shell.execute_reply.started":"2023-05-16T11:07:36.0277Z","shell.execute_reply":"2023-05-16T11:07:36.034271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Everything is fine, all bird species are in place. As a bonus, we have two types of datasets with different numbers of links to audio data","metadata":{}},{"cell_type":"markdown","source":"# Let's examine the Audio","metadata":{}},{"cell_type":"markdown","source":"A **spectrogram** is a visualization of the sound spectrum over time. It shows which frequency components are present in the audio file at a given moment. It is commonly used for sound analysis and highlighting the main frequency characteristics.\n\nA **waveform** is a visualization of the sound wave's oscillations over time. It shows how the sound's amplitude changes over time. It is commonly used for sound analysis and identifying features of the sound spectrum, such as loudness and frequency.\n\n**Both graphs** can help in understanding the structure and characteristics of an audio file, which can be useful for sound analysis and determining the sounds contained within the file.","metadata":{}},{"cell_type":"code","source":"import librosa\nimport librosa.display\n\n# Path to test_soundscapes\ntest_soundscapes = '/kaggle/input/birdclef-2023/test_soundscapes'\n\n# List files in the directory\nfiles = os.listdir(test_soundscapes)\n\n# Load audio and display spectrogram\nfor file in files:\n    signal, sr = librosa.load(os.path.join(test_soundscapes, file), sr=None)\n    plt.figure(figsize=(12, 3))\n    plt.title('Spectrogram of Test Audio')\n    librosa.display.specshow(librosa.amplitude_to_db(librosa.stft(signal), ref=np.max), sr=sr, x_axis='time', y_axis='log')\n    plt.colorbar(format='%+2.0f dB')\n    plt.xlabel('Time')\n    plt.ylabel('Frequency (Hz)')\n    plt.tight_layout()\n    plt.show()\n\n    # Display waveform\n    plt.figure(figsize=(12, 3))\n    plt.title('Waveform of Test Audio')\n    librosa.display.waveshow(signal, sr=sr, alpha=0.5)\n    plt.xlabel('Time')\n    plt.ylabel('Amplitude')\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-16T11:13:33.455086Z","iopub.execute_input":"2023-05-16T11:13:33.455683Z","iopub.status.idle":"2023-05-16T11:14:16.618662Z","shell.execute_reply.started":"2023-05-16T11:13:33.455639Z","shell.execute_reply":"2023-05-16T11:14:16.617583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Path to the training audio folder\ntrain_audio = '/kaggle/input/birdclef-2023/train_audio/'\n\n# List of directory paths\ndir_paths = ['/kaggle/input/birdclef-2023/train_audio/abethr1', '/kaggle/input/birdclef-2023/train_audio/abhori1']\n\nfor dir_path in dir_paths:\n    audio_files = [f for f in os.listdir(dir_path) if f.endswith(\".ogg\")]\n    if audio_files:\n        file_name = audio_files[0] # First file in the folder\n        file_path = os.path.join(dir_path, file_name)\n        print(\"Path:\", file_path)\n\n    # Load audio and display spectrogram\n    signal, sr = librosa.load(file_path, sr=None)\n    plt.figure(figsize=(12, 3))\n    plt.title(f'Spectrogram of {file_name} from {os.path.basename(dir_path)}')\n    librosa.display.specshow(librosa.amplitude_to_db(librosa.stft(signal), ref=np.max), sr=sr, x_axis='time', y_axis='log')\n    plt.colorbar(format='%+2.0f dB')\n    plt.xlabel('Time')\n    plt.ylabel('Frequency (Hz)')\n    plt.tight_layout()\n    plt.show()\n    \n    # Display waveform\n    plt.figure(figsize=(12, 3))\n    plt.title(f'Waveform of {file_name} from {os.path.basename(dir_path)}')\n    librosa.display.waveshow(signal, sr=sr, alpha=0.5)\n    plt.xlabel('Time')\n    plt.ylabel('Amplitude')\n    plt.show()\n    \n","metadata":{"execution":{"iopub.status.busy":"2023-05-16T11:19:35.877721Z","iopub.execute_input":"2023-05-16T11:19:35.878157Z","iopub.status.idle":"2023-05-16T11:19:41.213875Z","shell.execute_reply.started":"2023-05-16T11:19:35.878117Z","shell.execute_reply":"2023-05-16T11:19:41.212231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Spectrograms show the change in energy of a sound signal in both time and frequency domains. The vertical axis represents frequency, the horizontal axis represents time, and the color represents the energy level. Therefore, spectrograms allow us to see which frequencies dominate the sound at a given moment and how these frequencies change over time.\n\nTo interpret spectrograms more accurately, it is necessary to study the theory of sound processing and signal analysis. This may require additional knowledge in mathematics and physics.\n\nIt is important to keep in mind that spectrograms can be subject to various types of noise and artifacts, which can distort the representation of the sound signal. Therefore, to obtain more accurate results, signal processing techniques, including filtering and noise suppression, need to be applied.\n","metadata":{}},{"cell_type":"markdown","source":"**Conclusion**\n\n* The dataset eBird_Taxonomy_v2021.csv is practically useless, as a significant portion of the useful data is duplicated in the analog train_metadata. There are no meaningful interactions between the bird species. In my opinion, it would be more useful to add descriptions of the species based on their behavior. For example, whether they are solitary or not, their migration patterns throughout the seasons, and the territories they occupy. With such information, the coordinate data could be utilized more effectively.\n\n* The audio files for each species are unbalanced, with some having 500 audio samples while others have only one.\n\n* An erroneous labeling of one audio file under two different species was detected, suggesting that there may be other labeling errors in the dataset.\n\nWe will try to address the given task by reducing the audio file dataset.","metadata":{}},{"cell_type":"markdown","source":"# Preparing datasets for training\nWe will create datasets based on the coordinates of the audio files.\n\n**Dataset 1**: All unique coordinates within Kenya + unique coordinates outside Kenya\n\n**Dataset 2**: All coordinates within Kenya + unique coordinates outside Kenya within Kenya\n\nWe will use these datasets for further analysis and training.","metadata":{}},{"cell_type":"code","source":"# Dataset 1\n# Data merging\nall_unique_coords = pd.concat([unique_coords_in_Kenya, unique_coords_not_in_Kenya])\n\n# Add path to each value in the 'filename' column\nall_unique_coords['filename'] = all_unique_coords['filename'].apply(lambda x: '/kaggle/input/birdclef-2023/train_audio' + x)\nall_unique_coords","metadata":{"execution":{"iopub.status.busy":"2023-05-16T11:27:18.856659Z","iopub.execute_input":"2023-05-16T11:27:18.857083Z","iopub.status.idle":"2023-05-16T11:27:18.885623Z","shell.execute_reply.started":"2023-05-16T11:27:18.857054Z","shell.execute_reply":"2023-05-16T11:27:18.884286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving the result to a file\nall_unique_coords.to_csv('/kaggle/working/all_unique_coords.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-16T11:28:20.842688Z","iopub.execute_input":"2023-05-16T11:28:20.843097Z","iopub.status.idle":"2023-05-16T11:28:20.894257Z","shell.execute_reply.started":"2023-05-16T11:28:20.843068Z","shell.execute_reply":"2023-05-16T11:28:20.893002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine all coordinates within approximate Kenya boundaries + unique coordinates outside Kenya (all_Kenya)\nall_Kenya = pd.concat([df_in_Kenya, df_not_in_Kenya])\n\n# Add the file path to each value in the 'filename' column\nall_Kenya['filename'] = all_Kenya['filename'].apply(lambda x: '/kaggle/input/birdclef-2023/train_audio/' + x)\n\nall_Kenya","metadata":{"execution":{"iopub.status.busy":"2023-05-16T11:30:45.151452Z","iopub.execute_input":"2023-05-16T11:30:45.152809Z","iopub.status.idle":"2023-05-16T11:30:45.188063Z","shell.execute_reply.started":"2023-05-16T11:30:45.152751Z","shell.execute_reply":"2023-05-16T11:30:45.186883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_Kenya.to_csv('/kaggle/working/all_Kenya.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-16T11:31:08.86238Z","iopub.execute_input":"2023-05-16T11:31:08.862853Z","iopub.status.idle":"2023-05-16T11:31:08.938133Z","shell.execute_reply.started":"2023-05-16T11:31:08.862819Z","shell.execute_reply":"2023-05-16T11:31:08.93706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Models and Training","metadata":{}},{"cell_type":"markdown","source":"Summary of audio file features, CNN model, and training metrics.\n\nThe audio file features for the main datasets, all_unique_coords and all_Kenya, will be extracted using the librosa library.\n\n**Parameters for extracting features from audio files**:\n\n* max_len: Maximum length of features. If the length of extracted features is less than max_len, they will be padded with zero values to the required length.\n\n* DURATION: Duration of the audio segment set to 5 seconds.\n\n* SAMPLE_RATE: Sample rate of the audio file set to 32,000 samples per second.\n\n* max_len: Maximum length of features initially set to 128.\n\n* n_fft: Size of the window in the Fast Fourier Transform (FFT) algorithm when computing spectrograms. In this case, it is set to 2048.\n\n* hop_length: Hop length for window overlapping when computing spectrograms. Set to 512, which corresponds to 75% window overlap.\n\n* n_mels: Number of Mel filters when computing Mel spectrograms. Set to 128.\n\n\n","metadata":{}},{"cell_type":"markdown","source":"****CNN Model****.\n\n*Architecture*:\n\n* Number of convolutional layers: 4\n* Number of filters for the first two convolutional layers: 32\n* Kernel size for the first two convolutional layers: 3x3\n* Activation function for the first two convolutional layers: ReLU\n* Number of filters for the last two convolutional layers: 64\n* Kernel size for the last two convolutional layers: 3x3\n* Activation function for the last two convolutional layers: ReLU\n* Pooling window size: 2x2\n* Dropout rate for the first Dropout layer: 0.25\n* Dropout rate for the second Dropout layer: 0.5\n* Number of neurons for the fully connected layer: 512\n* Activation function for the fully connected layer: ReLU\n* Activation function for the output layer: softmax\n\n*Model compilation*:\n\n* Loss function: 'categorical_crossentropy' or 'sparse_categorical_crossentropy'\n* Optimizer: 'adam' (Adam optimization algorithm).\n* Metrics: accuracy, precision, and recall.\n\n*Model training*:\n\n* Batch size: 64.\n* Number of epochs: 25.\n\n*Training results*:\n\n* Accuracy indicates the proportion of examples that were classified correctly out of the total number of examples.\n* Loss is a measure of the discrepancy between the true labels and the predicted labels by the model.\n\nAccuracy is important for evaluating the model's performance, while Loss is important for optimization. The goal of model training is to minimize the Loss value, i.e., reduce the discrepancy between the true labels and the predicted labels on the training set. However, Accuracy is also important for understanding how well the model is performing the classification task.","metadata":{}},{"cell_type":"markdown","source":"**CNN** for the dataset of **all bird coordinates** within approximate boundaries of Kenya + outside of Kenya (**all_Kenya**)","metadata":{}},{"cell_type":"markdown","source":"**Feature Extraction**","metadata":{}},{"cell_type":"code","source":"from tqdm import tqdm\n\n# Dataset Path\nCSV_PATH = \"/kaggle/working/all_Kenya.csv\"\n\n# Audio Parameters\nDURATION = 5 # Duration in seconds\nSAMPLE_RATE = 32000 # Number of samples per second\nmax_len = 128 # Maximum length\n\n# Function to extract audio features\ndef extract_features(file_path, max_len):\n    signal, sr = librosa.load(file_path, sr=SAMPLE_RATE, duration=DURATION)\n    mel_spec = librosa.feature.melspectrogram(y=signal, sr=sr, n_fft=2048, hop_length=512, n_mels=128)\n    mel_spec = librosa.power_to_db(mel_spec, ref=np.max)\n    pad_width = max_len - mel_spec.shape[1]\n    if pad_width > 0:\n        mel_spec = np.pad(mel_spec, pad_width=((0, 0), (0, pad_width)), mode='constant')\n    return mel_spec\n\n# Load the dataset\ndata = pd.read_csv(CSV_PATH)\n\n# Extraction\nX = []\ny = []\nfor index, row in tqdm(data.iterrows(), total=len(data)):\n    file_path = row['filename']\n    target = row['primary_label']\n    features = extract_features(file_path, max_len)\n    X.append(features)\n    y.append(target)\n    if features.shape[1] > max_len:\n        max_len = features.shape[1]","metadata":{"execution":{"iopub.status.busy":"2023-05-16T11:45:04.859812Z","iopub.execute_input":"2023-05-16T11:45:04.860234Z","iopub.status.idle":"2023-05-16T11:51:29.302568Z","shell.execute_reply.started":"2023-05-16T11:45:04.860204Z","shell.execute_reply":"2023-05-16T11:51:29.300981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving Features\nnp.save('/kaggle/working/features_all_Kenya.npy', X)\n\n# Saving Labels\nnp.save('/kaggle/working/labels_all_Kenya.npy', y)","metadata":{"execution":{"iopub.status.busy":"2023-05-16T11:52:01.906385Z","iopub.execute_input":"2023-05-16T11:52:01.906807Z","iopub.status.idle":"2023-05-16T11:52:03.288786Z","shell.execute_reply.started":"2023-05-16T11:52:01.906776Z","shell.execute_reply":"2023-05-16T11:52:03.287451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Models**","metadata":{}},{"cell_type":"code","source":"# CNN + loss function \"categorical_crossentropy\"\n\nimport pandas as pd\nimport numpy as np\nimport librosa\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, Activation, Flatten, Conv2D, MaxPooling2D\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nimport tensorflow as tf\n\n# Data loading\nX = np.load('/kaggle/working/features_all_Kenya.npy')\ny = np.load('/kaggle/working/labels_all_Kenya.npy')\n\n# Labels\nlabel_encoder = LabelEncoder()\ny = label_encoder.fit_transform(y)\ny = to_categorical(y)\n\n# Convert to arrays\nX = np.array(X)\ny = np.array(y)\n\n# Split data into training and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Model architecture\nmodel = Sequential()\nmodel.add(Conv2D(32, (3, 3), padding='same', input_shape=(X_train.shape[1], X_train.shape[2], 1))) # исправьте форму входных данных\nmodel.add(Activation('relu'))\nmodel.add(Conv2D(32, (3, 3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Conv2D(64, (3, 3), padding='same'))\nmodel.add(Activation('relu'))\nmodel.add(Conv2D(64, (3, 3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Flatten())\nmodel.add(Dense(512))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(y.shape[1]))\nmodel.add(Activation('softmax'))\n\nmetrics = ['accuracy', tf.keras.metrics.Precision(), tf.keras.metrics.Recall()]\n\n# Compile the model using categorical crossentropy loss function and Adam optimizer\nmodel.compile(loss='categorical_crossentropy', optimizer='adam', metrics=metrics)\n\n# Path to save the trained model\nMODEL_PATH = \"/kaggle/working/all_Kenya_categorical_cross.h5\"\n\n# ModelCheckpoint callback to save the best model during training\ncheckpoint = ModelCheckpoint(MODEL_PATH, monitor='accuracy', verbose=1, save_best_only=True, mode='max')\n\n# Train the model\nhistory = model.fit(X_train, y_train, batch_size=64, epochs=25, validation_data=(X_test, y_test), callbacks=[checkpoint])\n","metadata":{"execution":{"iopub.status.busy":"2023-05-16T11:54:25.461581Z","iopub.execute_input":"2023-05-16T11:54:25.46204Z","iopub.status.idle":"2023-05-16T16:37:51.598241Z","shell.execute_reply.started":"2023-05-16T11:54:25.462008Z","shell.execute_reply":"2023-05-16T16:37:51.593786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CNN + loss function \"sparse_categorical_crossentropy\"\n\nimport pandas as pd\nimport numpy as np\nimport librosa\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.layers import Dense, Dropout, Activation, Flatten, Conv2D, MaxPooling2D\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nimport tensorflow as tf\n\n# Data loading\nX = np.load('/kaggle/working/features_all_Kenya.npy')\ny = np.load('/kaggle/working/labels_all_Kenya.npy')\n\n# Labels\nlabel_encoder = LabelEncoder()\ny = label_encoder.fit_transform(y)\ny = y.reshape(-1, 1) # Reshape labels to the required format\n\n# Number of classes\nnum_classes = len(label_encoder.classes_)\n\n# Convert to arrays\nX = np.array(X)\ny = np.array(y).ravel() # используем метод ravel(), чтобы преобразовать метки в нужный формат\n\n# Split into training and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Model architecture\nmodel = Sequential()\nmodel.add(Conv2D(32, (3, 3), padding='same', input_shape=(X.shape[1], X.shape[2], 1)))\nmodel.add(Activation('relu'))\nmodel.add(Conv2D(32, (3, 3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Conv2D(64, (3, 3), padding='same'))\nmodel.add(Activation('relu'))\nmodel.add(Conv2D(64, (3, 3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Flatten())\nmodel.add(Dense(512))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(264))\nmodel.add(Activation('softmax'))\n\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy', 'sparse_categorical_accuracy'])\n\n# Path to save the trained model\nMODEL_PATH = \"/kaggle/working/all_Kenya_sparse_categorical_crossentropy.h5\"\n\n# ModelCheckpoint callback to save the best model during training\ncheckpoint = ModelCheckpoint(MODEL_PATH, monitor='accuracy', verbose=1, save_best_only=True, mode='max')\n\n# Train the model\nhistory = model.fit(X_train, y_train, batch_size=64, epochs=25, validation_data=(X_test, y_test), callbacks=[checkpoint])\n","metadata":{"execution":{"iopub.status.busy":"2023-05-16T16:41:44.279514Z","iopub.execute_input":"2023-05-16T16:41:44.279986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**CNN** for the dataset of all unique bird coordinates within and outside of Kenya (**all_unique_coords**)\n\n**Feature Extraction**","metadata":{}},{"cell_type":"code","source":"# Dataset Path\nCSV_PATH = '/kaggle/working/all_unique_coords.csv'\n\n# Audio Parameters\nDURATION = 5 # Duration in seconds\nSAMPLE_RATE = 32000 # Number of samples per second\nmax_len = 128 # Maximum length\n\n# Function to extract audio features\ndef extract_features(file_path, max_len):\n    signal, sr = librosa.load(file_path, sr=SAMPLE_RATE, duration=DURATION)\n    mel_spec = librosa.feature.melspectrogram(y=signal, sr=sr, n_fft=2048, hop_length=512, n_mels=128)\n    mel_spec = librosa.power_to_db(mel_spec, ref=np.max)\n    pad_width = max_len - mel_spec.shape[1]\n    if pad_width > 0:\n        mel_spec = np.pad(mel_spec, pad_width=((0, 0), (0, pad_width)), mode='constant')\n    return mel_spec\n\n# Load the dataset\ndata = pd.read_csv(CSV_PATH)\n\n# Extraction\nX = []\ny = []\nfor index, row in tqdm(data.iterrows(), total=len(data)):\n    file_path = row['filename']\n    target = row['primary_label']\n    features = extract_features(file_path, max_len)\n    X.append(features)\n    y.append(target)\n    if features.shape[1] > max_len:\n        max_len = features.shape[1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving Features\nnp.save('/kaggle/working/features_all_unique_coords.npy', X)\n\n# Saving Labels\nnp.save('/kaggle/working/labels_all_unique_coords.npy', y)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Models**","metadata":{}},{"cell_type":"code","source":"# CNN + loss function \"categorical_crossentropy\"\n\n# Data loading\nX = np.load('/kaggle/working/features_all_unique_coords.npy')\ny = np.load('/kaggle/working/labels_all_unique_coords.npy')\n\n# Labels\nlabel_encoder = LabelEncoder()\ny = label_encoder.fit_transform(y)\ny = to_categorical(y)\n\n# Convert to arrays\nX = np.array(X)\ny = np.array(y)\n\n# Split data into training and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Model architecture\nmodel = Sequential()\nmodel.add(Conv2D(32, (3, 3), padding='same', input_shape=(X_train.shape[1], X_train.shape[2], 1))) # исправьте форму входных данных\nmodel.add(Activation('relu'))\nmodel.add(Conv2D(32, (3, 3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Conv2D(64, (3, 3), padding='same'))\nmodel.add(Activation('relu'))\nmodel.add(Conv2D(64, (3, 3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Flatten())\nmodel.add(Dense(512))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(y.shape[1]))\nmodel.add(Activation('softmax'))\n\nmetrics = ['accuracy', tf.keras.metrics.Precision(), tf.keras.metrics.Recall()]\n\n# Compile the model using categorical crossentropy loss function and Adam optimizer\nmodel.compile(loss='categorical_crossentropy', optimizer='adam', metrics=metrics)\n\n# Path to save the trained model\nMODEL_PATH = \"/kaggle/working/all_unique_coords_categorical_cross.h5\"\n\n# ModelCheckpoint callback to save the best model during training\ncheckpoint = ModelCheckpoint(MODEL_PATH, monitor='accuracy', verbose=1, save_best_only=True, mode='max')\n\n# Train the model\nhistory = model.fit(X_train, y_train, batch_size=64, epochs=25, validation_data=(X_test, y_test), callbacks=[checkpoint])\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CNN + loss function \"sparse_categorical_crossentropy\"\n\n# Data loading\nX = np.load('/kaggle/working/features_all_unique_coords.npy')\ny = np.load('/kaggle/working/labels_all_unique_coords.npy')\n\n# Labels\nlabel_encoder = LabelEncoder()\ny = label_encoder.fit_transform(y)\ny = y.reshape(-1, 1) # Reshape labels to the required format\n\n# Number of classes\nnum_classes = len(label_encoder.classes_)\n\n# Convert to arrays\nX = np.array(X)\ny = np.array(y).ravel() # используем метод ravel(), чтобы преобразовать метки в нужный формат\n\n# Split into training and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Model architecture\nmodel = Sequential()\nmodel.add(Conv2D(32, (3, 3), padding='same', input_shape=(X.shape[1], X.shape[2], 1)))\nmodel.add(Activation('relu'))\nmodel.add(Conv2D(32, (3, 3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Conv2D(64, (3, 3), padding='same'))\nmodel.add(Activation('relu'))\nmodel.add(Conv2D(64, (3, 3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Flatten())\nmodel.add(Dense(512))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(264))\nmodel.add(Activation('softmax'))\n\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy', 'sparse_categorical_accuracy'])\n\n# Path to save the trained model\nMODEL_PATH = \"/kaggle/working/all_unique_coords_sparse_categorical_crossentropy.h5\"\n\n# ModelCheckpoint callback to save the best model during training\ncheckpoint = ModelCheckpoint(MODEL_PATH, monitor='accuracy', verbose=1, save_best_only=True, mode='max')\n\n# Train the model\nhistory = model.fit(X_train, y_train, batch_size=64, epochs=25, validation_data=(X_test, y_test), callbacks=[checkpoint])\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For the model with the best accuracy metrics (dataset: all_unique_coords, loss function: sparse_categorical_crossentropy), we will make a submission.","metadata":{}},{"cell_type":"code","source":"import librosa\nimport pandas as pd\nimport numpy as np\nfrom tensorflow.keras.models import load_model\n\n# Load the model\nmodel = load_model('/kaggle/input/datasets-and-models/all_unique_coords/all_unique_coords_sparse_categorical_crossentropy.h5')\n\n# Load the test audio files\ntest_file_path = '/kaggle/input/birdclef-2023/test_soundscapes/soundscape_29201.ogg'\n\n# Prepare the submission DataFrame\nsubmit_df = pd.read_csv('/kaggle/input/birdclef-2023/sample_submission.csv')\nsubmit_df.iloc[:, 1:] = 0\n\n# Process each test audio file\nfor file_path in test_files:\n    # Load the file\n    y, sr = librosa.load(file_path, sr=32000, res_type='kaiser_fast')\n    duration = librosa.get_duration(y=y, sr=sr)\n\n    # Create intervals of 5 seconds with overlap\n    interval_duration = 5\n    overlap = 0.5\n    intervals = np.arange(0, duration - interval_duration, interval_duration * (1 - overlap))\n    intervals = np.append(intervals, duration - interval_duration)\n    interval_df = pd.DataFrame({'t_min': intervals, 't_max': intervals + interval_duration})\n\n    # Process each interval\n    for i, row in interval_df.iterrows():\n        t_min = row['t_min']\n        t_max = row['t_max']\n\n        # Extract MFCC features\n        interval = y[int(t_min*sr):int(t_max*sr)]\n        mfccs = librosa.feature.mfcc(y=interval, sr=sr, n_mfcc=128, n_fft=2048)\n        mfccs_scaled = (mfccs - np.mean(mfccs, axis=0)) / np.std(mfccs, axis=0)\n        mfccs_scaled = np.expand_dims(mfccs_scaled, axis=-1)\n\n        preds = model.predict(np.array([mfccs_scaled]), verbose=0)\n\n        row_id = file_path.split(\"/\")[-1].replace(\".ogg\", f\"_{i}\")\n        submit_df.loc[submit_df.row_id == row_id, submit_df.columns[1:]] = preds[0]\n\n# Save the submission DataFrame\nsubmit_df.to_csv(\"/kaggle/working/submission_all_unique_coords_sparse_categorical_crossentropy.csv\", index=False)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit_df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nimport pandas as pd\nimport numpy as np\nfrom tensorflow.keras.models import load_model\n\n# Load the model\nmodel = load_model('/content/drive/MyDrive/all_unique_coords_sparse_categorical_crossentropy.h5')\n\n# Specify the path to the test audio file\ntest_file_path = '/kaggle/input/birdclef-2023/test_soundscapes/soundscape_29201.ogg'\n\n# Prepare the submission DataFrame\nsubmit_df = pd.read_csv('/kaggle/input/birdclef-2023/sample_submission.csv')\nsubmit_df.iloc[:, 1:] = 0\n\n# Load the test audio file\ny, sr = librosa.load(test_file_path, sr=32000, res_type='kaiser_fast')\nduration = librosa.get_duration(y=y, sr=sr)\n\n# Create intervals of 5 seconds with overlap\ninterval_duration = 5\noverlap = 0.5\nintervals = np.arange(0, duration - interval_duration, interval_duration * (1 - overlap))\nintervals = np.append(intervals, duration - interval_duration)\ninterval_df = pd.DataFrame({'t_min': intervals, 't_max': intervals + interval_duration})\n\n# Process each interval\nfor i, row in interval_df.iterrows():\n    t_min = row['t_min']\n    t_max = row['t_max']\n\n    # Extract MFCC features\n    interval = y[int(t_min*sr):int(t_max*sr)]\n    mfccs = librosa.feature.mfcc(y=interval, sr=sr, n_mfcc=128, n_fft=2048)\n    mfccs_scaled = (mfccs - np.mean(mfccs, axis=0)) / np.std(mfccs, axis=0)\n    mfccs_scaled = np.expand_dims(mfccs_scaled, axis=-1)\n\n    preds = model.predict(np.array([mfccs_scaled]), verbose=0)\n\n    row_id = test_file_path.split(\"/\")[-1].replace(\".ogg\", f\"_{i}\")\n    submit_df.loc[submit_df.row_id == row_id, submit_df.columns[1:]] = preds[0]\n\n# Save the submission DataFrame\nsubmit_df.to_csv(\"/kaggle/working/submission_all_unique_coords_sparse_categorical_crossentropy.csv\", index=False)\n","metadata":{},"execution_count":null,"outputs":[]}]}