{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## The AIM of this post is to understand and analyze the provided data in the competitions.","metadata":{}},{"cell_type":"markdown","source":"![sholicola-ian_lockwood_inset](https://github.com/skj092/kaggle-BirdCLEF-2024/assets/43055935/2ea97f9e-0aa2-400b-af34-ad8401018323)","metadata":{}},{"cell_type":"code","source":"!pip install fastbook -q","metadata":{"execution":{"iopub.status.busy":"2024-05-02T03:07:18.012043Z","iopub.execute_input":"2024-05-02T03:07:18.012433Z","iopub.status.idle":"2024-05-02T03:07:37.118421Z","shell.execute_reply.started":"2024-05-02T03:07:18.012403Z","shell.execute_reply":"2024-05-02T03:07:37.116860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert sound of bird to spectrogram\nfrom fastai.vision.all import Path, get_files\nimport soundfile as sf\nimport librosa as lb\nimport librosa.display as lbd\nfrom IPython.display import Audio\nfrom soundfile import SoundFile\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport random\nimport os\nimport pandas as pd \nfrom fastbook import *\nfrom IPython.display import Image, display, Audio, Markdown\nimport plotly.express as px","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2024-05-02T03:07:37.121737Z","iopub.execute_input":"2024-05-02T03:07:37.122280Z","iopub.status.idle":"2024-05-02T03:07:38.155535Z","shell.execute_reply.started":"2024-05-02T03:07:37.122230Z","shell.execute_reply":"2024-05-02T03:07:38.154418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    sampling_rate = 32000\n    duration = 5\n    fmin = 0\n    fmax = None\n    data_dir = Path(\"/kaggle/input/birdclef-2024\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-05-02T03:07:38.156918Z","iopub.execute_input":"2024-05-02T03:07:38.157258Z","iopub.status.idle":"2024-05-02T03:07:38.163419Z","shell.execute_reply.started":"2024-05-02T03:07:38.157231Z","shell.execute_reply":"2024-05-02T03:07:38.162150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n**Some utility functions**","metadata":{}},{"cell_type":"code","source":"def get_audio_info(filepath):\n    \"\"\"Get some properties from  an audio file\"\"\"\n    with SoundFile(filepath) as f:\n        sr = f.samplerate\n        frames = f.frames\n        duration = float(frames)/sr\n    return {\"frames\": frames, \"sr\": sr, \"duration\": duration}","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-05-02T03:07:38.164899Z","iopub.execute_input":"2024-05-02T03:07:38.165832Z","iopub.status.idle":"2024-05-02T03:07:38.178669Z","shell.execute_reply.started":"2024-05-02T03:07:38.165788Z","shell.execute_reply":"2024-05-02T03:07:38.177420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def compute_melspec(y, sr, n_mels, fmin, fmax):\n    \"\"\"\n    Computes a mel-spectrogram and puts it at decibel scale\n    Arguments:\n        y {np array} -- signal\n        params {AudioParams} -- Parameters to use for the spectrogram. Expected to have the attributes sr, n_mels, f_min, f_max\n    Returns:\n        np array -- Mel-spectrogram\n    \"\"\"\n    melspec = lb.feature.melspectrogram(\n        y=y, sr=sr, n_mels=n_mels, fmin=fmin, fmax=fmax,\n    )\n\n    melspec = lb.power_to_db(melspec).astype(np.float32)\n    return melspec","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-05-02T03:07:38.182771Z","iopub.execute_input":"2024-05-02T03:07:38.183192Z","iopub.status.idle":"2024-05-02T03:07:38.197602Z","shell.execute_reply.started":"2024-05-02T03:07:38.183159Z","shell.execute_reply":"2024-05-02T03:07:38.196351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mono_to_color(X, eps=1e-6, mean=None, std=None):\n    mean = mean or X.mean()\n    std = std or X.std()\n    X = (X - mean) / (std + eps)\n\n    _min, _max = X.min(), X.max()\n\n    if (_max - _min) > eps:\n        V = np.clip(X, _min, _max)\n        V = 255 * (V - _min) / (_max - _min)\n        V = V.astype(np.uint8)\n    else:\n        V = np.zeros_like(X, dtype=np.uint8)\n\n    return V","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-05-02T03:07:38.199110Z","iopub.execute_input":"2024-05-02T03:07:38.199613Z","iopub.status.idle":"2024-05-02T03:07:38.211963Z","shell.execute_reply.started":"2024-05-02T03:07:38.199557Z","shell.execute_reply":"2024-05-02T03:07:38.210843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sr, n_mels, fmin, fmax  = Config.sampling_rate, 128, Config.fmin, Config.fmax\ndef audio_to_image(audio):\n    melspec = compute_melspec(audio, sr=sr, n_mels = n_mels, fmin=fmin, fmax=fmax)\n    image = mono_to_color(melspec)\n    return image","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-05-02T03:07:38.215068Z","iopub.execute_input":"2024-05-02T03:07:38.215569Z","iopub.status.idle":"2024-05-02T03:07:38.225189Z","shell.execute_reply.started":"2024-05-02T03:07:38.215524Z","shell.execute_reply":"2024-05-02T03:07:38.223923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_files = get_files(Config.data_dir / \"train_audio\", extensions=\".ogg\")\nprint(f\"Found {len(audio_files)} audio files\")","metadata":{"execution":{"iopub.status.busy":"2024-05-02T03:07:38.226685Z","iopub.execute_input":"2024-05-02T03:07:38.227017Z","iopub.status.idle":"2024-05-02T03:07:40.085071Z","shell.execute_reply.started":"2024-05-02T03:07:38.226991Z","shell.execute_reply":"2024-05-02T03:07:40.083907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**So we we 24459 audios of different length in the training data**","metadata":{}},{"cell_type":"markdown","source":"# Lets hear some audio and see their spectrogram to get some glimps","metadata":{}},{"cell_type":"code","source":"# take a random sample\naudio_path = random.choice(audio_files)\ninfo = get_audio_info(audio_path)\nprint(info)\n\n# Convert to spectrogram\naudio, sr = sf.read(audio_path)\nimg = audio_to_image(audio)\n\n# show spectrogra\nplt.imshow(img)\nplt.show()\n\n# play audio\ny, sr = lb.load(audio_path)\nAudio(y, rate=sr)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T03:07:40.086765Z","iopub.execute_input":"2024-05-02T03:07:40.087246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# take a random sample\naudio_path = random.choice(audio_files)\ninfo = get_audio_info(audio_path)\nprint(info)\n\n# Convert to spectrogram\naudio, sr = sf.read(audio_path)\nimg = audio_to_image(audio)\n\n# show spectrogra\nplt.imshow(img)\nplt.show()\n\n# play audio\ny, sr = lb.load(audio_path)\nAudio(y, rate=sr)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# take a random sample\naudio_path = random.choice(audio_files)\ninfo = get_audio_info(audio_path)\nprint(info)\n\n# Convert to spectrogram\naudio, sr = sf.read(audio_path)\nimg = audio_to_image(audio)\n\n# show spectrogra\nplt.imshow(img)\nplt.show()\n\n# play audio\ny, sr = lb.load(audio_path)\nAudio(y, rate=sr)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# take a random sample\naudio_path = random.choice(audio_files)\ninfo = get_audio_info(audio_path)\nprint(info)\n\n# Convert to spectrogram\naudio, sr = sf.read(audio_path)\nimg = audio_to_image(audio)\n\n# show spectrogra\nplt.imshow(img)\nplt.show()\n\n# play audio\ny, sr = lb.load(audio_path)\nAudio(y, rate=sr)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T03:07:56.984985Z","iopub.execute_input":"2024-05-02T03:07:56.985303Z","iopub.status.idle":"2024-05-02T03:07:57.544895Z","shell.execute_reply.started":"2024-05-02T03:07:56.985276Z","shell.execute_reply":"2024-05-02T03:07:57.543461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# take a random sample\naudio_path = random.choice(audio_files)\ninfo = get_audio_info(audio_path)\nprint(info)\n\n# Convert to spectrogram\naudio, sr = sf.read(audio_path)\nimg = audio_to_image(audio)\n\n# show spectrogra\nplt.imshow(img)\nplt.show()\n\n# play audio\ny, sr = lb.load(audio_path)\nAudio(y, rate=sr)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T03:07:57.546520Z","iopub.execute_input":"2024-05-02T03:07:57.546936Z","iopub.status.idle":"2024-05-02T03:07:57.985315Z","shell.execute_reply.started":"2024-05-02T03:07:57.546902Z","shell.execute_reply":"2024-05-02T03:07:57.984132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(Config.data_dir/'train_metadata.csv')\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-02T03:07:57.990827Z","iopub.execute_input":"2024-05-02T03:07:57.991362Z","iopub.status.idle":"2024-05-02T03:07:58.167418Z","shell.execute_reply.started":"2024-05-02T03:07:57.991304Z","shell.execute_reply":"2024-05-02T03:07:58.166262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**You will find that the bird sound is brighter on the spectrogram** \n\nSince `sample rate` is 32000, so a 56.976s audio when loaded using python it will become a array of length `56.976 * 32000 = 1823232` which is number of frame. \n\nIf you want to dig deepler into how sound is represented digitally - [check this blog](https://towardsdatascience.com/audio-deep-learning-made-simple-part-1-state-of-the-art-techniques-da1d3dff2504)","metadata":{}},{"cell_type":"markdown","source":"# Lets study the metadata to get more insights","metadata":{}},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T03:07:58.168888Z","iopub.execute_input":"2024-05-02T03:07:58.169351Z","iopub.status.idle":"2024-05-02T03:07:58.196552Z","shell.execute_reply.started":"2024-05-02T03:07:58.169273Z","shell.execute_reply":"2024-05-02T03:07:58.195415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.primary_label.nunique()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T03:07:58.197939Z","iopub.execute_input":"2024-05-02T03:07:58.198259Z","iopub.status.idle":"2024-05-02T03:07:58.212981Z","shell.execute_reply.started":"2024-05-02T03:07:58.198231Z","shell.execute_reply":"2024-05-02T03:07:58.211644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**There are total 182 unique birds sounds in the competitions**","metadata":{}},{"cell_type":"code","source":"value_counts = df['primary_label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T03:07:58.214571Z","iopub.execute_input":"2024-05-02T03:07:58.214951Z","iopub.status.idle":"2024-05-02T03:07:58.227898Z","shell.execute_reply.started":"2024-05-02T03:07:58.214919Z","shell.execute_reply":"2024-05-02T03:07:58.226627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting only the top N values\ntop_n = value_counts.head(50) # Adjust N as needed\ntop_n.plot(kind='bar', figsize=(20, 6))\n\nplt.title('Top N Value Counts of column_name')\nplt.xlabel('Unique Values')\nplt.ylabel('Counts')\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T03:07:58.229450Z","iopub.execute_input":"2024-05-02T03:07:58.229837Z","iopub.status.idle":"2024-05-02T03:07:58.924714Z","shell.execute_reply.started":"2024-05-02T03:07:58.229808Z","shell.execute_reply":"2024-05-02T03:07:58.923377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting only the top N values\ntop_n = value_counts.tail(50) # Adjust N as needed\ntop_n.plot(kind='bar', figsize=(20, 6))\n\nplt.title('Top N Value Counts of column_name')\nplt.xlabel('Unique Values')\nplt.ylabel('Counts')\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T03:07:58.926522Z","iopub.execute_input":"2024-05-02T03:07:58.926889Z","iopub.status.idle":"2024-05-02T03:07:59.612677Z","shell.execute_reply.started":"2024-05-02T03:07:58.926858Z","shell.execute_reply":"2024-05-02T03:07:59.611482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"``For few birds 500 samples are present while for some there are only 5``","metadata":{}},{"cell_type":"markdown","source":"# Destribution of bird on the global map","metadata":{}},{"cell_type":"code","source":"fig = px.scatter_mapbox(df, lat='latitude', lon='longitude', color='primary_label', \n                        hover_name='primary_label', hover_data=['latitude', 'longitude'], \n                        title='Geographical Distribution of Bird Species',\n                        zoom=1, height=600)\nfig.update_layout(mapbox_style=\"open-street-map\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T03:07:59.613980Z","iopub.execute_input":"2024-05-02T03:07:59.614305Z","iopub.status.idle":"2024-05-02T03:08:02.464364Z","shell.execute_reply.started":"2024-05-02T03:07:59.614276Z","shell.execute_reply":"2024-05-02T03:08:02.463218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data contain sound of all over the world, but there are two huge cluster on Asia and Europe.**","metadata":{}},{"cell_type":"markdown","source":"# Lets study about few birds","metadata":{}},{"cell_type":"code","source":"name = \"yebbab1\"\ntemp = df.loc[df['primary_label'] == name]\nprint(f\"total number of bird in the dataset: {len(temp)}\")\n# Download some images of the bird\n\nfig = px.scatter_mapbox(temp, lat='latitude', lon='longitude', color='primary_label', \n                        hover_name='primary_label', hover_data=['latitude', 'longitude'], \n                        title='Geographical Distribution of Bird Species',\n                        zoom=1, height=600)\nfig.update_layout(mapbox_style=\"open-street-map\")\nfig.show()\n\n# Assuming 'temp' is a DataFrame with bird data\nidx = random.randint(0, len(temp)-1)\nentry = temp.iloc[idx]\n\nfilename = entry['filename']\nscientific_name = entry['scientific_name']\ncommon_name = entry['common_name']\nurls = search_images_ddg(common_name, max_images=1)\n\n# Display bird information\ndisplay(Markdown(f\"### Bird Information\"))\ndisplay(Markdown(f\"**Scientific Name:** {scientific_name}\"))\ndisplay(Markdown(f\"**Common Name:** {common_name}\"))\ndisplay(Image(url=urls[0], width=300, height=300))\n\n# Audio information\naudio_path = os.path.join(Config.data_dir,'train_audio', filename)\ninfo = get_audio_info(audio_path)\ndisplay(Markdown(f\"### Audio Information\"))\nprint(f\"Audio Info: {info} \\n\")\n\n# Audio Spectrogram\ndisplay(Markdown(f\"### Audio Spectrogram\"))\naudio, sr = sf.read(audio_path)\nimg = audio_to_image(audio)\nplt.imshow(img)\nplt.axis('off')  # Optional: Hide axis\nplt.show()\n\n# Play audio\ny, sr = lb.load(audio_path)\ndisplay(Audio(y, rate=sr))","metadata":{"execution":{"iopub.status.busy":"2024-05-02T03:08:02.465938Z","iopub.execute_input":"2024-05-02T03:08:02.466322Z","iopub.status.idle":"2024-05-02T03:08:05.036071Z","shell.execute_reply.started":"2024-05-02T03:08:02.466288Z","shell.execute_reply":"2024-05-02T03:08:05.035021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Moipig1**","metadata":{}},{"cell_type":"code","source":"name = \"moipig1\"\ntemp = df.loc[df['primary_label'] == name]\nprint(f\"total number of bird in the dataset: {len(temp)}\")\n# Download some images of the bird\n\nfig = px.scatter_mapbox(temp, lat='latitude', lon='longitude', color='primary_label', \n                        hover_name='primary_label', hover_data=['latitude', 'longitude'], \n                        title='Geographical Distribution of Bird Species',\n                        zoom=1, height=600)\nfig.update_layout(mapbox_style=\"open-street-map\")\nfig.show()\n\n# Assuming 'temp' is a DataFrame with bird data\nidx = random.randint(0, len(temp)-1)\nentry = temp.iloc[idx]\n\nfilename = entry['filename']\nscientific_name = entry['scientific_name']\ncommon_name = entry['common_name']\nurls = search_images_ddg(common_name, max_images=1)\n\n# Display bird information\ndisplay(Markdown(f\"### Bird Information\"))\ndisplay(Markdown(f\"**Scientific Name:** {scientific_name}\"))\ndisplay(Markdown(f\"**Common Name:** {common_name}\"))\ndisplay(Image(url=urls[0], width=300, height=300))\n\n# Audio information\naudio_path = os.path.join(Config.data_dir,'train_audio', filename)\ninfo = get_audio_info(audio_path)\ndisplay(Markdown(f\"### Audio Information\"))\nprint(f\"Audio Info: {info} \\n\")\n\n# Audio Spectrogram\ndisplay(Markdown(f\"### Audio Spectrogram\"))\naudio, sr = sf.read(audio_path)\nimg = audio_to_image(audio)\nplt.imshow(img)\nplt.axis('off')  # Optional: Hide axis\nplt.show()\n\n# Play audio\ny, sr = lb.load(audio_path)\ndisplay(Audio(y, rate=sr))","metadata":{"execution":{"iopub.status.busy":"2024-05-02T03:08:05.037411Z","iopub.execute_input":"2024-05-02T03:08:05.037739Z","iopub.status.idle":"2024-05-02T03:08:07.636767Z","shell.execute_reply.started":"2024-05-02T03:08:05.037712Z","shell.execute_reply":"2024-05-02T03:08:07.635680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"name = \"integr\"\ntemp = df.loc[df['primary_label'] == name]\nprint(f\"total number of bird in the dataset: {len(temp)}\")\n# Download some images of the bird\n\nfig = px.scatter_mapbox(temp, lat='latitude', lon='longitude', color='primary_label', \n                        hover_name='primary_label', hover_data=['latitude', 'longitude'], \n                        title='Geographical Distribution of Bird Species',\n                        zoom=1, height=600)\nfig.update_layout(mapbox_style=\"open-street-map\")\nfig.show()\n\n# Assuming 'temp' is a DataFrame with bird data\nidx = random.randint(0, len(temp)-1)\nentry = temp.iloc[idx]\n\nfilename = entry['filename']\nscientific_name = entry['scientific_name']\ncommon_name = entry['common_name']\nurls = search_images_ddg(common_name, max_images=1)\n\n# Display bird information\ndisplay(Markdown(f\"### Bird Information\"))\ndisplay(Markdown(f\"**Scientific Name:** {scientific_name}\"))\ndisplay(Markdown(f\"**Common Name:** {common_name}\"))\ndisplay(Image(url=urls[0], width=300, height=300))\n\n# Audio information\naudio_path = os.path.join(Config.data_dir,'train_audio', filename)\ninfo = get_audio_info(audio_path)\ndisplay(Markdown(f\"### Audio Information\"))\nprint(f\"Audio Info: {info} \\n\")\n\n# Audio Spectrogram\ndisplay(Markdown(f\"### Audio Spectrogram\"))\naudio, sr = sf.read(audio_path)\nimg = audio_to_image(audio)\nplt.imshow(img)\nplt.axis('off')  # Optional: Hide axis\nplt.show()\n\n# Play audio\ny, sr = lb.load(audio_path)\ndisplay(Audio(y, rate=sr))","metadata":{"execution":{"iopub.status.busy":"2024-05-02T03:08:07.638944Z","iopub.execute_input":"2024-05-02T03:08:07.639877Z","iopub.status.idle":"2024-05-02T03:08:10.235213Z","shell.execute_reply.started":"2024-05-02T03:08:07.639834Z","shell.execute_reply":"2024-05-02T03:08:10.234039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}