{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"![](https://storage.googleapis.com/kaggle-competitions/kaggle/44224/logos/header.png)","metadata":{"_kg_hide-input":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os  \nimport ast\nimport librosa\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport plotly.express as px\nimport plotly\nimport plotly.graph_objs as go\nfrom collections import Counter\nimport matplotlib.pyplot as plt\nfrom IPython.display import Audio","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:10.208307Z","iopub.execute_input":"2023-03-14T09:52:10.208704Z","iopub.status.idle":"2023-03-14T09:52:13.342566Z","shell.execute_reply.started":"2023-03-14T09:52:10.208671Z","shell.execute_reply":"2023-03-14T09:52:13.341334Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Glimse of TRAIN DATA","metadata":{}},{"cell_type":"code","source":"trainmeta_df = pd.read_csv(\"/kaggle/input/birdclef-2023/train_metadata.csv\")\ntrainmeta_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:13.344388Z","iopub.execute_input":"2023-03-14T09:52:13.344692Z","iopub.status.idle":"2023-03-14T09:52:13.496819Z","shell.execute_reply.started":"2023-03-14T09:52:13.344660Z","shell.execute_reply":"2023-03-14T09:52:13.495874Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check for NULL Values","metadata":{}},{"cell_type":"code","source":"trainmeta_df.isna().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:13.498225Z","iopub.execute_input":"2023-03-14T09:52:13.498824Z","iopub.status.idle":"2023-03-14T09:52:13.513324Z","shell.execute_reply.started":"2023-03-14T09:52:13.498791Z","shell.execute_reply":"2023-03-14T09:52:13.512531Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- there were totally 454 NULL values in the dataset","metadata":{}},{"cell_type":"code","source":"trainmeta_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:13.515609Z","iopub.execute_input":"2023-03-14T09:52:13.516171Z","iopub.status.idle":"2023-03-14T09:52:13.547466Z","shell.execute_reply.started":"2023-03-14T09:52:13.516139Z","shell.execute_reply":"2023-03-14T09:52:13.546096Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*  There are total of 12 columns : 3 numerical , 9 categorical\n*  There are total 454 missing values in the dataframe","metadata":{}},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"markdown","source":"# histogram of Primary Label","metadata":{}},{"cell_type":"code","source":"\nfig = px.histogram(trainmeta_df, x=\"primary_label\",nbins=len(trainmeta_df[\"primary_label\"].unique()) )\nfig.update_layout(title_text=\"Distribution of Primary Labels\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:13.548930Z","iopub.execute_input":"2023-03-14T09:52:13.549466Z","iopub.status.idle":"2023-03-14T09:52:15.335593Z","shell.execute_reply.started":"2023-03-14T09:52:13.549421Z","shell.execute_reply":"2023-03-14T09:52:15.334649Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Histogram of primary label along with the Author","metadata":{}},{"cell_type":"code","source":"\nfig = px.histogram(trainmeta_df, x=\"primary_label\",color = \"author\",nbins=len(trainmeta_df[\"primary_label\"].unique()) )\nfig.update_layout(title_text=\"Distribution of Primary Labels based on author\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:15.337008Z","iopub.execute_input":"2023-03-14T09:52:15.337524Z","iopub.status.idle":"2023-03-14T09:52:19.787758Z","shell.execute_reply.started":"2023-03-14T09:52:15.337490Z","shell.execute_reply":"2023-03-14T09:52:19.786585Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Hisogram of secondary label ","metadata":{}},{"cell_type":"code","source":"secondary_df = trainmeta_df['secondary_labels'].str.replace('[','').str.replace(']','').str.replace('\\'','').str.split(',', expand=True).stack().reset_index(level=1, drop=True).rename('secondary_label')\nsecondary_df = pd.merge(trainmeta_df.drop(columns=['secondary_labels']), secondary_df, left_index=True, right_index=True)\n\nfig = px.histogram(secondary_df, x=\"secondary_label\", title=\"Distribution of Secondary Labels\")\nfig.update_layout(xaxis_title=\"Secondary Label\", yaxis_title=\"Count\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:19.789357Z","iopub.execute_input":"2023-03-14T09:52:19.789693Z","iopub.status.idle":"2023-03-14T09:52:20.012199Z","shell.execute_reply.started":"2023-03-14T09:52:19.789662Z","shell.execute_reply":"2023-03-14T09:52:20.010841Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bar chart of Type","metadata":{}},{"cell_type":"code","source":"# Flatten the list of labels in the \"type\" column\nlabels = [label.strip(\"[]'\") for sublist in trainmeta_df['type'].apply(ast.literal_eval) for label in sublist]\n\n# Count the occurrence of each label\nlabel_counts = Counter(labels)\n\n# Create a bar plot of the label counts\nfig = px.bar(x=list(label_counts.keys()), y=list(label_counts.values()))\nfig.update_layout(title_text=\"Distribution of Types\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:20.013774Z","iopub.execute_input":"2023-03-14T09:52:20.014184Z","iopub.status.idle":"2023-03-14T09:52:20.249084Z","shell.execute_reply.started":"2023-03-14T09:52:20.014137Z","shell.execute_reply":"2023-03-14T09:52:20.247815Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# BOX plot for latitude and longitude ","metadata":{}},{"cell_type":"code","source":"fig1 = px.box(trainmeta_df, y=\"latitude\")\nfig1.update_layout(title_text=\"Distribution of Latitude\",\n                   yaxis=dict(title=\"Latitude\"))\n\n# create a box plot for longitude\nfig2 = px.box(trainmeta_df, y=\"longitude\")\nfig2.update_layout(title_text=\"Distribution of Longitude\",\n                   yaxis=dict(title=\"Longitude\"))\n\n# create a box plot for rating\nfig3 = px.box(trainmeta_df, y=\"rating\")\nfig3.update_layout(title_text=\"Distribution of Ratings\",\n                   yaxis=dict(title=\"Rating\"))\n\n# show the figures\nfig1.show()\nfig2.show()\nfig3.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:20.250538Z","iopub.execute_input":"2023-03-14T09:52:20.250889Z","iopub.status.idle":"2023-03-14T09:52:20.448420Z","shell.execute_reply.started":"2023-03-14T09:52:20.250855Z","shell.execute_reply":"2023-03-14T09:52:20.447563Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Histogram of Scientific Names","metadata":{}},{"cell_type":"code","source":"fig = px.histogram(trainmeta_df, x=\"scientific_name\", nbins=len(trainmeta_df[\"scientific_name\"].unique()))\nfig.update_layout(title_text=\"Distribution of Scientific Names\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:20.451965Z","iopub.execute_input":"2023-03-14T09:52:20.452585Z","iopub.status.idle":"2023-03-14T09:52:20.578370Z","shell.execute_reply.started":"2023-03-14T09:52:20.452547Z","shell.execute_reply":"2023-03-14T09:52:20.577523Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Histogram of Common Name","metadata":{}},{"cell_type":"code","source":"fig = px.histogram(trainmeta_df, x=\"common_name\", nbins=len(trainmeta_df[\"common_name\"].unique()))\nfig.update_layout(title_text=\"Distribution of Common Names\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:20.579550Z","iopub.execute_input":"2023-03-14T09:52:20.580585Z","iopub.status.idle":"2023-03-14T09:52:20.703443Z","shell.execute_reply.started":"2023-03-14T09:52:20.580548Z","shell.execute_reply":"2023-03-14T09:52:20.702220Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Histogram of AUTHOR","metadata":{}},{"cell_type":"code","source":"fig = px.histogram(trainmeta_df, x=\"author\", nbins=len(trainmeta_df[\"author\"].unique()))\nfig.update_layout(title_text=\"Distribution of Authors\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:20.704888Z","iopub.execute_input":"2023-03-14T09:52:20.705238Z","iopub.status.idle":"2023-03-14T09:52:20.831599Z","shell.execute_reply.started":"2023-03-14T09:52:20.705205Z","shell.execute_reply":"2023-03-14T09:52:20.830446Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Histogram of Rating","metadata":{}},{"cell_type":"code","source":"fig = px.histogram(trainmeta_df, x=\"rating\", nbins=len(trainmeta_df[\"rating\"].unique()) , color_discrete_sequence=['red'])\nfig.update_layout(title_text=\"Distribution of Ratings\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:20.833068Z","iopub.execute_input":"2023-03-14T09:52:20.833424Z","iopub.status.idle":"2023-03-14T09:52:20.896876Z","shell.execute_reply.started":"2023-03-14T09:52:20.833391Z","shell.execute_reply":"2023-03-14T09:52:20.895643Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Corelation Matrix","metadata":{}},{"cell_type":"code","source":"# drop columns from correlation matrix\ncorr = trainmeta_df.corr()\n\n# create correlation heatmap\nfig = px.imshow(corr,\n                labels=dict(x=\"Columns\", y=\"Columns\", color=\"Correlation\"),\n                x=corr.columns,\n                y=corr.columns,\n                color_continuous_scale='RdBU',\n                zmin=-1,\n                zmax=1,\n                title=\"Correlation Heatmap\")\n                \n# add text annotations\nannotations = []\nfor i, row in enumerate(corr.values):\n    for j, value in enumerate(row):\n        text = '{:.2f}'.format(value)\n        annotations.append(dict(x=corr.columns[j], y=corr.columns[i], text=text, showarrow=False))\n\nfig.update_layout(width=800, height=800)\nfig.update_traces(showscale=True, colorbar_thickness=25, colorbar_len=0.75)\nfig.update_layout(margin=dict(l=50, r=50, b=100, t=100, pad=4))\nfig.update_layout(annotations=annotations)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:20.898669Z","iopub.execute_input":"2023-03-14T09:52:20.899407Z","iopub.status.idle":"2023-03-14T09:52:21.015440Z","shell.execute_reply.started":"2023-03-14T09:52:20.899369Z","shell.execute_reply":"2023-03-14T09:52:21.014247Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Scatter Plot on Longitude and Latitude ","metadata":{}},{"cell_type":"code","source":"fig = go.Figure(data=go.Scattergeo(lon=trainmeta_df['longitude'], lat = trainmeta_df['latitude'], mode='markers', marker=dict(size=4,\n                                            opacity=0.6,symbol='square',line=dict(width=1,\n                                                                                                                        color='white'),\n                                                                                                               colorscale='Blues',\n                                                                                                               color='blue')),)\nfig.update_layout(mapbox_style=\"open-street-map\")\nplotly.offline.iplot(fig)","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:21.016703Z","iopub.execute_input":"2023-03-14T09:52:21.017017Z","iopub.status.idle":"2023-03-14T09:52:21.081507Z","shell.execute_reply.started":"2023-03-14T09:52:21.016988Z","shell.execute_reply":"2023-03-14T09:52:21.080346Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting map using plotly\nfig = px.scatter_geo(trainmeta_df, lat = 'latitude', lon = 'longitude')\n\n# setting title for the map\nfig.update_layout(title = 'Countries', title_x = 0.5)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:21.083130Z","iopub.execute_input":"2023-03-14T09:52:21.083858Z","iopub.status.idle":"2023-03-14T09:52:21.191594Z","shell.execute_reply.started":"2023-03-14T09:52:21.083814Z","shell.execute_reply":"2023-03-14T09:52:21.190455Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Lets Dive into AUDIO FILES","metadata":{}},{"cell_type":"code","source":"Audio(\"/kaggle/input/birdclef-2023/train_audio/abethr1/XC128013.ogg\")","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:21.192755Z","iopub.execute_input":"2023-03-14T09:52:21.193085Z","iopub.status.idle":"2023-03-14T09:52:21.238496Z","shell.execute_reply.started":"2023-03-14T09:52:21.193052Z","shell.execute_reply":"2023-03-14T09:52:21.237373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def audio_eda(audio_path):\n    folder = audio_path.split(\"/\")[-2]\n    filename = audio_path.split(\"/\")[-1]\n   \n    # Load an audio file\n    samples, sample_rate = librosa.load(audio_path)\n\n    # Visualize the waveform\n    plt.figure(figsize=(14, 5))\n    librosa.display.waveshow(samples, sr=sample_rate)\n    plt.title(f'Waveform from {folder} on {filename}')\n\n    # Compute the spectrogram\n    spectrogram = librosa.stft(samples)\n    spectrogram_db = librosa.amplitude_to_db(abs(spectrogram))\n\n    # Visualize the spectrogram\n    plt.figure(figsize=(14, 5))\n    librosa.display.specshow(spectrogram_db, sr=sample_rate, x_axis='time', y_axis='log')\n    plt.colorbar(format='%+2.0f dB')\n    plt.title(f'Spectrogram (dB){folder} on {filename}')\n\n    # Compute the mel spectrogram\n\n\n    # Visualize the mel spectrogram\n    S = librosa.feature.melspectrogram(y=samples, sr=sample_rate)\n\n    # Visualize mel spectrogram\n    plt.figure(figsize=(10, 4))\n    librosa.display.specshow(librosa.power_to_db(S, ref=np.max), y_axis='mel', fmax=8000, x_axis='time')\n    plt.colorbar(format='%+2.0f dB')\n    plt.title(f'Mel spectrogram {folder} on {filename}')\n    plt.tight_layout()\n\n\n    # Compute the chromagram\n    chromagram = librosa.feature.chroma_stft( y = samples , sr = sample_rate)\n\n    # Visualize the chromagram\n    plt.figure(figsize=(14, 5))\n    librosa.display.specshow(chromagram, sr=sample_rate, x_axis='time', y_axis='chroma')\n    plt.colorbar()\n    plt.title(f'Chromagram {folder} on {filename}')\n\n    # Compute the MFCCs\n    mfccs = librosa.feature.mfcc(y=samples, sr=sample_rate, n_mfcc=13)\n\n    # Visualize the MFCCs\n    plt.figure(figsize=(14, 5))\n    librosa.display.specshow(mfccs, sr=sample_rate, x_axis='time')\n    plt.colorbar()\n    plt.title(f'MFCCs {folder} on {filename}')\n\n    # Show the plots\n    display(Audio(samples, rate=sample_rate))\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:21.239897Z","iopub.execute_input":"2023-03-14T09:52:21.240226Z","iopub.status.idle":"2023-03-14T09:52:21.253937Z","shell.execute_reply.started":"2023-03-14T09:52:21.240195Z","shell.execute_reply":"2023-03-14T09:52:21.252632Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_eda(\"/kaggle/input/birdclef-2023/train_audio/abethr1/XC128013.ogg\")","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:52:21.255807Z","iopub.execute_input":"2023-03-14T09:52:21.256981Z","iopub.status.idle":"2023-03-14T09:52:38.776653Z","shell.execute_reply.started":"2023-03-14T09:52:21.256929Z","shell.execute_reply":"2023-03-14T09:52:38.775336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Let's see the Waveform,spectogram,mel-spectogram,chromogram and MFCC","metadata":{}},{"cell_type":"code","source":"src_path = \"/kaggle/input/birdclef-2023/train_audio/\"\nfolders = os.listdir(src_path)[:10]\nfor fold in folders:\n    fold_path = src_path+\"/\"+fold\n    files = os.listdir(fold_path)[:3]\n    for file in files:\n        audio_path = fold_path+\"/\"+file\n        audio_eda(audio_path)\n        print(\"Moving to NEXT AUDIO\")\n","metadata":{"execution":{"iopub.status.busy":"2023-03-14T09:53:30.561302Z","iopub.execute_input":"2023-03-14T09:53:30.562640Z","iopub.status.idle":"2023-03-14T09:55:07.412703Z","shell.execute_reply.started":"2023-03-14T09:53:30.562593Z","shell.execute_reply":"2023-03-14T09:55:07.411394Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}