{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# BirdCLEF EDA + Audio Visualization","metadata":{}},{"cell_type":"code","source":"!pip install reverse_geocode","metadata":{"execution":{"iopub.status.busy":"2022-04-12T00:38:10.043164Z","iopub.execute_input":"2022-04-12T00:38:10.043641Z","iopub.status.idle":"2022-04-12T00:38:21.532831Z","shell.execute_reply.started":"2022-04-12T00:38:10.043554Z","shell.execute_reply":"2022-04-12T00:38:21.531933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport seaborn as sns\nimport os\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport reverse_geocode\nimport librosa\nimport librosa.display\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-12T00:38:21.535066Z","iopub.execute_input":"2022-04-12T00:38:21.535321Z","iopub.status.idle":"2022-04-12T00:38:25.100347Z","shell.execute_reply.started":"2022-04-12T00:38:21.535288Z","shell.execute_reply":"2022-04-12T00:38:25.099159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"train_md = pd.read_csv('../input/birdclef-2022/train_metadata.csv')","metadata":{"execution":{"iopub.status.busy":"2022-04-11T22:30:26.700768Z","iopub.execute_input":"2022-04-11T22:30:26.701084Z","iopub.status.idle":"2022-04-11T22:30:26.827933Z","shell.execute_reply.started":"2022-04-11T22:30:26.701045Z","shell.execute_reply":"2022-04-11T22:30:26.827267Z"}}},{"cell_type":"code","source":"train_md = pd.read_csv('../input/birdclef-2022/train_metadata.csv')\ntrain_md.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-04-12T00:38:25.102367Z","iopub.execute_input":"2022-04-12T00:38:25.102708Z","iopub.status.idle":"2022-04-12T00:38:25.233567Z","shell.execute_reply.started":"2022-04-12T00:38:25.102661Z","shell.execute_reply":"2022-04-12T00:38:25.232660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Thank Goodness! No missing data. ","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (20,10))\ncolors = sns.color_palette('pastel')\ncount_val = train_md.primary_label.value_counts(normalize = True).values\nlabels = train_md.primary_label.value_counts().keys()\nsns.barplot(x = labels,y = count_val, palette  = 'pastel')\nplt.xticks(rotation = 90)\nplt.title('Primary Label Composition in Training Data')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-12T00:38:25.235070Z","iopub.execute_input":"2022-04-12T00:38:25.235301Z","iopub.status.idle":"2022-04-12T00:38:27.158598Z","shell.execute_reply.started":"2022-04-12T00:38:25.235272Z","shell.execute_reply":"2022-04-12T00:38:27.157566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10,10))\ncolors = sns.color_palette('pastel')\ncount_val = train_md.primary_label.value_counts(normalize = True).values[:11].tolist()\nlabels = train_md.primary_label.value_counts().keys().tolist()[:11]\ncount_val.append(sum(train_md.primary_label.value_counts(normalize = True).values[11:]))\nlabels.append('others')\nplt.title('Top 10 Primary Labels by count in Training Data')\nplt.pie(count_val, labels = labels, colors = colors, autopct = '%0.1f%%')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-12T00:38:27.161193Z","iopub.execute_input":"2022-04-12T00:38:27.161450Z","iopub.status.idle":"2022-04-12T00:38:27.591236Z","shell.execute_reply.started":"2022-04-12T00:38:27.161417Z","shell.execute_reply":"2022-04-12T00:38:27.590208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10,10))\ncolors = sns.color_palette('pastel')\ncount_val = train_md.common_name.value_counts(normalize = True).values[:11].tolist()\nlabels = train_md.common_name.value_counts().keys().tolist()[:11]\ncount_val.append(sum(train_md.common_name.value_counts(normalize = True).values[11:]))\nlabels.append('others')\nplt.title('Top 10 Birds (Common Name) by count in Training Data')\nplt.pie(count_val, labels = labels, colors = colors, autopct = '%0.1f%%')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-12T00:38:27.593122Z","iopub.execute_input":"2022-04-12T00:38:27.593369Z","iopub.status.idle":"2022-04-12T00:38:27.846039Z","shell.execute_reply.started":"2022-04-12T00:38:27.593339Z","shell.execute_reply":"2022-04-12T00:38:27.845059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Seems like Barn Owl is the most common bird","metadata":{}},{"cell_type":"code","source":"sns.catplot(y = 'rating', data = train_md, kind=\"box\", palette = 'pastel')\nplt.ylabel('Rating')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-12T00:38:27.847132Z","iopub.execute_input":"2022-04-12T00:38:27.847346Z","iopub.status.idle":"2022-04-12T00:38:28.067317Z","shell.execute_reply.started":"2022-04-12T00:38:27.847312Z","shell.execute_reply":"2022-04-12T00:38:28.066338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Our median rating is 4. This is good. Atleast 50% of data is of high quality.","metadata":{}},{"cell_type":"code","source":"fig = px.scatter_geo(\n    train_md,\n    lat=\"latitude\",\n    lon=\"longitude\",\n    color=\"common_name\",\n    width=1_000,\n    height=500,\n    title=\"BirdCLEF 2022 Recording Geographical Locations\",\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-12T00:38:28.068645Z","iopub.execute_input":"2022-04-12T00:38:28.069167Z","iopub.status.idle":"2022-04-12T00:38:29.516237Z","shell.execute_reply.started":"2022-04-12T00:38:28.069134Z","shell.execute_reply":"2022-04-12T00:38:29.515207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def city_state_country(row):\n    coord = (row['latitude'], row['longitude']), (0,0)\n    location = reverse_geocode.search(coord)[0]['country']\n    row['country'] = location\n    return row\n\ntrain_md = train_md.apply(city_state_country, axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2022-04-12T00:38:29.517756Z","iopub.execute_input":"2022-04-12T00:38:29.518186Z","iopub.status.idle":"2022-04-12T00:38:41.042233Z","shell.execute_reply.started":"2022-04-12T00:38:29.518140Z","shell.execute_reply":"2022-04-12T00:38:41.041271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (20,10))\nsns.barplot(x = train_md.country.value_counts().keys(),y = train_md.country.value_counts().values, data = train_md, palette  = 'pastel')\nplt.xticks(rotation = 90)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-12T00:38:41.043559Z","iopub.execute_input":"2022-04-12T00:38:41.043865Z","iopub.status.idle":"2022-04-12T00:38:43.374834Z","shell.execute_reply.started":"2022-04-12T00:38:41.043824Z","shell.execute_reply":"2022-04-12T00:38:43.373971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Most of our data comes from United States. It is possible that if there is a location specific feature in our recordings, then it might cause poor generalization when we build and train our models.  ","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (10,10))\ncolors = sns.color_palette('pastel')\ncount_val = train_md.country.value_counts(normalize = True).values[:11].tolist()\nlabels = train_md.country.value_counts().keys().tolist()[:11]\ncount_val.append(sum(train_md.country.value_counts(normalize = True).values[11:]))\nlabels.append('others')\nplt.title('Top 10 Countries by count in Training Data')\nplt.pie(count_val, labels = labels, colors = colors, autopct = '%0.1f%%')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-12T00:38:43.376216Z","iopub.execute_input":"2022-04-12T00:38:43.376881Z","iopub.status.idle":"2022-04-12T00:38:43.616678Z","shell.execute_reply.started":"2022-04-12T00:38:43.376849Z","shell.execute_reply":"2022-04-12T00:38:43.615595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.scatter_geo(\n    train_md,\n    lat=\"latitude\",\n    lon=\"longitude\",\n    color=\"common_name\",\n    width=1_000,\n    height=500,\n    title=\"BirdCLEF 2022 Recording USA\",\n    scope = 'usa',\n    hover_name = 'country'\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-12T00:38:43.617801Z","iopub.execute_input":"2022-04-12T00:38:43.618558Z","iopub.status.idle":"2022-04-12T00:38:44.198667Z","shell.execute_reply.started":"2022-04-12T00:38:43.618522Z","shell.execute_reply":"2022-04-12T00:38:44.198075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Audio Files","metadata":{}},{"cell_type":"code","source":"AUDIO_DIR = '../input/birdclef-2022/train_audio'\nidx = np.random.randint(0, len(train_md), 10)","metadata":{"execution":{"iopub.status.busy":"2022-04-12T00:38:44.199810Z","iopub.execute_input":"2022-04-12T00:38:44.200177Z","iopub.status.idle":"2022-04-12T00:38:44.203744Z","shell.execute_reply.started":"2022-04-12T00:38:44.200146Z","shell.execute_reply":"2022-04-12T00:38:44.203208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows=10, figsize=(10, 10), sharex = True)\nfor i in range(10):\n    audio_file = AUDIO_DIR + '/' + train_md.iloc[idx[i]].filename\n    signal, sr = librosa.load(audio_file)\n    librosa.display.waveshow(signal, sr=sr, alpha = 0.5, color = 'blue', ax=ax[i])","metadata":{"execution":{"iopub.status.busy":"2022-04-12T00:38:44.205855Z","iopub.execute_input":"2022-04-12T00:38:44.206252Z","iopub.status.idle":"2022-04-12T00:38:58.860700Z","shell.execute_reply.started":"2022-04-12T00:38:44.206220Z","shell.execute_reply":"2022-04-12T00:38:58.857441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### We have to be cautious while prepraing the data. Our inputs would have varying lengths. Either we trim the signals to a minimum length or pad it to a maximum length.  ","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows=10, figsize=(10, 20))\nfor i in range(10):\n    audio_file = AUDIO_DIR + '/' + train_md.iloc[idx[i]].filename\n    signal, sr = librosa.load(audio_file)\n    D = librosa.amplitude_to_db(np.abs(librosa.stft(signal)), ref=np.max)\n    img = librosa.display.specshow(D, y_axis='linear', x_axis='time',sr=sr, ax=ax[i], cmap = 'cool')\nfig.colorbar(img, ax=ax, format=\"%+2.f dB\")","metadata":{"execution":{"iopub.status.busy":"2022-04-12T00:38:58.861913Z","iopub.execute_input":"2022-04-12T00:38:58.862723Z","iopub.status.idle":"2022-04-12T00:39:14.191907Z","shell.execute_reply.started":"2022-04-12T00:38:58.862689Z","shell.execute_reply":"2022-04-12T00:39:14.190983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### We can already see some patterns in the spectrograms. If these are distinct enough for different birds, then we can build a good model. ","metadata":{}}]}