{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-output":true,"collapsed":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import os\nimport numpy as np \nimport pandas as pd\nimport datetime as dt\nfrom sklearn import preprocessing as prep\nimport librosa as lb\nimport librosa.display as lbd\nimport librosa.feature as lbf\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.graph_objects as go\nfrom collections import Counter\nfrom plotly.subplots import make_subplots\nimport plotly.express as px\nfrom matplotlib import rcParams\nimport plotly.offline\nsns.set(style='darkgrid')\nplt.rcParams['figure.figsize'] = (16,8)\nimport IPython.display as ipd\nimport ipywidgets as ipw\nimport warnings\nwarnings.filterwarnings('ignore')\n\nlink = 'https://ebird.org/species/'\nPATH_AUDIO = '../input/birdsong-recognition/train_audio/'\n## configuring setup, constants and parameters\nPATH_TRAIN = \"../input/birdsong-recognition/train.csv\"\nPATH_TEST = \"../input/birdsong-recognition/test.csv\"\n\n# PATH_TRAIN_EXTENDED = \"../input/xeno-canto-bird-recordings-extended-a-m/train_extended.csv\"\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv(PATH_TRAIN)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(set(train.ebird_code))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### There are totally 264 types of bird species availabel in the dataset","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# zero_crossings = lb.zero_crossings(x[n0:n1], pad=False)\n# print(sum(zero_crossings))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_bird_map = train[[\"ebird_code\", \"species\"]].drop_duplicates()\n\nfor ebird_code in os.listdir(PATH_AUDIO)[:20]:\n    species = df_bird_map[df_bird_map.ebird_code == ebird_code].species.values[0]\n    audio_file = os.listdir(f\"{PATH_AUDIO}/{ebird_code}\")[0]\n    audio_path = f\"{PATH_AUDIO}/{ebird_code}/{audio_file}\"\n    ipd.display(ipd.HTML(f\"<h2>{ebird_code} ({species})</h2>\"))\n    ipd.display(ipd.Audio(audio_path))\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_for_one_species(Audio_path):\n    values = Audio_path.split(\"/\")\n    ipd.display(ipd.HTML(f\"<h2>{values[5]}</h2>\"))\n    ipd.display(ipd.Audio(Audio_path))\n    data , samplingrate = lb.load(Audio_path)\n    plt.figure(figsize=(12, 4))\n    plt.title(\"Visuvalizing the Audio\")\n    lb.display.waveplot(data, sr=samplingrate)\n    X = lb.stft(data)\n    Xdb = lb.amplitude_to_db(abs(X))\n    plt.figure(figsize=(14, 5))\n    lb.display.specshow(Xdb, sr=samplingrate, x_axis='time', y_axis='hz')\n    plt.colorbar()\n    plt.title(\"Spectrogram of the wave\")\n#     # Zooming in\n#     n0 = 9000\n#     n1 = 9100\n#     plt.figure(figsize=(14, 5))\n#     plt.plot(X[n0:n1])\n#     plt.grid()\n#     plt.title(\"Zero Crossing rate\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"path = '/kaggle/input/birdsong-recognition/train_audio/nutwoo/XC462016.mp3'\nplot_for_one_species(path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.scatter(data_frame=train, x='longitude', y='latitude', color='ebird_code')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.choropleth(data_frame=train,locations=\"country\",locationmode=\"country names\",hover_name=\"species\",title=\"Birds Location\")\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# displaying only the top 30 countries\ncountry = train.country.value_counts()\ncountry_df = pd.DataFrame({'country':country.index, 'frequency':country.values}).head(35)\n\nfig = px.bar(country_df, x=\"frequency\", y=\"country\",color='country', orientation='h',\n             hover_data=[\"country\", \"frequency\"],\n             height=1000,\n             title='Number of audio samples besed on country of recording')\nfig.show()\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# displaying only the top 30 countries\nauthors = train.author.value_counts()\nauthors_df = pd.DataFrame({'authors':authors.index, 'frequency':authors.values}).head(35)\nfig = px.bar(authors_df, x=\"frequency\", y=\"authors\",color='authors', orientation='h',\n             hover_data=[\"authors\", \"frequency\"],\n             height=1000,\n             title='Authors Contribution')\nfig.show()\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"rcParams[\"figure.figsize\"] = 20,8\ntrain['ebird_code'].value_counts().plot(kind='hist')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(20, 8))\ntrain['date'].value_counts().sort_index().plot();\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# def parser(row):\n#    # function to load files and extract features\n#    file_name = os.path.join(os.path.abspath(data_dir), str(row.ebird_code), str(row.filename))\n\n#    # handle exception to check if there isn't a file which is corrupted\n#    try:\n#         for i in tqdm(range(0,10)):\n#             with joblib.parallel_backend('dask'):\n#               # here kaiser_fast is a technique used for faster extraction\n#               X, sample_rate = lb.load(file_name, res_type='kaiser_fast') \n#               # we extract mfcc feature from data\n#               mfccs = np.mean(lb.feature.mfcc(y=X, sr=sample_rate, n_mfcc=40).T,axis=0) \n#    except Exception as e:\n#       print(\"Error encountered while parsing file: \", file)\n#       return None, None\n \n#    feature = mfccs\n#    label = row.ebird_code\n \n#    return [feature, label]\n\n# temp = train.apply(parser, axis=1)\n# temp.columns = ['feature', 'label']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}