{"cells":[{"metadata":{"trusted":true,"_kg_hide-output":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport os\n\n# Data Visualization\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom matplotlib.offsetbox import AnnotationBbox, OffsetImage\n\n\nimport seaborn as sns\nsns.set_style(\"darkgrid\")\n\n# Map 1 library\nimport plotly.express as px\n\n# Map 2 libraries\nimport descartes\nimport geopandas as gpd\nfrom shapely.geometry import Point, Polygon\n\nimport cv2\nfrom wordcloud import WordCloud, STOPWORDS\n\n#Text Color\nfrom termcolor import colored\n\n# Librosa Libraries\nimport librosa\nimport librosa.display\nimport IPython.display as ipd\n\n#Data Preprocessing\nimport sklearn\n\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\nfrom sklearn.model_selection import train_test_split, GridSearchCV, RandomizedSearchCV\n\n#NLP\nfrom sklearn.feature_extraction.text import CountVectorizer\n\n#WordCloud\nfrom wordcloud import WordCloud, STOPWORDS\n\n#Text Processing\nimport re\nimport nltk\nnltk.download('popular')\n\n#Language Detection\n!pip install langdetect\nimport langdetect\n\n#Sentiment\nfrom textblob import TextBlob\n\n#ner\nimport spacy\n\n#Vectorizer\nfrom sklearn import feature_extraction, manifold\n\n#Word Embedding\nimport gensim.downloader as gensim_api\n\n#Topic Modeling\nimport gensim\n\n# HTML\nfrom IPython.core.display import HTML\nimport plotly.graph_objects as go\nimport warnings\nwarnings.filterwarnings('ignore')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/birdclef-2021/train_metadata.csv\")\ntrain_labels = pd.read_csv(\"../input/birdclef-2021/train_soundscape_labels.csv\")\ntest = pd.read_csv('../input/birdclef-2021/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_labels.head(100)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_labels.tail(100)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels = []\nfor row in train_labels.index:\n    labels.extend(train_labels.loc[row, 'birds'].split(' '))\nlabels = list(set(labels))\n\nprint('Number of unique bird labels:', len(labels))\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# **********Not all classes present in training data******"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_labels.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# # row_id consists of AudioID_SiteID_TimeInSeconds"},{"metadata":{"trusted":true},"cell_type":"code","source":"test.info()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Column-wise Unique Values"},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Data: train\")\nprint(\"-----------\")\nfor col in train.columns:\n    print(col + \":\" + str(len(train[col].unique())))\n\nprint(\"\\nData: train_labels\")\nprint(\"-----------\")\nfor col in train_labels.columns:\n    print(col + \":\" + str(len(train_labels[col].unique())))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"species = train['primary_label'].value_counts()\nfig = go.Figure(data=[go.Bar(y=species.values, x=species.index)],\n                layout=go.Layout(margin=go.layout.Margin(l=0, r=1, b=5, t=10)))\n\nfig.update_layout(title='Number of traning samples per species')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_distribution(x, title):\n    \n    sns.displot(x)\n\n    plt.title(title, fontsize = 15)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"label_list = [(train.primary_label.value_counts().values, \"Primary Label: Count Distribution\")]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(train.primary_label.value_counts())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for count, title in label_list:\n    plot_distribution(count, title)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Value Counts and Distribution Plot"},{"metadata":{"trusted":true},"cell_type":"code","source":"name_list = [(train.scientific_name.value_counts().values, \"Scientific Names: Count Distribution\"),\n            (train.common_name.value_counts().values, \"Common Name: Count Distribution\"),\n            ]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for count, title in name_list:\n    plot_distribution(count, title)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f\"The Ratings of songs are in range {train.rating.min()} upto {train.rating.max()} seconds\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nax = sns.countplot(train['rating'], palette=\"hls\", order = train['rating'].value_counts().index)\n\nplt.title(\"Rating of Song\", fontsize = 16)\nplt.xticks(fontsize = 13)\nplt.yticks(fontsize = 13)\nplt.ylabel(\"Frequency\", fontsize = 14)\nplt.xlabel(\"Rating\", fontsize = 14)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# **Most of the songs time is 4 seconds**"},{"metadata":{"trusted":true},"cell_type":"code","source":"train.type[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"adjusted_type = train['type'].apply(lambda x: x.replace('[', ''))\nadjusted_type = adjusted_type.apply(lambda x: x.replace(']', ''))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"adjusted_type = adjusted_type.apply(lambda x: x.split(',')).reset_index().explode(\"type\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Strip of white spaces and convert to lower chars\nadjusted_type = adjusted_type['type'].apply(lambda x: x.strip().lower()).reset_index()\nadjusted_type['type'] = adjusted_type['type'].replace('calls', 'call')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Create Top 8 list with song types\ntop_8 = list(adjusted_type['type'].value_counts().head(8).reset_index()['index'])\ndata = adjusted_type[adjusted_type['type'].isin(top_8)]\n\nplt.figure(figsize=(16, 6))\nax = sns.countplot(data['type'], palette=\"hls\", order = data['type'].value_counts().index)\n\nplt.title(\"Top 8 Song Types\", fontsize=16)\nplt.ylabel(\"Frequency\", fontsize = 14)\nplt.yticks(fontsize = 13) \nplt.xticks(rotation = 45, fontsize = 13) \nplt.xlabel(\"Type\", fontsize = 14);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(len(os.listdir('../input/birdclef-2021/test_soundscapes')))\nprint(\"Train ShortAudio folder has \",len(os.listdir('../input/birdclef-2021/train_short_audio')),'folders each one is a type of birds and contain different number of audio files with different durations')\nprint(\"Train Soundscape folder has \",len(os.listdir('../input/birdclef-2021/train_soundscapes')),'files each one 10 minutes duration')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with open('../input/birdclef-2021/test_soundscapes/SNE_recording_location.txt','r') as f:\n        line = f.readline()\n        while line:\n            line = f.readline()\n            print(line)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# **Location,Latitude,Longitude**"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}