{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":19596,"databundleVersionId":1292430,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os \nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go\nimport plotly.express as px\nimport folium \nfrom folium import Marker , GeoJson , Choropleth , Circle\nfrom folium.plugins import HeatMap , MarkerCluster\nimport librosa.display \nfrom IPython.display import Audio","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-29T11:33:42.312646Z","iopub.execute_input":"2023-11-29T11:33:42.313092Z","iopub.status.idle":"2023-11-29T11:33:44.451821Z","shell.execute_reply.started":"2023-11-29T11:33:42.313044Z","shell.execute_reply":"2023-11-29T11:33:44.450196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df =pd.read_csv(\"/kaggle/input/birdsong-recognition/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:33:44.453313Z","iopub.execute_input":"2023-11-29T11:33:44.453934Z","iopub.status.idle":"2023-11-29T11:33:44.903806Z","shell.execute_reply.started":"2023-11-29T11:33:44.453898Z","shell.execute_reply":"2023-11-29T11:33:44.902308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:33:44.90539Z","iopub.execute_input":"2023-11-29T11:33:44.905758Z","iopub.status.idle":"2023-11-29T11:33:44.94178Z","shell.execute_reply.started":"2023-11-29T11:33:44.905703Z","shell.execute_reply":"2023-11-29T11:33:44.940509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets check the class distribution of the data","metadata":{}},{"cell_type":"code","source":"df[\"species\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:33:44.945815Z","iopub.execute_input":"2023-11-29T11:33:44.946295Z","iopub.status.idle":"2023-11-29T11:33:44.962039Z","shell.execute_reply.started":"2023-11-29T11:33:44.946248Z","shell.execute_reply":"2023-11-29T11:33:44.9608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The data is highly imbalanced as we can see some of the speicies audio files are less compared to some other ","metadata":{"execution":{"iopub.status.busy":"2023-11-29T09:39:24.808211Z","iopub.execute_input":"2023-11-29T09:39:24.808648Z","iopub.status.idle":"2023-11-29T09:39:24.816543Z","shell.execute_reply.started":"2023-11-29T09:39:24.808616Z","shell.execute_reply":"2023-11-29T09:39:24.814736Z"}}},{"cell_type":"code","source":"# visualize the data ","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:33:44.964847Z","iopub.execute_input":"2023-11-29T11:33:44.96581Z","iopub.status.idle":"2023-11-29T11:33:44.971676Z","shell.execute_reply.started":"2023-11-29T11:33:44.965758Z","shell.execute_reply":"2023-11-29T11:33:44.970198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"year\"] = df[\"date\"].apply(lambda x : x.split(\"-\")[0])\ndf[\"month\"] = df[\"date\"].apply(lambda x : x.split(\"-\")[1])\ngroup_year = df.groupby([\"year\"]).size().reset_index(name = \"counts\")\ngroup_year = group_year.iloc[3:]\ngroup_month = df.groupby([\"month\"]).size().reset_index(name = \"counts\")\n","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:33:44.973468Z","iopub.execute_input":"2023-11-29T11:33:44.973841Z","iopub.status.idle":"2023-11-29T11:33:45.0255Z","shell.execute_reply.started":"2023-11-29T11:33:44.973809Z","shell.execute_reply":"2023-11-29T11:33:45.024579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = make_subplots(rows=2, cols=1, subplot_titles = ('Number of recordings accdn year', 'Number of recordings accdn month'))\n\nfig.append_trace(go.Bar(\n    x=group_year['year'],\n    y=group_year['counts'],\n    #tickmode='linear'\n), row=1, col=1)\n\nfig.append_trace(go.Bar(\n    x=group_month['month'],\n    y=group_month['counts'],\n), row=2, col=1)\n\n\n\nfig.update_layout(height=1000, width=700, showlegend=False,  xaxis = dict(\n        tickmode = 'linear',\n    ), xaxis2 = dict(tickmode='linear'))\nfig.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:33:45.027306Z","iopub.execute_input":"2023-11-29T11:33:45.027766Z","iopub.status.idle":"2023-11-29T11:33:45.404552Z","shell.execute_reply.started":"2023-11-29T11:33:45.027729Z","shell.execute_reply":"2023-11-29T11:33:45.403108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets see how the data distributed","metadata":{}},{"cell_type":"code","source":"fig = make_subplots(rows=1, cols=2, specs=[[{\"type\": \"pie\"}, {\"type\": \"pie\"}]], subplot_titles=('Distribution of Channels', 'Distribution of Sampling rate'))\n\ngroup_ch = df.groupby([\"channels\"]).size().reset_index(name = \"counts\")\nfig.append_trace(go.Pie(labels = group_ch[\"channels\"] , \n                       values = group_ch[\"counts\"],),\n                row = 1 , col =1)\ngroup_sr = df.groupby([\"sampling_rate\"]).size().reset_index(name = \"counts\")\nfig.append_trace(go.Pie(labels = group_sr[\"sampling_rate\"] , \n                       values = group_sr[\"counts\"],),\n                row = 1 , col =2)","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:33:45.406368Z","iopub.execute_input":"2023-11-29T11:33:45.406807Z","iopub.status.idle":"2023-11-29T11:33:45.450759Z","shell.execute_reply.started":"2023-11-29T11:33:45.406771Z","shell.execute_reply":"2023-11-29T11:33:45.449588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:33:45.452295Z","iopub.execute_input":"2023-11-29T11:33:45.45266Z","iopub.status.idle":"2023-11-29T11:33:45.46255Z","shell.execute_reply.started":"2023-11-29T11:33:45.452629Z","shell.execute_reply":"2023-11-29T11:33:45.46098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_path = '../input/birdsong-recognition/train_audio/aldfly/XC134874.mp3'\nx, sr = librosa.load(audio_path)\nAudio(x, rate=sr)","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:33:45.464515Z","iopub.execute_input":"2023-11-29T11:33:45.465023Z","iopub.status.idle":"2023-11-29T11:33:47.260038Z","shell.execute_reply.started":"2023-11-29T11:33:45.464975Z","shell.execute_reply":"2023-11-29T11:33:47.258853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets see how this audio looks likes","metadata":{}},{"cell_type":"code","source":"!pip install --upgrade librosa\n\nfig, ax = plt.subplots(4, figsize = (20, 9))\nfig.suptitle('Waveplots', fontsize=16)\naudio_path1 = '../input/birdsong-recognition/train_audio/aldfly/XC134874.mp3'\naudio_path2 = '../input/birdsong-recognition/train_audio/amepip/XC111040.mp3'\naudio_path3 = '../input/birdsong-recognition/train_audio/banswa/XC138517.mp3'\naudio_path4 = '../input/birdsong-recognition/train_audio/bkhgro/XC109305.mp3'\n\ny1, sr1 = librosa.load(audio_path1)\ny2, sr2 = librosa.load(audio_path2)\ny3, sr3 = librosa.load(audio_path3)\ny4, sr4 = librosa.load(audio_path4)\n\nlibrosa.display.waveshow(y=y1, sr=sr1, color = \"#3371FF\", ax=ax[0])\nlibrosa.display.waveshow(y=y2 , sr=sr2, color = \"#F7A81E\", ax=ax[1])\nlibrosa.display.waveshow(y=y3 , sr=sr3, color = \"#2BF71E\", ax=ax[2])\nlibrosa.display.waveshow(y=y4 , sr=sr4, color = \"#F71E6D\", ax=ax[3])","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:33:47.261459Z","iopub.execute_input":"2023-11-29T11:33:47.262939Z","iopub.status.idle":"2023-11-29T11:34:05.996744Z","shell.execute_reply.started":"2023-11-29T11:33:47.262894Z","shell.execute_reply":"2023-11-29T11:34:05.995273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Feature Extraction\nThe rate at which the signal changes from positive to zero to negative or from negative to zero to positive","metadata":{}},{"cell_type":"code","source":"# Visualize an STFT power spectrum\n\naudio_path = '../input/birdsong-recognition/train_audio/aldfly/XC134874.mp3'\ny, sr = librosa.load(audio_path)\nplt.figure(figsize=(12, 8))\nD = librosa.amplitude_to_db(librosa.stft(y))\nplt.subplot(4, 2, 1)\nlibrosa.display.specshow(D, y_axis='linear')\nplt.colorbar(format='%+2.0f dB')\nplt.title('Linear-frequency power spectrogram')\n\n# logarithmic scale\n\nplt.subplot(4, 2, 2)\nlibrosa.display.specshow(D, y_axis='log')\nplt.colorbar(format='%+2.0f dB')\nplt.title('Log-frequency power spectrogram')\n\n#CQT scale\n\nCQT = librosa.amplitude_to_db(librosa.cqt(y, sr=sr), ref=np.max)\nplt.subplot(4, 2, 3)\nlibrosa.display.specshow(CQT, y_axis='cqt_hz')\nplt.colorbar(format='%+2.0f dB')\nplt.title('Constant-Q power spectrogram (Hz)')\n\nCQT = librosa.amplitude_to_db(librosa.cqt(y, sr=sr), ref=np.max)\nplt.subplot(4, 2, 4)\nlibrosa.display.specshow(CQT, y_axis='cqt_note')\nplt.colorbar(format='%+2.0f dB')\nplt.title('Constant-Q power spectrogram (note)')\n\n#Chromagram\nC = librosa.feature.chroma_cqt(y=y, sr=sr)\nplt.subplot(4, 2, 5)\nlibrosa.display.specshow(C, y_axis='chroma')\nplt.colorbar()\nplt.title('Chromagram')\n\n# Log power spectrogram\nplt.subplot(4, 2, 6)\nlibrosa.display.specshow(D, x_axis='time', y_axis='log')\nplt.colorbar(format='%+2.0f dB')\nplt.title('Log power spectrogram')","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:34:05.998344Z","iopub.execute_input":"2023-11-29T11:34:05.998737Z","iopub.status.idle":"2023-11-29T11:34:13.048245Z","shell.execute_reply.started":"2023-11-29T11:34:05.998685Z","shell.execute_reply":"2023-11-29T11:34:13.047102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let's zoom in \nn0 = 7000\nn1 = 7100\nplt.figure(figsize=(14, 5))\nplt.plot(y[n0:n1])","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:34:13.053526Z","iopub.execute_input":"2023-11-29T11:34:13.054198Z","iopub.status.idle":"2023-11-29T11:34:13.380043Z","shell.execute_reply.started":"2023-11-29T11:34:13.054156Z","shell.execute_reply":"2023-11-29T11:34:13.378784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"zero_crossings = librosa.zero_crossings(y[n0:n1], pad=False)\nzero_crossings.shape\nzcrs = librosa.feature.zero_crossing_rate(y)\nprint(zcrs.shape)\nplt.figure(figsize=(14, 5))\nplt.plot(zcrs[0])","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:34:13.381673Z","iopub.execute_input":"2023-11-29T11:34:13.38205Z","iopub.status.idle":"2023-11-29T11:34:13.720195Z","shell.execute_reply.started":"2023-11-29T11:34:13.382014Z","shell.execute_reply":"2023-11-29T11:34:13.718782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:34:13.721815Z","iopub.execute_input":"2023-11-29T11:34:13.722266Z","iopub.status.idle":"2023-11-29T11:34:13.731499Z","shell.execute_reply.started":"2023-11-29T11:34:13.722141Z","shell.execute_reply":"2023-11-29T11:34:13.730097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s = (df.dtypes == \"object\")\nlist1= list(s[s].index)\nprint(list1)","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:34:13.733131Z","iopub.execute_input":"2023-11-29T11:34:13.734162Z","iopub.status.idle":"2023-11-29T11:34:13.7441Z","shell.execute_reply.started":"2023-11-29T11:34:13.734121Z","shell.execute_reply":"2023-11-29T11:34:13.742588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.drop([\"filename\"  ,\"url\",\"country\" , \"author\" ,\"secondary_labels\" ,\"latitude\",\"elevation\" ,\"description\" ,\"background\" ,\"license\",\"date\"] , axis =1)","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:34:13.746084Z","iopub.execute_input":"2023-11-29T11:34:13.746888Z","iopub.status.idle":"2023-11-29T11:34:13.766653Z","shell.execute_reply.started":"2023-11-29T11:34:13.746834Z","shell.execute_reply":"2023-11-29T11:34:13.765332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for cols in df.columns:\n    unique = df[cols].unique()\n    print(f\"The values are {cols} : {unique}\")\n# Define the desired order\n","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:34:13.768946Z","iopub.execute_input":"2023-11-29T11:34:13.769532Z","iopub.status.idle":"2023-11-29T11:34:13.850436Z","shell.execute_reply.started":"2023-11-29T11:34:13.769481Z","shell.execute_reply":"2023-11-29T11:34:13.84918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:34:13.851848Z","iopub.execute_input":"2023-11-29T11:34:13.85222Z","iopub.status.idle":"2023-11-29T11:34:13.916232Z","shell.execute_reply.started":"2023-11-29T11:34:13.852186Z","shell.execute_reply":"2023-11-29T11:34:13.914874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.dropna(axis=0)\ndf.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-11-29T12:17:39.118313Z","iopub.execute_input":"2023-11-29T12:17:39.119734Z","iopub.status.idle":"2023-11-29T12:17:39.134885Z","shell.execute_reply.started":"2023-11-29T12:17:39.119661Z","shell.execute_reply":"2023-11-29T12:17:39.133618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# Assuming df is your DataFrame\nlabel_encoder = LabelEncoder()\nmapping_dict = {}\n\nfor column in df.columns:\n    df[column] = label_encoder.fit_transform(df[column])\n    mapping_dict[column] = dict(zip(label_encoder.classes_, label_encoder.transform(label_encoder.classes_)))\n\n# Now df is transformed, and mapping_dict contains the mapping information\n","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:35:59.004817Z","iopub.execute_input":"2023-11-29T11:35:59.005273Z","iopub.status.idle":"2023-11-29T11:35:59.081497Z","shell.execute_reply.started":"2023-11-29T11:35:59.005239Z","shell.execute_reply":"2023-11-29T11:35:59.080013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# mapping_dict","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:36:19.878601Z","iopub.execute_input":"2023-11-29T11:36:19.879049Z","iopub.status.idle":"2023-11-29T11:36:19.885185Z","shell.execute_reply.started":"2023-11-29T11:36:19.879013Z","shell.execute_reply":"2023-11-29T11:36:19.883471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-11-29T11:36:22.743586Z","iopub.execute_input":"2023-11-29T11:36:22.74416Z","iopub.status.idle":"2023-11-29T11:36:22.778302Z","shell.execute_reply.started":"2023-11-29T11:36:22.744114Z","shell.execute_reply":"2023-11-29T11:36:22.777087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score\n\n# Assuming your DataFrame is named 'df'\n# Drop any non-numeric columns or encode them if needed\n# Here, I'm dropping non-numeric columns for simplicity\n\n\n# Split the data into features (X) and target variable (y)\nX = df_numeric.drop('rating', axis=1)\ny = df_numeric['rating']\n\n# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Create a RandomForestClassifier\nmodel = RandomForestClassifier(random_state=42)\n\n# Train the model\nmodel.fit(X_train, y_train)\n\n# Make predictions on the test set\ny_pred = model.predict(X_test)\n\n# Calculate accuracy\naccuracy = accuracy_score(y_test, y_pred)\nprint(f\"Accuracy: {accuracy * 100:.2f}%\")","metadata":{"execution":{"iopub.status.busy":"2023-11-29T12:19:43.585417Z","iopub.execute_input":"2023-11-29T12:19:43.585932Z","iopub.status.idle":"2023-11-29T12:19:47.392896Z","shell.execute_reply.started":"2023-11-29T12:19:43.585894Z","shell.execute_reply":"2023-11-29T12:19:47.391825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}