{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n##for dirname, _, filenames in os.walk('/kaggle/input'):\n   ## for filename in filenames:\n        ##print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-26T23:00:38.388692Z","iopub.execute_input":"2023-05-26T23:00:38.389546Z","iopub.status.idle":"2023-05-26T23:00:38.427281Z","shell.execute_reply.started":"2023-05-26T23:00:38.389482Z","shell.execute_reply":"2023-05-26T23:00:38.426121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set paths to input and output data\n\nTrain_DIR = '/kaggle/input/birdclef-2023/train_audio'\nOUTPUT_DIR = '/kaggle/working/'\n\n\n# Print names of 10 WAV files from the input path\n\nfor path, subdirs, filenames in os.walk(Train_DIR):\n    if filenames:\n        print(os.path.join(path, min(filenames)))   \n        ","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:00:38.429834Z","iopub.execute_input":"2023-05-26T23:00:38.430390Z","iopub.status.idle":"2023-05-26T23:00:42.715603Z","shell.execute_reply.started":"2023-05-26T23:00:38.430336Z","shell.execute_reply":"2023-05-26T23:00:42.714120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot first 20 WAV files for different classes as a waveform and a frequency spectrum\n\n\nimport os\nimport wave\nimport pylab\nimport librosa.display\nimport matplotlib.pyplot as plt\nfrom scipy import signal\nfrom scipy.io import wavfile\nfrom IPython.display import Audio\n\ni =1\n\nfor path, subdirs, filenames in os.walk(Train_DIR):\n    if filenames:\n        samples, sample_rate = librosa.load(os.path.join(path, min(filenames)), sr=None)\n    \n    \n        print(min(filenames))\n        \n        # use the decibel scale to get the final Mel Spectrogram\n        \n        sgram = librosa.stft(samples)\n        sgram_mag, _ = librosa.magphase(sgram)\n        mel_scale_sgram = librosa.feature.melspectrogram(S=sgram_mag, sr=sample_rate)\n        mel_sgram = librosa.amplitude_to_db(mel_scale_sgram, ref=np.min)\n        librosa.display.specshow(mel_sgram, sr=sample_rate, x_axis='time', y_axis='mel')\n        plt.colorbar(format='%+2.0f dB')\n        \n        \n        plt.figure(figsize=(14, 5))\n        librosa.display.waveshow(samples, sr=sample_rate)\n        \n        \n        plt.show()\n        \n         \n        i = i+1\n        if i > 20:\n            break\n\n            \n    ","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:00:42.717879Z","iopub.execute_input":"2023-05-26T23:00:42.719317Z","iopub.status.idle":"2023-05-26T23:01:22.933789Z","shell.execute_reply.started":"2023-05-26T23:00:42.719262Z","shell.execute_reply":"2023-05-26T23:01:22.932306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read metadata file\nmetadata_file = '/kaggle/input/birdclef-2023/train_metadata.csv'\ndf = pd.read_csv(metadata_file)\ndf\n","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:01:22.936993Z","iopub.execute_input":"2023-05-26T23:01:22.938246Z","iopub.status.idle":"2023-05-26T23:01:23.099599Z","shell.execute_reply.started":"2023-05-26T23:01:22.938200Z","shell.execute_reply":"2023-05-26T23:01:23.098233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['primary_label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:01:23.101427Z","iopub.execute_input":"2023-05-26T23:01:23.101806Z","iopub.status.idle":"2023-05-26T23:01:23.121117Z","shell.execute_reply.started":"2023-05-26T23:01:23.101769Z","shell.execute_reply":"2023-05-26T23:01:23.119300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 264 classes of birds sound ","metadata":{}},{"cell_type":"code","source":"# Take relevant columns\ndf_train  = df[['filename', 'primary_label']]\ndf_train","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:01:23.122791Z","iopub.execute_input":"2023-05-26T23:01:23.123324Z","iopub.status.idle":"2023-05-26T23:01:23.146359Z","shell.execute_reply.started":"2023-05-26T23:01:23.123271Z","shell.execute_reply":"2023-05-26T23:01:23.144897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load an audio file. Return the signal as a tensor and the sample rate\ndef open(audio_file):\n    sig, sr = torchaudio.load(audio_file)\n    return (sig, sr)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:01:23.149229Z","iopub.execute_input":"2023-05-26T23:01:23.149784Z","iopub.status.idle":"2023-05-26T23:01:23.157952Z","shell.execute_reply.started":"2023-05-26T23:01:23.149731Z","shell.execute_reply":"2023-05-26T23:01:23.156871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\n\n# Load the audio file\nimport os\n\nAUDIO_FILE = '/kaggle/input/birdclef-2023/train_audio/abethr1/XC128013.ogg'\nsamples, sample_rate = librosa.load(AUDIO_FILE, sr=None)\n\nprint(samples)\nprint(sample_rate)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:01:23.159266Z","iopub.execute_input":"2023-05-26T23:01:23.159806Z","iopub.status.idle":"2023-05-26T23:01:23.221956Z","shell.execute_reply.started":"2023-05-26T23:01:23.159713Z","shell.execute_reply":"2023-05-26T23:01:23.220674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mel_spec = librosa.feature.melspectrogram(y=samples, sr=sample_rate)\nprint(mel_spec)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:01:23.223692Z","iopub.execute_input":"2023-05-26T23:01:23.224085Z","iopub.status.idle":"2023-05-26T23:01:23.333687Z","shell.execute_reply.started":"2023-05-26T23:01:23.224048Z","shell.execute_reply":"2023-05-26T23:01:23.332039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mfccs = librosa.feature.mfcc(y=samples, sr=sample_rate, n_mfcc=40)\nprint(mfccs)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:01:23.342739Z","iopub.execute_input":"2023-05-26T23:01:23.344686Z","iopub.status.idle":"2023-05-26T23:01:23.526652Z","shell.execute_reply.started":"2023-05-26T23:01:23.344603Z","shell.execute_reply":"2023-05-26T23:01:23.524885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def features_extractor(file_name):\n    audio, sample_rate = librosa.load(file_name) \n    mfccs_features = librosa.feature.mfcc(y=audio, sr=sample_rate, n_mfcc=30)\n    mfccs_scaled_features = np.mean(mfccs_features.T,axis=0)\n    \n    return mfccs_scaled_features\n    \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:01:23.528857Z","iopub.execute_input":"2023-05-26T23:01:23.530667Z","iopub.status.idle":"2023-05-26T23:01:23.540552Z","shell.execute_reply.started":"2023-05-26T23:01:23.530590Z","shell.execute_reply":"2023-05-26T23:01:23.538739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" features_extractor('/kaggle/input/birdclef-2023/train_audio/abethr1/XC128013.ogg')\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:01:23.542935Z","iopub.execute_input":"2023-05-26T23:01:23.544889Z","iopub.status.idle":"2023-05-26T23:01:23.753465Z","shell.execute_reply.started":"2023-05-26T23:01:23.544810Z","shell.execute_reply":"2023-05-26T23:01:23.751767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\n# Define the directory containing the audio files\nTrain_DIR = '/kaggle/input/birdclef-2023/train_audio'\n\nimport numpy as np\nfrom tqdm import tqdm\n### Now we iterate through every audio file and extract features \n### using Mel-Frequency Cepstral Coefficients\nextracted_features=[]\nfor index_num,row in tqdm(df_train.iterrows()):\n    file_name = os.path.join(Train_DIR+'/'+ str(row[\"filename\"]))\n    final_class_labels=row[\"primary_label\"]\n    data=features_extractor(file_name)\n    extracted_features.append([data,final_class_labels])\n            \n            ","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:01:23.755661Z","iopub.execute_input":"2023-05-26T23:01:23.757255Z","iopub.status.idle":"2023-05-26T23:59:54.999246Z","shell.execute_reply.started":"2023-05-26T23:01:23.757180Z","shell.execute_reply":"2023-05-26T23:59:54.996480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"extracted_features_df=pd.DataFrame(extracted_features,columns=['feature','class'])\nextracted_features_df.tail()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:59:55.008859Z","iopub.execute_input":"2023-05-26T23:59:55.010614Z","iopub.status.idle":"2023-05-26T23:59:55.073185Z","shell.execute_reply.started":"2023-05-26T23:59:55.010518Z","shell.execute_reply":"2023-05-26T23:59:55.071311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Split the dataset into independent and dependent dataset\nX=np.array(extracted_features_df['feature'].tolist())\ny=np.array(extracted_features_df['class'].tolist())","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:59:55.081502Z","iopub.execute_input":"2023-05-26T23:59:55.082951Z","iopub.status.idle":"2023-05-26T23:59:55.120243Z","shell.execute_reply.started":"2023-05-26T23:59:55.082861Z","shell.execute_reply":"2023-05-26T23:59:55.118734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:59:55.122100Z","iopub.execute_input":"2023-05-26T23:59:55.122602Z","iopub.status.idle":"2023-05-26T23:59:55.131846Z","shell.execute_reply.started":"2023-05-26T23:59:55.122549Z","shell.execute_reply":"2023-05-26T23:59:55.130250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sklearn\nfrom sklearn.preprocessing import StandardScaler\n\nX = sklearn.preprocessing.minmax_scale(X)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:59:55.134481Z","iopub.execute_input":"2023-05-26T23:59:55.135360Z","iopub.status.idle":"2023-05-26T23:59:55.240253Z","shell.execute_reply.started":"2023-05-26T23:59:55.135218Z","shell.execute_reply":"2023-05-26T23:59:55.238554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Label Encoding\ny=np.array(pd.get_dummies(y))","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:59:55.246937Z","iopub.execute_input":"2023-05-26T23:59:55.247384Z","iopub.status.idle":"2023-05-26T23:59:55.275979Z","shell.execute_reply.started":"2023-05-26T23:59:55.247346Z","shell.execute_reply":"2023-05-26T23:59:55.274768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:59:55.278132Z","iopub.execute_input":"2023-05-26T23:59:55.278867Z","iopub.status.idle":"2023-05-26T23:59:55.285625Z","shell.execute_reply.started":"2023-05-26T23:59:55.278828Z","shell.execute_reply":"2023-05-26T23:59:55.284494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Train Test Split\nfrom sklearn.model_selection import train_test_split\nX_train,X_test,y_train,y_test=train_test_split(X,y,test_size=0.2,random_state=0)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:59:55.287561Z","iopub.execute_input":"2023-05-26T23:59:55.288308Z","iopub.status.idle":"2023-05-26T23:59:55.387485Z","shell.execute_reply.started":"2023-05-26T23:59:55.288269Z","shell.execute_reply":"2023-05-26T23:59:55.386135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X)\nprint(y)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T23:59:55.389170Z","iopub.execute_input":"2023-05-26T23:59:55.389579Z","iopub.status.idle":"2023-05-26T23:59:55.397633Z","shell.execute_reply.started":"2023-05-26T23:59:55.389541Z","shell.execute_reply":"2023-05-26T23:59:55.396473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense,Dropout,Activation,Flatten\nfrom tensorflow.keras.optimizers import Adam\nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2023-05-27T00:39:21.959562Z","iopub.execute_input":"2023-05-27T00:39:21.960188Z","iopub.status.idle":"2023-05-27T00:39:21.967669Z","shell.execute_reply.started":"2023-05-27T00:39:21.960136Z","shell.execute_reply":"2023-05-27T00:39:21.966101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model=Sequential()\n###first layer\nmodel.add(Dense(20,input_shape=(30,)))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\n\n###second layer\nmodel.add(Dense(500))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\n\n###third layer\nmodel.add(Dense(1000))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\n\n###second layer\nmodel.add(Dense(2000))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\n\n###third layer\nmodel.add(Dense(1000))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\n\n###final layer\nmodel.add(Dense(y.shape[1]))\nmodel.add(Activation('softmax'))","metadata":{"execution":{"iopub.status.busy":"2023-05-27T00:43:14.669483Z","iopub.execute_input":"2023-05-27T00:43:14.670054Z","iopub.status.idle":"2023-05-27T00:43:14.857572Z","shell.execute_reply.started":"2023-05-27T00:43:14.669986Z","shell.execute_reply":"2023-05-27T00:43:14.856239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T00:43:18.137182Z","iopub.execute_input":"2023-05-27T00:43:18.138792Z","iopub.status.idle":"2023-05-27T00:43:18.197594Z","shell.execute_reply.started":"2023-05-27T00:43:18.138708Z","shell.execute_reply":"2023-05-27T00:43:18.196329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss='categorical_crossentropy',metrics=['accuracy'],optimizer='adam')","metadata":{"execution":{"iopub.status.busy":"2023-05-27T00:43:24.395951Z","iopub.execute_input":"2023-05-27T00:43:24.396471Z","iopub.status.idle":"2023-05-27T00:43:24.413431Z","shell.execute_reply.started":"2023-05-27T00:43:24.396431Z","shell.execute_reply":"2023-05-27T00:43:24.411863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Trianing my model\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom datetime import datetime \n\nnum_epochs = 100\nnum_batch_size = 32\n\ncheckpointer = ModelCheckpoint(filepath='saved_models/audio_classification.hdf5', \n                               verbose=1, save_best_only=True)\nstart = datetime.now()\n\nmodel.fit(X_train, y_train, batch_size=num_batch_size, epochs=num_epochs, validation_data=(X_test, y_test), callbacks=[checkpointer], verbose=1)\n\n\nduration = datetime.now() - start\nprint(\"Training completed in time: \", duration)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T00:43:27.153846Z","iopub.execute_input":"2023-05-27T00:43:27.154288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_accuracy=model.evaluate(X_test,y_test,verbose=0)\nprint(test_accuracy[1])","metadata":{"execution":{"iopub.status.busy":"2023-05-27T00:02:40.880538Z","iopub.execute_input":"2023-05-27T00:02:40.881090Z","iopub.status.idle":"2023-05-27T00:02:41.148642Z","shell.execute_reply.started":"2023-05-27T00:02:40.881009Z","shell.execute_reply":"2023-05-27T00:02:41.146779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}