{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install comet_ml","metadata":{"execution":{"iopub.status.busy":"2023-03-10T06:31:03.631516Z","iopub.execute_input":"2023-03-10T06:31:03.632279Z","iopub.status.idle":"2023-03-10T06:31:13.529748Z","shell.execute_reply.started":"2023-03-10T06:31:03.632231Z","shell.execute_reply":"2023-03-10T06:31:13.528403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport librosa\nimport glob\nimport matplotlib.pyplot as plt\n\nimport csv\nimport io\n\nimport plotly.express as px\n\nfrom IPython.display import Audio","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:53:18.404658Z","iopub.execute_input":"2023-03-10T17:53:18.405012Z","iopub.status.idle":"2023-03-10T17:53:24.030518Z","shell.execute_reply.started":"2023-03-10T17:53:18.404981Z","shell.execute_reply":"2023-03-10T17:53:24.029489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics \nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split \nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Activation\nfrom keras.optimizers import Adam\nfrom keras.utils import to_categorical","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:53:28.240841Z","iopub.execute_input":"2023-03-10T17:53:28.241783Z","iopub.status.idle":"2023-03-10T17:53:35.325804Z","shell.execute_reply.started":"2023-03-10T17:53:28.241731Z","shell.execute_reply":"2023-03-10T17:53:35.324696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set up logging with comet_ml\n# from comet_ml import Experiment\n# from kaggle_secrets import UserSecretsClient\n# user_secrets = UserSecretsClient()\n\n# api_key = user_secrets.get_secret(\"comet_api_key\")\n# project_name = user_secrets.get_secret(\"comet_project_name\")\n# workspace = user_secrets.get_secret(\"comet_workspace\")\n\n# Create an experiment with your api key\n# experiment = Experiment(\n#     api_key=api_key,\n#     project_name=project_name,\n#     workspace=workspace,\n# )","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:53:42.565005Z","iopub.execute_input":"2023-03-10T17:53:42.565780Z","iopub.status.idle":"2023-03-10T17:53:42.571744Z","shell.execute_reply.started":"2023-03-10T17:53:42.565729Z","shell.execute_reply":"2023-03-10T17:53:42.570511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rel_path = \"/kaggle/input/birdclef-2023/train_audio/\"","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:53:43.462853Z","iopub.execute_input":"2023-03-10T17:53:43.463219Z","iopub.status.idle":"2023-03-10T17:53:43.468556Z","shell.execute_reply.started":"2023-03-10T17:53:43.463186Z","shell.execute_reply":"2023-03-10T17:53:43.467506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_files = glob.glob(\"/kaggle/input/birdclef-2023/train_audio/*/*.ogg\")\ndf = pd.read_csv(\"/kaggle/input/birdclef-2023/train_metadata.csv\")\ndf","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:53:44.082558Z","iopub.execute_input":"2023-03-10T17:53:44.082937Z","iopub.status.idle":"2023-03-10T17:53:47.281003Z","shell.execute_reply.started":"2023-03-10T17:53:44.082906Z","shell.execute_reply":"2023-03-10T17:53:47.279933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot sample densities globally\nfig = px.density_mapbox(df, lat='latitude', lon='longitude', z='rating', radius=10,\n                        center=dict(lat=0, lon=180), zoom=0,\n                        mapbox_style=\"stamen-terrain\")\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:53:47.283155Z","iopub.execute_input":"2023-03-10T17:53:47.283551Z","iopub.status.idle":"2023-03-10T17:53:48.544535Z","shell.execute_reply.started":"2023-03-10T17:53:47.283510Z","shell.execute_reply":"2023-03-10T17:53:48.543467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prim_labels = df[\"primary_label\"].unique()[:6]\nscientific_names = df[\"scientific_name\"].unique()[:6]\n\nlabels = [f'{prim_label}-{scientific_name}' for prim_label, scientific_name in zip(prim_labels, scientific_names)]\nlabels","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:53:48.546561Z","iopub.execute_input":"2023-03-10T17:53:48.546963Z","iopub.status.idle":"2023-03-10T17:53:48.559632Z","shell.execute_reply.started":"2023-03-10T17:53:48.546924Z","shell.execute_reply":"2023-03-10T17:53:48.558418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files = {}\n\nfor i in range(len(prim_labels)):\n    temp = df[df[\"primary_label\"] == prim_labels[i]].iloc[0]\n    path = rel_path + temp[\"filename\"]\n    files[labels[i]] = path\n\nfiles","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:53:55.545166Z","iopub.execute_input":"2023-03-10T17:53:55.545542Z","iopub.status.idle":"2023-03-10T17:53:55.564106Z","shell.execute_reply.started":"2023-03-10T17:53:55.545508Z","shell.execute_reply":"2023-03-10T17:53:55.562989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(15, 15))\n# experiment.log_image('/kaggle/working/species_examples.png')\nfig.subplots_adjust(hspace=0.4, wspace=0.4)\nfor i, label in enumerate(labels):\n    fn = files[label]\n    fig.add_subplot(5, 2, i+1)\n    plt.title(label)\n    data, sample_rate = librosa.load(fn)\n    librosa.display.waveshow(data, sr=sample_rate)\n    \nplt.savefig('/kaggle/working/species_examples.png')","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:53:55.762320Z","iopub.execute_input":"2023-03-10T17:53:55.763038Z","iopub.status.idle":"2023-03-10T17:54:08.722108Z","shell.execute_reply.started":"2023-03-10T17:53:55.762998Z","shell.execute_reply":"2023-03-10T17:54:08.721013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Log graphic of waveforms to Comet\n# experiment.log_image('/kaggle/working/species_examples.png')","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:54:08.723885Z","iopub.execute_input":"2023-03-10T17:54:08.725571Z","iopub.status.idle":"2023-03-10T17:54:08.730355Z","shell.execute_reply.started":"2023-03-10T17:54:08.725528Z","shell.execute_reply":"2023-03-10T17:54:08.729027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Log audio files to Comet for debugging\n# for label in labels:\n#     fn = files[label]\n#     experiment.log_audio(fn, metadata = {'name': label})","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:54:08.732073Z","iopub.execute_input":"2023-03-10T17:54:08.732964Z","iopub.status.idle":"2023-03-10T17:54:08.741896Z","shell.execute_reply.started":"2023-03-10T17:54:08.732923Z","shell.execute_reply":"2023-03-10T17:54:08.740826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract features from the audio files using librosa\ndef extract_features(file_name):\n    audio, sample_rate = librosa.load(file_name)\n    # mel frequency cepstral coefficients\n    mfcc = librosa.feature.mfcc(y=audio, sr=sample_rate, n_mfcc=40)\n    return np.mean(mfcc.T, axis=0)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:54:08.744048Z","iopub.execute_input":"2023-03-10T17:54:08.744325Z","iopub.status.idle":"2023-03-10T17:54:08.751662Z","shell.execute_reply.started":"2023-03-10T17:54:08.744299Z","shell.execute_reply":"2023-03-10T17:54:08.750649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Credit to Gideon Mendels https://towardsdatascience.com/how-to-apply-machine-learning-and-deep-learning-methods-to-audio-analysis-615e286fcbbc","metadata":{}},{"cell_type":"code","source":"def generate_features(df):\n    features = []\n    for index, row in df.iterrows():\n        if index > 1000:\n            break\n        filename = rel_path + row[\"filename\"]\n\n        class_label = row[\"primary_label\"]\n        mfcc = extract_features(filename)\n        features.append([mfcc, class_label])\n    return features","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:55:41.917507Z","iopub.execute_input":"2023-03-10T17:55:41.918216Z","iopub.status.idle":"2023-03-10T17:55:41.924099Z","shell.execute_reply.started":"2023-03-10T17:55:41.918173Z","shell.execute_reply":"2023-03-10T17:55:41.923052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\ndurations = []\nfor _ in range(10):\n    start = time.time()\n    generate_features(df[:10])\n    end = time.time()\n    durations.append(end-start)\n\nprint(f'mean time:{np.mean(durations)}, min time: {np.min(durations)}, max time: {np.max(durations)}, range: {np.max(durations) - np.min(durations)}')","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:55:46.208438Z","iopub.execute_input":"2023-03-10T17:55:46.209139Z","iopub.status.idle":"2023-03-10T17:56:00.693525Z","shell.execute_reply.started":"2023-03-10T17:55:46.209099Z","shell.execute_reply":"2023-03-10T17:56:00.690234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"If the function takes 1.5 seconds for a dataframe of 10 rows, I can extrapolate to the larger dataset to find the estimated time to complete generating the features for the entire dataset.","metadata":{}},{"cell_type":"code","source":"time_to_complete = (len(df) / 10) * np.mean(durations)\nmax_time_to_complete = (len(df) / 10) * np.max(durations)\nmin_time_to_complete = (len(df) / 10) * np.min(durations)\n\nrange_completion = max_time_to_complete - min_time_to_complete\n\nprint(f'mean time: {time_to_complete/60}, min time: {min_time_to_complete/60}, max time: {max_time_to_complete/60}, range: {range_completion/60}')","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:54:32.685730Z","iopub.execute_input":"2023-03-10T17:54:32.686755Z","iopub.status.idle":"2023-03-10T17:54:32.695812Z","shell.execute_reply.started":"2023-03-10T17:54:32.686712Z","shell.execute_reply":"2023-03-10T17:54:32.694749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Generating the features for the dataframe will take on average 41.6 minutes with a worst case scenario of 51.77 minutes","metadata":{}},{"cell_type":"code","source":"start = time.time()\nfeatures = generate_features(df)\nend = time.time()\n\nfeatures_df = pd.DataFrame(features, columns=[\"feature\", \"primary_label\"])\nfeatures_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-10T17:56:58.184324Z","iopub.execute_input":"2023-03-10T17:56:58.184870Z","iopub.status.idle":"2023-03-10T17:59:41.086678Z","shell.execute_reply.started":"2023-03-10T17:56:58.184817Z","shell.execute_reply":"2023-03-10T17:59:41.085316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print((end - start) / 60)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T18:00:05.173984Z","iopub.execute_input":"2023-03-10T18:00:05.174622Z","iopub.status.idle":"2023-03-10T18:00:05.180538Z","shell.execute_reply.started":"2023-03-10T18:00:05.174566Z","shell.execute_reply":"2023-03-10T18:00:05.179487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfrom keras.utils import to_categorical\n# Convert features and corresponding classification labels into numpy arrays\nX = np.array(features_df.feature.tolist())\ny = np.array(features_df.primary_label.tolist())\n# Encode the classification labels\nle = LabelEncoder()\nyy = to_categorical(le.fit_transform(y))","metadata":{"execution":{"iopub.status.busy":"2023-03-10T18:00:09.806939Z","iopub.execute_input":"2023-03-10T18:00:09.807285Z","iopub.status.idle":"2023-03-10T18:00:09.816434Z","shell.execute_reply.started":"2023-03-10T18:00:09.807253Z","shell.execute_reply":"2023-03-10T18:00:09.815191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split the dataset \nfrom sklearn.model_selection import train_test_split \nx_train, x_test, y_train, y_test = train_test_split(X, yy, test_size=0.2, random_state = 127)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T18:00:10.931912Z","iopub.execute_input":"2023-03-10T18:00:10.932847Z","iopub.status.idle":"2023-03-10T18:00:10.943711Z","shell.execute_reply.started":"2023-03-10T18:00:10.932794Z","shell.execute_reply":"2023-03-10T18:00:10.942654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_labels = yy.shape[1]\nfilter_size = 2\ndef build_model_graph(input_shape=(40,)):\n    model = Sequential()\n    model.add(Dense(256))\n    model.add(Activation('relu'))\n    model.add(Dropout(0.5))\n    model.add(Dense(256))\n    model.add(Activation('relu'))\n    model.add(Dropout(0.5))\n    model.add(Dense(num_labels))\n    model.add(Activation('softmax'))\n    # Compile the model\n    model.compile(loss='categorical_crossentropy', metrics=['accuracy'], optimizer='adam')\n    return model\nmodel = build_model_graph()","metadata":{"execution":{"iopub.status.busy":"2023-03-10T18:00:11.565515Z","iopub.execute_input":"2023-03-10T18:00:11.566223Z","iopub.status.idle":"2023-03-10T18:00:15.303392Z","shell.execute_reply.started":"2023-03-10T18:00:11.566185Z","shell.execute_reply":"2023-03-10T18:00:15.302386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.callbacks import ModelCheckpoint \nfrom datetime import datetime \nnum_epochs = 100\nnum_batch_size = 32\nmodel.fit(x_train, y_train, batch_size=num_batch_size, epochs=num_epochs, validation_data=(x_test, y_test), verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T18:00:56.122762Z","iopub.execute_input":"2023-03-10T18:00:56.123372Z","iopub.status.idle":"2023-03-10T18:01:17.595617Z","shell.execute_reply.started":"2023-03-10T18:00:56.123335Z","shell.execute_reply":"2023-03-10T18:01:17.594584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluating the model on the training and testing set\nscore = model.evaluate(x_train, y_train, verbose=0)\nprint(\"Training Accuracy: {0:.2%}\".format(score[1]))\nscore = model.evaluate(x_test, y_test, verbose=0)\nprint(\"Testing Accuracy: {0:.2%}\".format(score[1]))","metadata":{"execution":{"iopub.status.busy":"2023-03-10T18:01:43.268379Z","iopub.execute_input":"2023-03-10T18:01:43.268783Z","iopub.status.idle":"2023-03-10T18:01:43.457312Z","shell.execute_reply.started":"2023-03-10T18:01:43.268748Z","shell.execute_reply":"2023-03-10T18:01:43.456318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}