{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport warnings\nwarnings.filterwarnings(action='ignore')\n\ntrain = pd.read_csv('../input/birdclef-2022/train_metadata.csv',)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T20:15:51.807369Z","iopub.execute_input":"2023-05-15T20:15:51.807838Z","iopub.status.idle":"2023-05-15T20:15:51.912921Z","shell.execute_reply.started":"2023-05-15T20:15:51.807808Z","shell.execute_reply":"2023-05-15T20:15:51.912085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Code adapted from https://www.kaggle.com/shahules/bird-watch-complete-eda-fe\n# Make sure to check out the entire notebook.\n\nimport plotly.graph_objects as go\n\n# Unique eBird codes\nspecies = train['primary_label'].value_counts()\n\n# Make bar chart\nfig = go.Figure(data=[go.Bar(y=species.values, x=species.index)],\n                layout=go.Layout(margin=go.layout.Margin(l=0, r=0, b=10, t=50)))\n\n# Show chart\nfig.update_layout(title='Number of traning samples per species')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T19:31:27.520454Z","iopub.execute_input":"2023-05-15T19:31:27.521834Z","iopub.status.idle":"2023-05-15T19:31:27.870072Z","shell.execute_reply.started":"2023-05-15T19:31:27.521793Z","shell.execute_reply":"2023-05-15T19:31:27.868644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom types import SimpleNamespace\nimport numpy as np\ncfg = SimpleNamespace()\ncfg.data_dir = \"../input/birdclef-2022/\"\ncfg.train_data_folder = cfg.data_dir + \"train_audio/\"\ncfg.val_data_folder = cfg.data_dir + \"train_audio/\"","metadata":{"execution":{"iopub.status.busy":"2023-05-15T20:19:27.502439Z","iopub.execute_input":"2023-05-15T20:19:27.502903Z","iopub.status.idle":"2023-05-15T20:19:27.509236Z","shell.execute_reply.started":"2023-05-15T20:19:27.502868Z","shell.execute_reply":"2023-05-15T20:19:27.507886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom scipy.io import wavfile\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow import keras\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense\nfrom tensorflow.keras.utils import to_categorical\n\n# Load and preprocess the dataset\ncsv_path = \"../input/birdclef-2022/train_metadata.csv\"\n\n# Read CSV file\ndata_df = pd.read_csv(csv_path)\n\n# Extract audio paths and labels\naudio_paths = data_df['filename'].tolist()\nlabels = data_df['common_name'].tolist()\n\nX = []\ny = []\n\n# Iterate over audio paths\nfor audio_path, label in zip(audio_paths, labels):\n    sample_rate, audio = wavfile.read(audio_path)\n    \n    # Preprocessing steps (e.g., pad, normalize, spectrogram conversion)\n    # ...\n    \n    # Append preprocessed audio and label to X and y\n    X.append(preprocessed_audio)\n    y.append(label)\n\nX = np.array(X)\ny = np.array(y)\n\n# Split the dataset into train and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Normalize the input features\nX_train = X_train / np.max(X_train)\nX_test = X_test / np.max(X_test)\n\n# Convert labels to categorical format\nnum_classes = len(np.unique(labels))\ny_train = to_categorical(y_train, num_classes)\ny_test = to_categorical(y_test, num_classes)\n\n# Define the CNN architecture\nmodel = keras.Sequential([\n    Conv2D(32, kernel_size=(3, 3), activation='relu', input_shape=(X_train.shape[1:])),\n    MaxPooling2D(pool_size=(2, 2)),\n    Conv2D(64, kernel_size=(3, 3), activation='relu'),\n    MaxPooling2D(pool_size=(2, 2)),\n    Flatten(),\n    Dense(128, activation='relu'),\n    Dense(num_classes, activation='softmax')\n])\n\n# Compile the model\nmodel.compile(loss=keras.losses.categorical_crossentropy,\n              optimizer=keras.optimizers.Adam(),\n              metrics=['accuracy'])\n\n# Train the model\nmodel.fit(X_train, y_train, batch_size=32, epochs=10, validation_data=(X_test, y_test))\n\n# Evaluate the model on test data\ntest_loss, test_accuracy = model.evaluate(X_test, y_test)\nprint(\"Test Loss:\", test_loss)\nprint(\"Test Accuracy:\", test_accuracy)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-15T20:20:59.243291Z","iopub.execute_input":"2023-05-15T20:20:59.243768Z","iopub.status.idle":"2023-05-15T20:20:59.384837Z","shell.execute_reply.started":"2023-05-15T20:20:59.243731Z","shell.execute_reply":"2023-05-15T20:20:59.383328Z"},"trusted":true},"execution_count":null,"outputs":[]}]}