{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n#         print(os.path.join(dirname, filename))\n        pass\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-14T15:54:35.165825Z","iopub.execute_input":"2022-10-14T15:54:35.166200Z","iopub.status.idle":"2022-10-14T15:54:35.292728Z","shell.execute_reply.started":"2022-10-14T15:54:35.166171Z","shell.execute_reply":"2022-10-14T15:54:35.291637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport ast\nimport random\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nfrom tqdm import tqdm\nimport torchaudio\nimport IPython.display as ipd\nfrom collections import Counter\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import f1_score\n\nimport torch\nimport torch.nn as nn\nfrom torch.optim import Adam\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import models\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class config:\n    seed = 2022\n    num_fold = 5\n    sample_rate = 32_000\n    n_fft = 1024\n    hop_length = 512\n    n_mels = 64\n    duration = 7\n    num_classes = 152\n    train_batch_size = 32\n    valid_batch_size = 64\n    model_name = 'resnet50'\n    epochs = 2\n    device = 'cuda' if torch.cuda.is_available() else 'cpu'\n    learning_rate = 1e-4","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n\nseed_everything(config.seed)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/birdclef-2022/train_metadata.csv')\ndf.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 6))\n\nsns.countplot(df['primary_label'])\nplt.xticks(rotation=90)\nplt.title('Distribution of Primary Labels', fontsize=20)\n\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['type'] = df['type'].apply(lambda x : ast.literal_eval(x))\n\ntop = Counter([typ.lower() for lst in df['type'] for typ in lst])\n\ntop = dict(top.most_common(10))\n\nplt.figure(figsize=(20, 6))\n\nsns.barplot(x=list(top.keys()), y=list(top.values()), palette='hls')\nplt.title(\"Top 10 song types\")\n\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filename_1 = df['filename'].values[0]\n\nipd.Audio(f\"/kaggle/input/birdclef-2022/train_audio/{filename_1}\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filename_2 = df['filename'].values[-1]\nipd.Audio(f'/kaggle/input/birdclef-2022/train_audio/{filename_2}')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(2, 1, figsize=(20, 10))\nfig.suptitle(\"Sound Waves\", fontsize=15)\n\nsignal_1, sr = torchaudio.load(f\"/kaggle/input/birdclef-2022/train_audio/{filename_1}\")\n\n# Audio comes in as the sound (signal), and sample rate, (sr), which tells how many samples per second there are\n\nsns.lineplot(x=np.arange(len(signal_1[0,:].detach().numpy())), y=signal_1[0,:].detach().numpy(), ax=ax[0], color='#4400FF')\nax[0].set_title(\"Audio 1\")\n\n\nsignal_2, sr = torchaudio.load(f\"/kaggle/input/birdclef-2022/train_audio/{filename_2}\")\nsns.lineplot(x=np.arange(len(signal_2[0,:].detach().numpy())), y=signal_2[0,:].detach().numpy(), ax=ax[1], color='#4400FF')\nax[1].set_title(\"Audio 2\")\n\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoder = LabelEncoder()\n\ndf['primary_label_encoded'] = encoder.fit_transform(df['primary_label'])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=config.num_fold)\nfor k, (_, val_ind) in enumerate(skf.split(X=df, y=df['primary_label_encoded'])):\n    df.loc[val_ind, 'fold'] = k","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(20, 7))\nfig.suptitle(\"Mel Spectrogram\", fontsize=15)\n\nmel_spectrogram = torchaudio.transforms.MelSpectrogram(sample_rate=config.sample_rate, n_fft=config.n_fft, hop_length=config.hop_length, n_mels=config.n_mels)\n\nmel_1 = mel_spectrogram(signal_1)\nax[0].imshow(mel_1.log2()[0,:,:].detach().numpy(), aspect='auto', cmap='cool')\nax[0].set_title('Audio 1')\n\nmel_2 = mel_spectrogram(signal_2)\nax[1].imshow(mel_2.log2()[0,:,:].detach().numpy(), aspect='auto', cmap='cool')\n\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BirdClefDataset(Dataset):\n    def __init__(self, df, transformation, target_sample_rate, duration):\n        self.audio_paths = df['filename'].values\n        self.labels = df['primary_label_encoded'].values\n        self.transformation = transformation\n        self.target_sample_rate = target_sample_rate\n        self.num_samples = target_sample_rate * duration\n    \n    def __len__(self):\n        return len(self.audio_paths)\n    \n    def __getitem__(self, index):\n        audio_path = f'/kaggle/input/birdclef-2022/train_audio/{self.audio_paths[index]}'\n        signal, sr = torchaudio.load(audio_path)\n\n        # Ensure all audio is uniform sample rate\n        if sr != self.target_sample_rate:\n            resampler = torchaudio.transforms.Resample(sr, self.target_sample_rate)\n            signal = resampler(signal)\n        \n        # Make sure all the signals are the same length\n        if signal.shape[1] > self.num_samples:\n            signal = signal[:, :self.num_samples]\n        \n        if signal.shape[1] < self.num_samples:\n            num_missing_samples = self.num_samples - signal.shape[1]\n            last_dim_padding = (0, num_missing_samples)\n            signal = F.pad(signal, last_dim_padding)\n        \n        # Get the mel spectrogram from the signal\n        mel = self.transformation(signal)\n\n        # Since we are using pretrained image recognition models, we need 'RGB', which we will simulate\n        image = torch.cat([mel, mel, mel])\n\n        # Normalize the 'image'\n        max_val = torch.abs(image).max()\n        image = image / max_val\n\n        label = torch.tensor(self.labels[index])\n\n        return image, label\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_data(fold):\n    train_df = df[df['fold'] != fold].reset_index(drop=True)\n    valid_df = df[df['fold'] == fold].reset_index(drop=True)\n\n    train_dataset = BirdClefDataset(train_df, mel_spectrogram, config.sample_rate, config.duration)\n    valid_dataset = BirdClefDataset(valid_df, mel_spectrogram, config.sample_rate, config.duration)\n\n    train_loader = DataLoader(train_dataset, batch_size=config.train_batch_size, shuffle=True)\n    valid_loader = DataLoader(valid_dataset, batch_size=config.valid_batch_size, shuffle=True)\n\n    return train_loader, valid_loader","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv('/kaggle/input/birdclef-2022/sample_submission.csv')\nfor i in range(len(sample_submission)):\n    sample = sample_submission.row_id[i]\n    key = sample.split(\"_\")[0] + \"_\" + sample.split(\"_\")[1] + \"_\" + sample.split(\"_\")[3]\n    target_bird = sample.split(\"_\")[2]\n    print(key, target_bird)\n\n    sample_submission.iat[i, 1] = True # This will eventually be where we do predictions, but for now just true\nsample_submission.to_csv(\"submission.csv\", index=False)","metadata":{},"execution_count":null,"outputs":[]}]}