{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-14T00:09:34.654854Z","iopub.execute_input":"2023-04-14T00:09:34.655221Z","iopub.status.idle":"2023-04-14T00:09:34.661667Z","shell.execute_reply.started":"2023-04-14T00:09:34.655190Z","shell.execute_reply":"2023-04-14T00:09:34.660324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport os\nimport io\nimport librosa\nimport torchaudio\nimport numpy as np\nimport matplotlib.pyplot as plt    \nfrom torch.utils.data import Dataset, TensorDataset, DataLoader\nimport glob\nimport pandas as pd \nfrom torch.utils.data import Dataset, TensorDataset, DataLoader\nimport torch\nimport librosa\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt    \nimport numpy as np\nfrom pathlib import Path\nimport torchvision\nfrom torchvision import datasets, models, transforms\ndevice = torch.device(\"cpu\")","metadata":{"execution":{"iopub.status.busy":"2023-04-14T00:09:35.859239Z","iopub.execute_input":"2023-04-14T00:09:35.859616Z","iopub.status.idle":"2023-04-14T00:09:39.196968Z","shell.execute_reply.started":"2023-04-14T00:09:35.859585Z","shell.execute_reply":"2023-04-14T00:09:39.196109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_samples = list(glob.glob(\"/kaggle/input/birdclef-2023/test_soundscapes/*.ogg\"))\ntest_samples","metadata":{"execution":{"iopub.status.busy":"2023-04-14T00:10:06.492673Z","iopub.execute_input":"2023-04-14T00:10:06.493033Z","iopub.status.idle":"2023-04-14T00:10:06.502215Z","shell.execute_reply.started":"2023-04-14T00:10:06.492999Z","shell.execute_reply":"2023-04-14T00:10:06.501358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TrainAudioDataset(Dataset):\n    def __init__(self):\n        self.test_samples = list(glob.glob(\"/kaggle/input/birdclef-2023/test_soundscapes/*.ogg\"))\n         \n    def __len__(self):\n        return len(self.test_samples)\n\n    def __getitem__(self, idx): \n        path = self.test_samples[idx]\n        file_id = path.split(\".ogg\")[0].split(\"/\")[-1]\n        \n        waveform, sample_rate = torchaudio.load(path)\n        return ((waveform[0], sample_rate), file_id)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T00:10:09.000965Z","iopub.execute_input":"2023-04-14T00:10:09.001340Z","iopub.status.idle":"2023-04-14T00:10:09.008566Z","shell.execute_reply.started":"2023-04-14T00:10:09.001303Z","shell.execute_reply":"2023-04-14T00:10:09.007151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def stride_trick(a, stride_length, stride_step):\n     \"\"\"\n     apply framing using the stride trick from numpy.\n\n     Args:\n         a (array) : signal array.\n         stride_length (int) : length of the stride.\n         stride_step (int) : stride step.\n\n     Returns:\n         blocked/framed array.\n     \"\"\"\n     nrows = ((a.size - stride_length) // stride_step) + 1\n     n = a.strides[0]\n     return np.lib.stride_tricks.as_strided(a,\n                                            shape=(nrows, stride_length),\n                                            strides=(stride_step*n, n))\n\ndef framing(sig, fs=16000, win_len=0.025, win_hop=0.01):\n     \"\"\"\n     transform a signal into a series of overlapping frames (=Frame blocking).\n\n     Args:\n         sig     (array) : a mono audio signal (Nx1) from which to compute features.\n         fs        (int) : the sampling frequency of the signal we are working with.\n                           Default is 16000.\n         win_len (float) : window length in sec.\n                           Default is 0.025.\n         win_hop (float) : step between successive windows in sec.\n                           Default is 0.01.\n\n     Returns:\n         array of frames.\n         frame length.\n\n     Notes:\n     ------\n         Uses the stride trick to accelerate the processing.\n     \"\"\"\n     # run checks and assertions\n     if win_len < win_hop: print(\"ParameterError: win_len must be larger than win_hop.\")\n\n     # compute frame length and frame step (convert from seconds to samples)\n     frame_length = win_len * fs\n     frame_step = win_hop * fs\n     signal_length = len(sig)\n     frames_overlap = frame_length - frame_step\n\n     # compute number of frames and left sample in order to pad if needed to make\n     # sure all frames have equal number of samples  without truncating any samples\n     # from the original signal\n     rest_samples = np.abs(signal_length - frames_overlap) % np.abs(frame_length - frames_overlap)\n     pad_signal = np.append(sig, np.array([0] * int(frame_step - rest_samples) * int(rest_samples != 0.)))\n\n     # apply stride trick\n     frames = stride_trick(pad_signal, int(frame_length), int(frame_step))\n     return frames, frame_length","metadata":{"execution":{"iopub.status.busy":"2023-04-14T00:10:11.125420Z","iopub.execute_input":"2023-04-14T00:10:11.125791Z","iopub.status.idle":"2023-04-14T00:10:11.135793Z","shell.execute_reply.started":"2023-04-14T00:10:11.125759Z","shell.execute_reply":"2023-04-14T00:10:11.134674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataProcessing: \n    @staticmethod\n    def record_to_frames(waveform, sample_rate, frame_size=5):\n        p1d = (1, sample_rate * frame_size)\n        out = torch.nn.functional.pad(waveform, p1d, \"constant\", 0)\n        return out.unfold(0, sample_rate * frame_size, sample_rate * frame_size)\n\n    @staticmethod\n    def my_collate(batch):\n        frames = []\n        labels = []\n        for (data,file_id) in batch: \n                (waveform, sample_rate) = data\n                l_frames, frame_length = framing(waveform, sample_rate, 5.0, 5.0)\n   \n                for index in range(len(l_frames)):\n                    frame = l_frames[index]\n                    # audio_spectogram = spectogram(frame)\n                    # audio_spectogram = audio_spectogram.repeat(3, 1, 1)\n                    frames.append((frame, sample_rate, f'{file_id}_{(index+1)*5}'))\n                    labels.append(file_id) \n        return [frames, labels]\n    \n    @staticmethod\n    def melgram_v2(audio, sample_rate,  to_file):\n         \n        plt.figure(figsize=(3,2))\n        plt.axis('off')  # no axis\n        plt.axes([0., 0., 1., 1.], frameon=False, xticks=[], yticks=[])  # Remove the white edge\n        melspectrogram = librosa.feature.melspectrogram(y=audio, sr=sample_rate)\n       \n        # height = S.shape[0]  \n        # image_cropped = S[int(height*0.2  ):int(height*0.8 ),:]\n        librosa.display.specshow(librosa.power_to_db(melspectrogram, ref=np.max))\n        plt.savefig(to_file, bbox_inches=None, pad_inches=0)\n        plt.close()","metadata":{"execution":{"iopub.status.busy":"2023-04-14T00:10:24.251424Z","iopub.execute_input":"2023-04-14T00:10:24.251796Z","iopub.status.idle":"2023-04-14T00:10:24.263827Z","shell.execute_reply.started":"2023-04-14T00:10:24.251764Z","shell.execute_reply":"2023-04-14T00:10:24.262513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataSet = TrainAudioDataset()\nbatch_size = 1\ndataLoader = DataLoader(dataSet, batch_size=batch_size, collate_fn=DataProcessing.my_collate)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T00:10:26.761754Z","iopub.execute_input":"2023-04-14T00:10:26.762421Z","iopub.status.idle":"2023-04-14T00:10:26.769336Z","shell.execute_reply.started":"2023-04-14T00:10:26.762380Z","shell.execute_reply":"2023-04-14T00:10:26.768371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfor (X, Y) in dataLoader:\n    for batch, x in tqdm(enumerate(X)):\n        fileName = Y[batch]\n        (frame, sample_rate, name) = x\n        path =  f'/test_melspectrogram/{fileName}/'\n        Path(path).mkdir(parents=True, exist_ok=True)\n        DataProcessing.melgram_v2(frame , sample_rate, f'{path}/{name}.png')\n            ","metadata":{"execution":{"iopub.status.busy":"2023-04-14T00:10:28.470677Z","iopub.execute_input":"2023-04-14T00:10:28.471053Z","iopub.status.idle":"2023-04-14T00:10:57.243589Z","shell.execute_reply.started":"2023-04-14T00:10:28.471017Z","shell.execute_reply":"2023-04-14T00:10:57.242791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = sorted(os.listdir('/kaggle/input/birdclef-2023/train_audio'))\n\nmodel_ft = models.resnet18(pretrained=False)\nnum_ftrs = model_ft.fc.in_features\nmodel_ft.fc = torch.nn.Linear(num_ftrs, len(class_names))\nmodel_ft = model_ft.to(device)\ncriterion = torch.nn.CrossEntropyLoss()\n\n# Observe that all parameters are being optimized\noptimizer_ft = torch.optim.SGD(model_ft.parameters(), lr=0.001, momentum=0.9)\n\ncheckpoint = torch.load('/kaggle/input/resnet18/model2023-04-028.pth', map_location=torch.device('cpu'))\nmodel_ft.load_state_dict(checkpoint['model_state_dict'])\noptimizer_ft.load_state_dict(checkpoint['optimizer_state_dict'])\nmodel_ft.eval()","metadata":{"execution":{"iopub.status.busy":"2023-04-14T00:12:32.641615Z","iopub.execute_input":"2023-04-14T00:12:32.642005Z","iopub.status.idle":"2023-04-14T00:12:32.891389Z","shell.execute_reply.started":"2023-04-14T00:12:32.641972Z","shell.execute_reply":"2023-04-14T00:12:32.890493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transform = transforms.Compose([ transforms.ToTensor() ])\ndataset = datasets.ImageFolder('/test_melspectrogram/', transform=transform  )\nloader = torch.utils.data.DataLoader(dataset, batch_size=5 )\n","metadata":{"execution":{"iopub.status.busy":"2023-04-14T00:12:07.450713Z","iopub.execute_input":"2023-04-14T00:12:07.451495Z","iopub.status.idle":"2023-04-14T00:12:07.459817Z","shell.execute_reply.started":"2023-04-14T00:12:07.451430Z","shell.execute_reply":"2023-04-14T00:12:07.457693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transform = transforms.Compose([ transforms.ToTensor() ])\ndataset = datasets.ImageFolder('/test_melspectrogram', transform=transform,  )\ntestloader = torch.utils.data.DataLoader(dataset, batch_size=64, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-14T00:13:32.447633Z","iopub.execute_input":"2023-04-14T00:13:32.448011Z","iopub.status.idle":"2023-04-14T00:13:32.455410Z","shell.execute_reply.started":"2023-04-14T00:13:32.447978Z","shell.execute_reply":"2023-04-14T00:13:32.454227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"/kaggle/input/birdclef-2023/sample_submission.csv\")\nsample_sub[class_names] = sample_sub[class_names].astype(np.float32)\nsample_sub.drop(sample_sub.index, inplace=True)\nsample_sub.head()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-04-14T00:13:34.351547Z","iopub.execute_input":"2023-04-14T00:13:34.352540Z","iopub.status.idle":"2023-04-14T00:13:34.427460Z","shell.execute_reply.started":"2023-04-14T00:13:34.352486Z","shell.execute_reply":"2023-04-14T00:13:34.426191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"indexToClass = {v: k for k, v in dataset.class_to_idx.items()}\ncolumns = []\n \nwith torch.no_grad():\n    for batch, (X, y) in tqdm(enumerate(testloader)):\n        pred = model_ft(X)\n        for index, predItem in enumerate(pred):\n            m = torch.nn.Softmax(dim=0)\n            output = m(predItem) \n            print(output.argmax(0))\n            fileName = dataset.imgs[batch * 64 + index][0].split('/')[-1].split('.')[0]\n             \n            columns = sample_sub.columns\n            \n            collection = np.append([fileName], output.detach().numpy())\n    \n            newRow = pd.DataFrame([collection], columns=sample_sub.columns)\n            pd.DataFrame(newRow)\n            sample_sub = pd.concat([sample_sub, newRow], ignore_index=True) ","metadata":{"execution":{"iopub.status.busy":"2023-04-14T00:13:36.445075Z","iopub.execute_input":"2023-04-14T00:13:36.445428Z","iopub.status.idle":"2023-04-14T00:13:43.285696Z","shell.execute_reply.started":"2023-04-14T00:13:36.445398Z","shell.execute_reply":"2023-04-14T00:13:43.284910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-09T19:35:13.866471Z","iopub.execute_input":"2023-04-09T19:35:13.866768Z","iopub.status.idle":"2023-04-09T19:35:13.895431Z","shell.execute_reply.started":"2023-04-09T19:35:13.866738Z","shell.execute_reply":"2023-04-09T19:35:13.893669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-04-09T19:42:45.758538Z","iopub.execute_input":"2023-04-09T19:42:45.759007Z","iopub.status.idle":"2023-04-09T19:42:45.770619Z","shell.execute_reply.started":"2023-04-09T19:42:45.758946Z","shell.execute_reply":"2023-04-09T19:42:45.769458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" ","metadata":{"execution":{"iopub.status.busy":"2023-04-09T19:44:56.350615Z","iopub.execute_input":"2023-04-09T19:44:56.350960Z","iopub.status.idle":"2023-04-09T19:44:56.361089Z","shell.execute_reply.started":"2023-04-09T19:44:56.350931Z","shell.execute_reply":"2023-04-09T19:44:56.358818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" ","metadata":{"execution":{"iopub.status.busy":"2023-04-09T19:48:39.626087Z","iopub.execute_input":"2023-04-09T19:48:39.626503Z","iopub.status.idle":"2023-04-09T19:48:39.636462Z","shell.execute_reply.started":"2023-04-09T19:48:39.626465Z","shell.execute_reply":"2023-04-09T19:48:39.634951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" ","metadata":{"execution":{"iopub.status.busy":"2023-04-09T19:48:42.142262Z","iopub.execute_input":"2023-04-09T19:48:42.142693Z","iopub.status.idle":"2023-04-09T19:48:43.162785Z","shell.execute_reply.started":"2023-04-09T19:48:42.142653Z","shell.execute_reply":"2023-04-09T19:48:43.160814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# os.remove(\"submission.csv\")\n\nsample_sub.to_csv(\"submission.csv\", index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-14T00:14:38.557088Z","iopub.execute_input":"2023-04-14T00:14:38.557501Z","iopub.status.idle":"2023-04-14T00:14:38.581751Z","shell.execute_reply.started":"2023-04-14T00:14:38.557456Z","shell.execute_reply":"2023-04-14T00:14:38.579933Z"},"trusted":true},"execution_count":null,"outputs":[]}]}