{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8084829,"sourceType":"datasetVersion","datasetId":4772325},{"sourceId":8138141,"sourceType":"datasetVersion","datasetId":4811146},{"sourceId":32381,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":27115}],"dockerImageVersionId":30684,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install /kaggle/input/fastxtend-0-1-7/fastxtend-0.1.7-py3-none-any.whl -qq\n!pip install /kaggle/input/fastxtend-0-1-7/colorednoise-2.2.0-py3-none-any.whl -qq\n!pip install /kaggle/input/fastxtend-0-1-7/primePy-1.3-py3-none-any.whl -qq","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-04-17T00:12:00.493176Z","iopub.execute_input":"2024-04-17T00:12:00.497913Z","iopub.status.idle":"2024-04-17T00:13:49.917184Z","shell.execute_reply.started":"2024-04-17T00:12:00.497767Z","shell.execute_reply":"2024-04-17T00:13:49.915400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from fastai.vision.all import *\nfrom fastxtend.audio.all import *\nfrom fastcore.parallel import *\nimport librosa\nimport ast\nimport shutil","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-04-17T00:13:49.920544Z","iopub.execute_input":"2024-04-17T00:13:49.921342Z","iopub.status.idle":"2024-04-17T00:14:28.008156Z","shell.execute_reply.started":"2024-04-17T00:13:49.921291Z","shell.execute_reply":"2024-04-17T00:14:28.006603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Background","metadata":{}},{"cell_type":"markdown","source":"In my [previous notebook](https://www.kaggle.com/code/vishalbakshi/birdclef24-fastxtend-starter-train/) I trained a quick-and-dirty ResNet18 model with a 67% accuracy in classifying bird species based on a small subset of 5-second audio spectrograms. In this notebook I'll complete my first submission, starting by looking at the sample submission file.","metadata":{}},{"cell_type":"code","source":"path = Path('/kaggle/input/birdclef-2024')\n\nfor el in path.ls():\n    print(el)","metadata":{"execution":{"iopub.status.busy":"2024-04-17T00:14:28.010773Z","iopub.execute_input":"2024-04-17T00:14:28.011710Z","iopub.status.idle":"2024-04-17T00:14:28.020671Z","shell.execute_reply.started":"2024-04-17T00:14:28.011660Z","shell.execute_reply":"2024-04-17T00:14:28.018602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss = pd.read_csv(path/'sample_submission.csv')\nss","metadata":{"execution":{"iopub.status.busy":"2024-04-17T00:14:28.025693Z","iopub.execute_input":"2024-04-17T00:14:28.026458Z","iopub.status.idle":"2024-04-17T00:14:28.117749Z","shell.execute_reply.started":"2024-04-17T00:14:28.026392Z","shell.execute_reply":"2024-04-17T00:14:28.116093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the competition page:\n\n> **test_soundscapes/**\n> \n> When you submit a notebook, the **test_soundscapes** directory will be populated with approximately 1,100 recordings to be used for scoring.  \n> They are 4 minutes long and in ogg audio format. The file names are randomized but have the general form of soundscape_xxxxxx.ogg.  \n> It should take your submission notebook approximately five minutes to load all of the test soundscapes.","metadata":{}},{"cell_type":"markdown","source":"I'll load in the test audio files (of which there are currently none):","metadata":{}},{"cell_type":"code","source":"(path/'test_soundscapes').ls()","metadata":{"execution":{"iopub.status.busy":"2024-04-16T14:31:46.360341Z","iopub.execute_input":"2024-04-16T14:31:46.360686Z","iopub.status.idle":"2024-04-16T14:31:46.370783Z","shell.execute_reply.started":"2024-04-16T14:31:46.360657Z","shell.execute_reply":"2024-04-16T14:31:46.369371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tst_files = get_files(path/'test_soundscapes', extensions=['.ogg'])","metadata":{"execution":{"iopub.status.busy":"2024-04-16T14:31:46.372235Z","iopub.execute_input":"2024-04-16T14:31:46.372695Z","iopub.status.idle":"2024-04-16T14:31:46.379351Z","shell.execute_reply.started":"2024-04-16T14:31:46.372655Z","shell.execute_reply":"2024-04-16T14:31:46.378406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(tst_files)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T14:31:46.380731Z","iopub.execute_input":"2024-04-16T14:31:46.381200Z","iopub.status.idle":"2024-04-16T14:31:46.397101Z","shell.execute_reply.started":"2024-04-16T14:31:46.381162Z","shell.execute_reply":"2024-04-16T14:31:46.396129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tst_files","metadata":{"execution":{"iopub.status.busy":"2024-04-16T14:31:46.400267Z","iopub.execute_input":"2024-04-16T14:31:46.400792Z","iopub.status.idle":"2024-04-16T14:31:46.409898Z","shell.execute_reply.started":"2024-04-16T14:31:46.400749Z","shell.execute_reply":"2024-04-16T14:31:46.408798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create `DataLoaders`","metadata":{}},{"cell_type":"markdown","source":"I'll run all of the code from my [Train] notebook necessary to create the `DataLoaders` used during training.","metadata":{}},{"cell_type":"code","source":"trn_path = Path('/kaggle/input/birdclef-2024-training-data-subset/notebooks/train_chunks')","metadata":{"execution":{"iopub.status.busy":"2024-04-17T00:14:28.119518Z","iopub.execute_input":"2024-04-17T00:14:28.119936Z","iopub.status.idle":"2024-04-17T00:14:28.127139Z","shell.execute_reply.started":"2024-04-17T00:14:28.119903Z","shell.execute_reply":"2024-04-17T00:14:28.125507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class PowerToDB(Transform):\n    order = 75\n    def __init__(self, sr=32000, n_fft=512): \n        self.sr = sr\n        self.n_fft=n_fft\n        self.mel = MelSpectrogram(sample_rate=self.sr, n_fft=self.n_fft)\n    def encodes(self, x:TensorAudio):\n        return TensorMelSpec.create(librosa.power_to_db(self.mel(x).cpu()), settings={})","metadata":{"execution":{"iopub.status.busy":"2024-04-17T00:14:28.129419Z","iopub.execute_input":"2024-04-17T00:14:28.130020Z","iopub.status.idle":"2024-04-17T00:14:28.141807Z","shell.execute_reply.started":"2024-04-17T00:14:28.129969Z","shell.execute_reply":"2024-04-17T00:14:28.140191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class LoadTensorAudio(Transform):\n    def __init__(self, sr=32000):\n        self.sr = sr\n    def encodes(self, x:Path):\n        x = torch.load(x)\n        return TensorAudio(x, sr=self.sr)","metadata":{"execution":{"iopub.status.busy":"2024-04-17T00:14:28.143489Z","iopub.execute_input":"2024-04-17T00:14:28.143973Z","iopub.status.idle":"2024-04-17T00:14:28.153745Z","shell.execute_reply.started":"2024-04-17T00:14:28.143914Z","shell.execute_reply":"2024-04-17T00:14:28.152643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"auds = DataBlock(blocks = (TransformBlock, CategoryBlock),  \n                 get_items=get_files,\n                 item_tfms = [LoadTensorAudio, PowerToDB(sr=32000)], \n                 splitter=RandomSplitter(valid_pct=0.2, seed=42),\n                 get_y = lambda o: o.parent.name)","metadata":{"execution":{"iopub.status.busy":"2024-04-17T00:14:28.156828Z","iopub.execute_input":"2024-04-17T00:14:28.157344Z","iopub.status.idle":"2024-04-17T00:14:28.288791Z","shell.execute_reply.started":"2024-04-17T00:14:28.157301Z","shell.execute_reply":"2024-04-17T00:14:28.287276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls = auds.dataloaders(Path(trn_path), bs=64)\ndls.show_batch(figsize=(10, 5))","metadata":{"execution":{"iopub.status.busy":"2024-04-17T00:14:28.290909Z","iopub.execute_input":"2024-04-17T00:14:28.291651Z","iopub.status.idle":"2024-04-17T00:14:39.396480Z","shell.execute_reply.started":"2024-04-17T00:14:28.291603Z","shell.execute_reply":"2024-04-17T00:14:39.394679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(dls.vocab)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T16:34:34.877424Z","iopub.execute_input":"2024-04-16T16:34:34.878183Z","iopub.status.idle":"2024-04-16T16:34:34.886075Z","shell.execute_reply.started":"2024-04-16T16:34:34.878144Z","shell.execute_reply":"2024-04-16T16:34:34.884936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.vocab","metadata":{"execution":{"iopub.status.busy":"2024-04-16T16:34:35.735784Z","iopub.execute_input":"2024-04-16T16:34:35.736214Z","iopub.status.idle":"2024-04-16T16:34:35.744279Z","shell.execute_reply.started":"2024-04-16T16:34:35.736168Z","shell.execute_reply":"2024-04-16T16:34:35.742813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create `learner`","metadata":{}},{"cell_type":"code","source":"learn = vision_learner(\n    dls, \n    resnet18,\n    n_in=1,\n    loss_func=CrossEntropyLossFlat(),\n    metrics=[accuracy],\n    pretrained=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-17T00:29:50.011296Z","iopub.execute_input":"2024-04-17T00:29:50.012596Z","iopub.status.idle":"2024-04-17T00:29:50.407723Z","shell.execute_reply.started":"2024-04-17T00:29:50.012541Z","shell.execute_reply":"2024-04-17T00:29:50.406782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.model_dir = '/kaggle/working/'\nlearn.load('/kaggle/input/birdclef-2024-resnet18/pytorch/2/1/birdclef24_resnet18_v2')","metadata":{"execution":{"iopub.status.busy":"2024-04-17T00:29:51.578412Z","iopub.execute_input":"2024-04-17T00:29:51.579241Z","iopub.status.idle":"2024-04-17T00:29:52.239029Z","shell.execute_reply.started":"2024-04-17T00:29:51.579189Z","shell.execute_reply":"2024-04-17T00:29:52.238103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission File Overview","metadata":{}},{"cell_type":"markdown","source":"With the `DataLoaders` created, I can now create test `DataLoaders` from the `tst_files` and then use the model for inference.\n\nI found [this notebook](https://www.kaggle.com/code/aikhmelnytskyy/birdclef24-pretraining-is-all-you-need-infer#Submission-Time-%E2%8F%B0) which uses the **unlabeled_soundscapes/** directory as the test data.\n\nAlso big thanks to [this notebook by Myso](https://www.kaggle.com/code/myso1987/how-to-submit-to-birdclef-2024/notebook) which walks through how to submit to this competition. The key thing that I overlooked was the following parts of the submission description (emphasis mine):\n\n> For each row_id, you should predict the probability that a given bird species was present. There is one column per bird species, so you will need to provide 182 predictions per row. **Each row covers a five-second window of audio.**\n\nSo I'm actually supposed to split the 4-minute audio in to 5-second chunks and _then_ calculated predictions on those 5-second chunks. The submission `row_id` description makes more sense now:\n\n> row_id: A slug of `soundscape_[soundscape_id]_[end_time]` for the prediction.","metadata":{}},{"cell_type":"markdown","source":"## Splitting Audio into 5-Second Chunks","metadata":{}},{"cell_type":"markdown","source":"There are 240 seconds in 4 minutes, which equals 48 five-second chunks. I need to split each audio file into 48 chunks, annotate the file name according to the provided format, and then calculate predictions. I'll create a temporary directory which will hold the 5-second-long chunks with the appropriate filenames.","metadata":{}},{"cell_type":"markdown","source":"I'll start by doing this for one audio file.","metadata":{}},{"cell_type":"code","source":"unlabeled_test_files = (path/'unlabeled_soundscapes').ls()[:5]","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:51:54.773597Z","iopub.execute_input":"2024-04-16T18:51:54.775050Z","iopub.status.idle":"2024-04-16T18:51:54.938368Z","shell.execute_reply.started":"2024-04-16T18:51:54.775009Z","shell.execute_reply":"2024-04-16T18:51:54.936926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fname = unlabeled_test_files[0]\nfname","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:51:56.750812Z","iopub.execute_input":"2024-04-16T18:51:56.751612Z","iopub.status.idle":"2024-04-16T18:51:56.759375Z","shell.execute_reply.started":"2024-04-16T18:51:56.751575Z","shell.execute_reply":"2024-04-16T18:51:56.758082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ta = TensorAudio.create(fname)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:51:57.584698Z","iopub.execute_input":"2024-04-16T18:51:57.585142Z","iopub.status.idle":"2024-04-16T18:51:57.976695Z","shell.execute_reply.started":"2024-04-16T18:51:57.585112Z","shell.execute_reply":"2024-04-16T18:51:57.975515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The `TensorAudio` object I believe is ultimately a tensor and so I can perform tensor operations on it:","metadata":{}},{"cell_type":"code","source":"ta[0].shape, ta[0].sr, ta[0].duration","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:51:58.597718Z","iopub.execute_input":"2024-04-16T18:51:58.598135Z","iopub.status.idle":"2024-04-16T18:51:58.609544Z","shell.execute_reply.started":"2024-04-16T18:51:58.598107Z","shell.execute_reply":"2024-04-16T18:51:58.608190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"5-seconds at 32000 Hz is 160k samples:","metadata":{}},{"cell_type":"code","source":"5 * 32000","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:51:59.816206Z","iopub.execute_input":"2024-04-16T18:51:59.816679Z","iopub.status.idle":"2024-04-16T18:51:59.825059Z","shell.execute_reply.started":"2024-04-16T18:51:59.816646Z","shell.execute_reply":"2024-04-16T18:51:59.823512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Splitting a 32 kHz audio file into 5-second chunks is the same as splitting it into chunks of 160k samples.","metadata":{}},{"cell_type":"code","source":"ta[0][:160000], ta[0][:160000].shape","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:52:00.628600Z","iopub.execute_input":"2024-04-16T18:52:00.628991Z","iopub.status.idle":"2024-04-16T18:52:00.691951Z","shell.execute_reply.started":"2024-04-16T18:52:00.628963Z","shell.execute_reply":"2024-04-16T18:52:00.690718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'll do one audio file, note that in order to maintain the shape of the `TensorAudio`, I'll follow what is done in fastxtend in [`RandomCropPad`](https://github.com/warner-benjamin/fastxtend/blob/ff594c81e16b1ca062744ab46bd519968d6ef4dc/fastxtend/audio/augment.py#L96):\n\n```python\nif crop_start is not None:\n        x = x[...,crop_start:crop_start+end_len]\n```","metadata":{}},{"cell_type":"code","source":"chunks = [ta[...,(i * 160000):(i + 1) * 160000] for i in range(48)]","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:52:01.495935Z","iopub.execute_input":"2024-04-16T18:52:01.496998Z","iopub.status.idle":"2024-04-16T18:52:01.505707Z","shell.execute_reply.started":"2024-04-16T18:52:01.496965Z","shell.execute_reply":"2024-04-16T18:52:01.504241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(chunks) # should be 48","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:52:01.932133Z","iopub.execute_input":"2024-04-16T18:52:01.932543Z","iopub.status.idle":"2024-04-16T18:52:01.940568Z","shell.execute_reply.started":"2024-04-16T18:52:01.932514Z","shell.execute_reply":"2024-04-16T18:52:01.939330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chunks[0].shape","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:52:02.397643Z","iopub.execute_input":"2024-04-16T18:52:02.398023Z","iopub.status.idle":"2024-04-16T18:52:02.406439Z","shell.execute_reply.started":"2024-04-16T18:52:02.397997Z","shell.execute_reply":"2024-04-16T18:52:02.405305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The math looks good---I'll check it visually to see if it looks okay:","metadata":{}},{"cell_type":"code","source":"ta.show();","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:52:03.300336Z","iopub.execute_input":"2024-04-16T18:52:03.300721Z","iopub.status.idle":"2024-04-16T18:52:11.605796Z","shell.execute_reply.started":"2024-04-16T18:52:03.300696Z","shell.execute_reply":"2024-04-16T18:52:11.604242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chunks[0].show();","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:52:11.607966Z","iopub.execute_input":"2024-04-16T18:52:11.608379Z","iopub.status.idle":"2024-04-16T18:52:12.454047Z","shell.execute_reply.started":"2024-04-16T18:52:11.608349Z","shell.execute_reply":"2024-04-16T18:52:12.452916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chunks[0].sr, chunks[0].duration","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:52:12.455570Z","iopub.execute_input":"2024-04-16T18:52:12.456055Z","iopub.status.idle":"2024-04-16T18:52:12.464192Z","shell.execute_reply.started":"2024-04-16T18:52:12.456026Z","shell.execute_reply":"2024-04-16T18:52:12.462688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chunks[-1].show();","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:52:12.466545Z","iopub.execute_input":"2024-04-16T18:52:12.466891Z","iopub.status.idle":"2024-04-16T18:52:13.253052Z","shell.execute_reply.started":"2024-04-16T18:52:12.466866Z","shell.execute_reply":"2024-04-16T18:52:13.251790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chunks[-1].sr, chunks[1].duration","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:52:13.255424Z","iopub.execute_input":"2024-04-16T18:52:13.256117Z","iopub.status.idle":"2024-04-16T18:52:13.264768Z","shell.execute_reply.started":"2024-04-16T18:52:13.256053Z","shell.execute_reply":"2024-04-16T18:52:13.263411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Tough to tell if it exactly looks the same but it does look similar and does sound the same as the 4-minute long file. Although the 5-second chunks are playing much louder (the y-axis values around 0.01 look the same as the 4-minute audio plot).","metadata":{}},{"cell_type":"markdown","source":"## Storing 5-Second Audio Segments as Tensors","metadata":{}},{"cell_type":"markdown","source":"Now that I have chunks, I'll store them as PyTorch tensors (which is how I stored the training chunks).","metadata":{}},{"cell_type":"markdown","source":"I'll create a temporary directory to hold these 5-second tensors","metadata":{}},{"cell_type":"code","source":"test_chunks = '/kaggle/temp/test_chunks'\nos.makedirs(test_chunks, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:52:15.959988Z","iopub.execute_input":"2024-04-16T18:52:15.960431Z","iopub.status.idle":"2024-04-16T18:52:15.967383Z","shell.execute_reply.started":"2024-04-16T18:52:15.960402Z","shell.execute_reply":"2024-04-16T18:52:15.965289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'll then run through a loop and save each of the 48 chunks as a separate `.pt` file in my `test_chunks` directory:","metadata":{}},{"cell_type":"code","source":"for i in range(48):\n    # create chunk tensor file name stem\n    end_time = (i + 1) * 5\n    fn_stem = f\"{fname.stem}_{end_time}\" \n        \n    # create full file path for chunk tensor\n    fn = Path(test_chunks)/(fn_stem + '.pt')\n\n    # load a chunk of the TensorAudio\n    ta = chunks[i]\n\n    # convert it to regular tenros\n    ta = ta.clone().detach()\n\n    # save it\n    torch.save(ta, fn)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:52:20.184524Z","iopub.execute_input":"2024-04-16T18:52:20.184969Z","iopub.status.idle":"2024-04-16T18:52:20.323864Z","shell.execute_reply.started":"2024-04-16T18:52:20.184936Z","shell.execute_reply":"2024-04-16T18:52:20.322514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking the directory:","metadata":{}},{"cell_type":"code","source":"Path(test_chunks).ls()","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:52:21.344956Z","iopub.execute_input":"2024-04-16T18:52:21.345465Z","iopub.status.idle":"2024-04-16T18:52:21.355306Z","shell.execute_reply.started":"2024-04-16T18:52:21.345423Z","shell.execute_reply":"2024-04-16T18:52:21.353399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking the first and the last chunk:","metadata":{}},{"cell_type":"code","source":"TensorAudio(torch.load(Path(test_chunks)/'1872382287_5.pt'), sr=32000).show();","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:52:24.339042Z","iopub.execute_input":"2024-04-16T18:52:24.339552Z","iopub.status.idle":"2024-04-16T18:52:25.182068Z","shell.execute_reply.started":"2024-04-16T18:52:24.339504Z","shell.execute_reply":"2024-04-16T18:52:25.180922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TensorAudio(torch.load(Path(test_chunks)/'1872382287_240.pt'), sr=32000).show();","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:52:25.184068Z","iopub.execute_input":"2024-04-16T18:52:25.184456Z","iopub.status.idle":"2024-04-16T18:52:25.969786Z","shell.execute_reply.started":"2024-04-16T18:52:25.184426Z","shell.execute_reply":"2024-04-16T18:52:25.968691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"That looks and sounds good!","metadata":{}},{"cell_type":"markdown","source":"## Calculating Predictions","metadata":{}},{"cell_type":"markdown","source":"With my `Learner` and test data prepared, I can now calculate predictions.","metadata":{}},{"cell_type":"code","source":"# get test files\ntst_files = get_files(Path(test_chunks))","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:52:30.292385Z","iopub.execute_input":"2024-04-16T18:52:30.293542Z","iopub.status.idle":"2024-04-16T18:52:30.301920Z","shell.execute_reply.started":"2024-04-16T18:52:30.293492Z","shell.execute_reply":"2024-04-16T18:52:30.300206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(tst_files)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:52:39.214576Z","iopub.execute_input":"2024-04-16T18:52:39.215257Z","iopub.status.idle":"2024-04-16T18:52:39.222135Z","shell.execute_reply.started":"2024-04-16T18:52:39.215224Z","shell.execute_reply":"2024-04-16T18:52:39.220986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create test DataLoaders\ntst_dl = dls.test_dl(tst_files)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:56:25.463593Z","iopub.execute_input":"2024-04-16T18:56:25.464016Z","iopub.status.idle":"2024-04-16T18:56:25.471693Z","shell.execute_reply.started":"2024-04-16T18:56:25.463988Z","shell.execute_reply":"2024-04-16T18:56:25.470555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculate probabilities\nprobs,_= learn.get_preds(dl=tst_dl)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:57:27.878850Z","iopub.execute_input":"2024-04-16T18:57:27.879352Z","iopub.status.idle":"2024-04-16T18:57:32.631018Z","shell.execute_reply.started":"2024-04-16T18:57:27.879318Z","shell.execute_reply":"2024-04-16T18:57:32.629606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looking at the shape of `probs`---I have 182 predictions for 48 items.","metadata":{}},{"cell_type":"code","source":"probs.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:57:34.180652Z","iopub.execute_input":"2024-04-16T18:57:34.182081Z","iopub.status.idle":"2024-04-16T18:57:34.190817Z","shell.execute_reply.started":"2024-04-16T18:57:34.182041Z","shell.execute_reply":"2024-04-16T18:57:34.189534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next, I'll prepare the submission CSV, starting with creating a `DataFrame` of probabilities with `dls.vocab` as the columns:","metadata":{}},{"cell_type":"code","source":"probs_df = pd.DataFrame(probs, columns=dls.vocab)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:57:39.484866Z","iopub.execute_input":"2024-04-16T18:57:39.485316Z","iopub.status.idle":"2024-04-16T18:57:39.492898Z","shell.execute_reply.started":"2024-04-16T18:57:39.485285Z","shell.execute_reply":"2024-04-16T18:57:39.491517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Then, I'll create `row_ids` which is the stem of the filepath in `tst_files`","metadata":{}},{"cell_type":"code","source":"tst_files[0].stem","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:57:40.533134Z","iopub.execute_input":"2024-04-16T18:57:40.533696Z","iopub.status.idle":"2024-04-16T18:57:40.542780Z","shell.execute_reply.started":"2024-04-16T18:57:40.533658Z","shell.execute_reply":"2024-04-16T18:57:40.541174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"row_ids = pd.Series([o.stem for o in tst_files])\nprobs_df['row_id'] = row_ids","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:57:45.231948Z","iopub.execute_input":"2024-04-16T18:57:45.233363Z","iopub.status.idle":"2024-04-16T18:57:45.243938Z","shell.execute_reply.started":"2024-04-16T18:57:45.233323Z","shell.execute_reply":"2024-04-16T18:57:45.242268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probs_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:57:45.918571Z","iopub.execute_input":"2024-04-16T18:57:45.919024Z","iopub.status.idle":"2024-04-16T18:57:45.949903Z","shell.execute_reply.started":"2024-04-16T18:57:45.918992Z","shell.execute_reply":"2024-04-16T18:57:45.948312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next, I'll re-order the columns so that `row_id` is first:","metadata":{}},{"cell_type":"code","source":"cols = ['row_id'] + list(dls.vocab)\nprobs_df = probs_df[cols]\nprobs_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:57:50.718143Z","iopub.execute_input":"2024-04-16T18:57:50.718587Z","iopub.status.idle":"2024-04-16T18:57:50.752275Z","shell.execute_reply.started":"2024-04-16T18:57:50.718557Z","shell.execute_reply":"2024-04-16T18:57:50.750888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks good! I'll now package the code for creating 5-second chunks from the 4-minute test data into a function, starting by emptying the `test_chunks` directory and creating a blank one.","metadata":{}},{"cell_type":"code","source":"shutil.rmtree('/kaggle/temp/test_chunks')","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:58:12.231205Z","iopub.execute_input":"2024-04-16T18:58:12.232596Z","iopub.status.idle":"2024-04-16T18:58:12.248600Z","shell.execute_reply.started":"2024-04-16T18:58:12.232545Z","shell.execute_reply":"2024-04-16T18:58:12.247533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_chunks = '/kaggle/temp/test_chunks'\nos.makedirs(test_chunks, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:58:12.874347Z","iopub.execute_input":"2024-04-16T18:58:12.874766Z","iopub.status.idle":"2024-04-16T18:58:12.881540Z","shell.execute_reply.started":"2024-04-16T18:58:12.874736Z","shell.execute_reply":"2024-04-16T18:58:12.879630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Path('/kaggle/temp/test_chunks').ls()","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:58:13.582716Z","iopub.execute_input":"2024-04-16T18:58:13.583457Z","iopub.status.idle":"2024-04-16T18:58:13.593736Z","shell.execute_reply.started":"2024-04-16T18:58:13.583423Z","shell.execute_reply":"2024-04-16T18:58:13.591865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_test_chunks(full_tst_files, test_chunks):\n    for fname in full_tst_files:\n        # prepare 48 five-second chunks\n        ta = TensorAudio.create(fname)\n        chunks = [ta[...,(i * 160000):(i + 1) * 160000] for i in range(48)]\n        \n        for i in range(48):\n            # create chunk tensor file name stem\n            end_time = (i + 1) * 5\n            fn_stem = f\"{fname.stem}_{end_time}\" \n\n            # create full file path for chunk tensor\n            fn = Path(test_chunks)/(fn_stem + '.pt')\n\n            # load a chunk of the TensorAudio\n            ta = chunks[i]\n\n            # convert it to regular tensor\n            ta = ta.clone().detach()\n\n            # save it\n            torch.save(ta, fn)","metadata":{"execution":{"iopub.status.busy":"2024-04-17T00:14:39.400392Z","iopub.execute_input":"2024-04-17T00:14:39.400816Z","iopub.status.idle":"2024-04-17T00:14:39.411419Z","shell.execute_reply.started":"2024-04-17T00:14:39.400782Z","shell.execute_reply":"2024-04-17T00:14:39.409497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Then I'll create a function to calculate and submit predictions.","metadata":{}},{"cell_type":"code","source":"def submit(tst_files):\n    # create test DataLoaders\n    tst_dl = dls.test_dl(tst_files)\n    \n    # calculate predictions\n    probs,_= learn.get_preds(dl=tst_dl)\n    \n    # create DataFrame with probabilities\n    probs_df = pd.DataFrame(probs, columns=dls.vocab)\n    \n    # add row_ids\n    row_ids = pd.Series([o.stem for o in tst_files])\n    probs_df['row_id'] = row_ids\n    \n    # re-order columns so row_id is the first columns\n    cols = ['row_id'] + list(dls.vocab)\n    probs_df = probs_df[cols]\n\n    # export to CSV\n    probs_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-17T00:14:39.413482Z","iopub.execute_input":"2024-04-17T00:14:39.414201Z","iopub.status.idle":"2024-04-17T00:14:39.454189Z","shell.execute_reply.started":"2024-04-17T00:14:39.414150Z","shell.execute_reply":"2024-04-17T00:14:39.452598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Finally, I'll have a code block where if it's running in submission, it runs the code above, otherwise it creates a `submission.csv` file with the correct header.","metadata":{}},{"cell_type":"code","source":"# get full tst_files\nfull_tst_files = get_files(path/'test_soundscapes', extensions=['.ogg'])\n    \nif len(full_tst_files) > 0:\n    # create fresh temp directory\n    shutil.rmtree('/kaggle/temp/test_chunks')\n    test_chunks = '/kaggle/temp/test_chunks'\n    os.makedirs(test_chunks, exist_ok=True)\n\n    # create 5-second chunks\n    create_test_chunks(full_tst_files, test_chunks)\n    \n    # get 5-second chunk test files\n    tst_files = get_files(Path(test_chunks))\n    \n    # submit predictions on 5-second chunks\n    submit(tst_files)\n    \nelse:\n    # create empty submission.csv\n    cols = ['row_id'] + list(dls.vocab)\n    empty_probs = pd.DataFrame(columns=cols)\n    empty_probs.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T19:15:04.291437Z","iopub.execute_input":"2024-04-16T19:15:04.291902Z","iopub.status.idle":"2024-04-16T19:15:04.323154Z","shell.execute_reply.started":"2024-04-16T19:15:04.291871Z","shell.execute_reply":"2024-04-16T19:15:04.322080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2024-04-16T19:15:09.036855Z","iopub.execute_input":"2024-04-16T19:15:09.037324Z","iopub.status.idle":"2024-04-16T19:15:10.184983Z","shell.execute_reply.started":"2024-04-16T19:15:09.037292Z","shell.execute_reply":"2024-04-16T19:15:10.183723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Checks","metadata":{}},{"cell_type":"markdown","source":"Just to be sure---I'll run my `create_test_chunks` and `submit` functions on about 100 unlabeled soundscapes files to make sure the outputs are as expected. Some of the unlabeled soundscape files are shorter than 4 minutes so I'll filter them out.","metadata":{}},{"cell_type":"code","source":"full_tst_files = get_files(path/'test_soundscapes', extensions=['.ogg'])\nunlabeled_tst_files = get_files(path/'unlabeled_soundscapes')[:150]","metadata":{"execution":{"iopub.status.busy":"2024-04-17T01:52:09.052776Z","iopub.execute_input":"2024-04-17T01:52:09.053625Z","iopub.status.idle":"2024-04-17T01:52:13.531482Z","shell.execute_reply.started":"2024-04-17T01:52:09.053584Z","shell.execute_reply":"2024-04-17T01:52:13.530157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# filter out audio files less than 4 minutes long as actual test files are 4 minutes long\nunlabeled_tst_files = [f for f in unlabeled_tst_files if TensorAudio.create(f).shape[-1] >= 7680000]","metadata":{"execution":{"iopub.status.busy":"2024-04-17T01:52:57.611032Z","iopub.execute_input":"2024-04-17T01:52:57.611528Z","iopub.status.idle":"2024-04-17T01:53:31.401445Z","shell.execute_reply.started":"2024-04-17T01:52:57.611494Z","shell.execute_reply":"2024-04-17T01:53:31.400158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(unlabeled_tst_files)","metadata":{"execution":{"iopub.status.busy":"2024-04-17T01:54:09.542329Z","iopub.execute_input":"2024-04-17T01:54:09.542837Z","iopub.status.idle":"2024-04-17T01:54:09.551762Z","shell.execute_reply.started":"2024-04-17T01:54:09.542804Z","shell.execute_reply":"2024-04-17T01:54:09.550105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if len(full_tst_files) == 0: # only run this outside of a submission\n    # create fresh temp directory\n    shutil.rmtree('/kaggle/temp/test_chunks')\n    test_chunks = '/kaggle/temp/test_chunks'\n    os.makedirs(test_chunks, exist_ok=True)\n\n    # create 5-second chunks\n    create_test_chunks(unlabeled_tst_files, test_chunks)\n\n    # get 5-second chunk test files\n    tst_files = get_files(Path(test_chunks))\n\n    # submit predictions on 5-second chunks\n    submit(tst_files)","metadata":{"execution":{"iopub.status.busy":"2024-04-17T01:54:16.409108Z","iopub.execute_input":"2024-04-17T01:54:16.409597Z","iopub.status.idle":"2024-04-17T02:07:24.009959Z","shell.execute_reply.started":"2024-04-17T01:54:16.409554Z","shell.execute_reply":"2024-04-17T02:07:24.008038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check that the length of test_chunks is correct, 48 five-second chunks per file\nlen(tst_files) == 48 * len(unlabeled_tst_files)","metadata":{"execution":{"iopub.status.busy":"2024-04-17T02:08:28.571367Z","iopub.execute_input":"2024-04-17T02:08:28.571957Z","iopub.status.idle":"2024-04-17T02:08:28.582661Z","shell.execute_reply.started":"2024-04-17T02:08:28.571914Z","shell.execute_reply":"2024-04-17T02:08:28.580936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(tst_files), len(unlabeled_tst_files)","metadata":{"execution":{"iopub.status.busy":"2024-04-17T02:08:30.894766Z","iopub.execute_input":"2024-04-17T02:08:30.895314Z","iopub.status.idle":"2024-04-17T02:08:30.904984Z","shell.execute_reply.started":"2024-04-17T02:08:30.895276Z","shell.execute_reply":"2024-04-17T02:08:30.903435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check the output\nsubm_df = pd.read_csv(\"submission.csv\")\nsubm_df.shape # items x classes ","metadata":{"execution":{"iopub.status.busy":"2024-04-17T02:08:54.640833Z","iopub.execute_input":"2024-04-17T02:08:54.641327Z","iopub.status.idle":"2024-04-17T02:08:55.045420Z","shell.execute_reply.started":"2024-04-17T02:08:54.641293Z","shell.execute_reply":"2024-04-17T02:08:55.044285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subm_df[list(dls.vocab)].sum(axis=1).sum() # close enough for floating point math","metadata":{"execution":{"iopub.status.busy":"2024-04-17T02:10:39.183622Z","iopub.execute_input":"2024-04-17T02:10:39.184930Z","iopub.status.idle":"2024-04-17T02:10:39.204603Z","shell.execute_reply.started":"2024-04-17T02:10:39.184893Z","shell.execute_reply":"2024-04-17T02:10:39.203048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks like my functions are working as expected.","metadata":{}}]}