{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import cv2\nimport audioread\nimport logging\nimport os\nimport sys\nsys.path.append('../input/pytorch-image-models/pytorch-image-models-master')\nimport random\nimport time\nimport warnings\nimport glob\n\nimport librosa\nimport numpy as np\nimport pandas as pd\nimport soundfile as sf\nimport timm\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.utils.data as torchdata\n\nfrom contextlib import contextmanager\nfrom pathlib import Path\nfrom typing import List\nfrom typing import Optional\nfrom sklearn import metrics\n\nfrom tqdm import tqdm\n\nimport albumentations as A\nimport albumentations.pytorch.transforms as T\n\nimport concurrent.futures","metadata":{"execution":{"iopub.status.busy":"2023-04-13T22:53:49.429938Z","iopub.execute_input":"2023-04-13T22:53:49.430364Z","iopub.status.idle":"2023-04-13T22:53:49.43981Z","shell.execute_reply.started":"2023-04-13T22:53:49.430326Z","shell.execute_reply":"2023-04-13T22:53:49.437977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")","metadata":{"execution":{"iopub.status.busy":"2023-04-13T22:53:49.72137Z","iopub.execute_input":"2023-04-13T22:53:49.721827Z","iopub.status.idle":"2023-04-13T22:53:49.728267Z","shell.execute_reply.started":"2023-04-13T22:53:49.721791Z","shell.execute_reply":"2023-04-13T22:53:49.726678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def init_layer(layer):\n    nn.init.xavier_uniform_(layer.weight)\n\n    if hasattr(layer, \"bias\"):\n        if layer.bias is not None:\n            layer.bias.data.fill_(0.)\n\n\ndef init_bn(bn):\n    bn.bias.data.fill_(0.)\n    bn.weight.data.fill_(1.0)\n\n\ndef init_weights(model):\n    classname = model.__class__.__name__\n    if classname.find(\"Conv2d\") != -1:\n        nn.init.xavier_uniform_(model.weight, gain=np.sqrt(2))\n        model.bias.data.fill_(0)\n    elif classname.find(\"BatchNorm\") != -1:\n        model.weight.data.normal_(1.0, 0.02)\n        model.bias.data.fill_(0)\n    elif classname.find(\"GRU\") != -1:\n        for weight in model.parameters():\n            if len(weight.size()) > 1:\n                nn.init.orghogonal_(weight.data)\n    elif classname.find(\"Linear\") != -1:\n        model.weight.data.normal_(0, 0.01)\n        model.bias.data.zero_()\n\n\ndef interpolate(x: torch.Tensor, ratio: int):\n    \"\"\"Interpolate data in time domain. This is used to compensate the\n    resolution reduction in downsampling of a CNN.\n    Args:\n      x: (batch_size, time_steps, classes_num)\n      ratio: int, ratio to interpolate\n    Returns:\n      upsampled: (batch_size, time_steps * ratio, classes_num)\n    \"\"\"\n    (batch_size, time_steps, classes_num) = x.shape\n    upsampled = x[:, :, None, :].repeat(1, 1, ratio, 1)\n    upsampled = upsampled.reshape(batch_size, time_steps * ratio, classes_num)\n    return upsampled\n\n\ndef pad_framewise_output(framewise_output: torch.Tensor, frames_num: int):\n    \"\"\"Pad framewise_output to the same length as input frames. The pad value\n    is the same as the value of the last frame.\n    Args:\n      framewise_output: (batch_size, frames_num, classes_num)\n      frames_num: int, number of frames to pad\n    Outputs:\n      output: (batch_size, frames_num, classes_num)\n    \"\"\"\n    output = F.interpolate(\n        framewise_output.unsqueeze(1),\n        size=(frames_num, framewise_output.size(2)),\n        align_corners=True,\n        mode=\"bilinear\").squeeze(1)\n\n    return output\n\n\nclass AttBlockV2(nn.Module):\n    def __init__(self,\n                 in_features: int,\n                 out_features: int,\n                 activation=\"linear\"):\n        super().__init__()\n\n        self.activation = activation\n        self.att = nn.Conv1d(\n            in_channels=in_features,\n            out_channels=out_features,\n            kernel_size=1,\n            stride=1,\n            padding=0,\n            bias=True)\n        self.cla = nn.Conv1d(\n            in_channels=in_features,\n            out_channels=out_features,\n            kernel_size=1,\n            stride=1,\n            padding=0,\n            bias=True)\n\n        self.init_weights()\n\n    def init_weights(self):\n        init_layer(self.att)\n        init_layer(self.cla)\n\n    def forward(self, x):\n        # x: (n_samples, n_in, n_time)\n        norm_att = torch.softmax(torch.tanh(self.att(x)), dim=-1)\n        cla = self.nonlinear_transform(self.cla(x))\n        x = torch.sum(norm_att * cla, dim=2)\n        return x, norm_att, cla\n\n    def nonlinear_transform(self, x):\n        if self.activation == 'linear':\n            return x\n        elif self.activation == 'sigmoid':\n            return torch.sigmoid(x)\n\n\nclass TimmSED(nn.Module):\n    def __init__(\n        self, \n        base_model_name: str, \n        config=None,\n        pretrained=False, \n        num_classes=24, \n        in_channels=1\n    ):\n        super().__init__()\n        \n        self.config = config\n\n        self.bn0 = nn.BatchNorm2d(self.config.n_mels)\n\n        base_model = timm.create_model(\n            base_model_name, \n            pretrained=pretrained, \n            num_classes=0,\n            global_pool=\"\",\n            in_chans=in_channels,\n        )\n        \n        layers = list(base_model.children())[:-2]\n        self.encoder = nn.Sequential(*layers)\n\n        in_features = base_model.num_features\n\n        self.fc1 = nn.Linear(in_features, in_features, bias=True)\n        self.att_block = AttBlockV2(\n            in_features, num_classes, activation=\"sigmoid\")\n\n        self.init_weight()\n\n    def init_weight(self):\n        init_bn(self.bn0)\n        init_layer(self.fc1)\n        \n    def forward(self, input_data):\n        if self.config.in_channels == 3:\n            x = input_data\n        else:\n            x = input_data[:, [0], :, :] # (batch_size, 1, time_steps, mel_bins)\n\n        frames_num = x.shape[2]\n\n        x = x.transpose(1, 3)\n        x = self.bn0(x)\n        x = x.transpose(1, 3)\n\n\n        x = x.transpose(2, 3)\n\n        x = self.encoder(x)\n        \n        # Aggregate in frequency axis\n        x = torch.mean(x, dim=2)\n\n        x1 = F.max_pool1d(x, kernel_size=3, stride=1, padding=1)\n        x2 = F.avg_pool1d(x, kernel_size=3, stride=1, padding=1)\n        x = x1 + x2\n\n        x = x.transpose(1, 2)\n        x = F.relu_(self.fc1(x))\n        x = x.transpose(1, 2)\n\n        (clipwise_output, norm_att, segmentwise_output) = self.att_block(x)\n\n        output_dict = {\n            \"clipwise_output\": clipwise_output,\n        }\n\n        return output_dict","metadata":{"execution":{"iopub.status.busy":"2023-04-13T22:53:49.822836Z","iopub.execute_input":"2023-04-13T22:53:49.823867Z","iopub.status.idle":"2023-04-13T22:53:49.855651Z","shell.execute_reply.started":"2023-04-13T22:53:49.823811Z","shell.execute_reply":"2023-04-13T22:53:49.8543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean = (0.485) # R only for RGB\nstd = (0.229) # R only for RGB\n\nalbu_transforms = {\n    'train' : A.Compose([\n            A.Normalize(mean, std),\n    ]),\n    'valid' : A.Compose([\n            A.Normalize(mean, std),\n    ]),\n}\n\n\nmean2 = (0.485, 0.456, 0.406) # RGB\nstd2 = (0.229, 0.224, 0.225) # RGB\n\nalbu_transforms2 = {\n    'train' : A.Compose([\n            A.Normalize(mean2, std2),\n    ]),\n    'valid' : A.Compose([\n            A.Normalize(mean2, std2),\n    ]),\n}","metadata":{"execution":{"iopub.status.busy":"2023-04-13T22:53:50.0155Z","iopub.execute_input":"2023-04-13T22:53:50.016328Z","iopub.status.idle":"2023-04-13T22:53:50.024278Z","shell.execute_reply.started":"2023-04-13T22:53:50.016284Z","shell.execute_reply":"2023-04-13T22:53:50.023017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG_tf_efficientnet_b0:\n    batch_size=4\n    num_workers=4\n\n    n_mels = 256\n    fmin = 16\n    fmax = 16386\n    n_fft = 2048\n    hop_length = 512\n    sr = 32000\n\n    target_columns = \"abethr1 abhori1 abythr1 afbfly1 afdfly1 afecuc1 affeag1 afgfly1 afghor1 afmdov1 afpfly1 afpkin1 afpwag1 afrgos1 afrgrp1 afrjac1 afrthr1 amesun2 augbuz1 bagwea1 barswa bawhor2 bawman1 bcbeat1 beasun2 bkctch1 bkfruw1 blacra1 blacuc1 blakit1 blaplo1 blbpuf2 blcapa2 blfbus1 blhgon1 blhher1 blksaw1 blnmou1 blnwea1 bltapa1 bltbar1 bltori1 blwlap1 brcale1 brcsta1 brctch1 brcwea1 brican1 brobab1 broman1 brosun1 brrwhe3 brtcha1 brubru1 brwwar1 bswdov1 btweye2 bubwar2 butapa1 cabgre1 carcha1 carwoo1 categr ccbeat1 chespa1 chewea1 chibat1 chtapa3 chucis1 cibwar1 cohmar1 colsun2 combul2 combuz1 comsan crefra2 crheag1 crohor1 darbar1 darter3 didcuc1 dotbar1 dutdov1 easmog1 eaywag1 edcsun3 egygoo equaka1 eswdov1 eubeat1 fatrav1 fatwid1 fislov1 fotdro5 gabgos2 gargan gbesta1 gnbcam2 gnhsun1 gobbun1 gobsta5 gobwea1 golher1 grbcam1 grccra1 grecor greegr grewoo2 grwpyt1 gryapa1 grywrw1 gybfis1 gycwar3 gyhbus1 gyhkin1 gyhneg1 gyhspa1 gytbar1 hadibi1 hamerk1 hartur1 helgui hipbab1 hoopoe huncis1 hunsun2 joygre1 kerspa2 klacuc1 kvbsun1 laudov1 lawgol lesmaw1 lessts1 libeat1 litegr litswi1 litwea1 loceag1 lotcor1 lotlap1 luebus1 mabeat1 macshr1 malkin1 marsto1 marsun2 mcptit1 meypar1 moccha1 mouwag1 ndcsun2 nobfly1 norbro1 norcro1 norfis1 norpuf1 nubwoo1 pabspa1 palfly2 palpri1 piecro1 piekin1 pitwhy purgre2 pygbat1 quailf1 ratcis1 raybar1 rbsrob1 rebfir2 rebhor1 reboxp1 reccor reccuc1 reedov1 refbar2 refcro1 reftin1 refwar2 rehblu1 rehwea1 reisee2 rerswa1 rewsta1 rindov rocmar2 rostur1 ruegls1 rufcha2 sacibi2 sccsun2 scrcha1 scthon1 shesta1 sichor1 sincis1 slbgre1 slcbou1 sltnig1 sobfly1 somgre1 somtit4 soucit1 soufis1 spemou2 spepig1 spewea1 spfbar1 spfwea1 spmthr1 spwlap1 squher1 strher strsee1 stusta1 subbus1 supsta1 tacsun1 tafpri1 tamdov1 thrnig1 trobou1 varsun2 vibsta2 vilwea1 vimwea1 walsta1 wbgbir1 wbrcha2 wbswea1 wfbeat1 whbcan1 whbcou1 whbcro2 whbtit5 whbwea1 whbwhe3 whcpri2 whctur2 wheslf1 whhsaw1 whihel1 whrshr1 witswa1 wlwwar wookin1 woosan wtbeat1 yebapa1 yebbar1 yebduc1 yebere1 yebgre1 yebsto1 yeccan1 yefcan yelbis1 yenspu1 yertin1 yesbar1 yespet1 yetgre1 yewgre1\".split()\n\n    base_model_name = \"tf_efficientnet_b0_ns\"\n    pretrained = False\n    num_classes = 264\n    in_channels = 1\n    \n    ckpt_path = glob.glob(f\"/kaggle/input/tfeb0-ns-0411/*\")","metadata":{"execution":{"iopub.status.busy":"2023-04-13T22:53:50.441186Z","iopub.execute_input":"2023-04-13T22:53:50.442662Z","iopub.status.idle":"2023-04-13T22:53:50.459832Z","shell.execute_reply.started":"2023-04-13T22:53:50.442589Z","shell.execute_reply":"2023-04-13T22:53:50.457978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config_tf_efficientnet_b0 = CFG_tf_efficientnet_b0()","metadata":{"execution":{"iopub.status.busy":"2023-04-13T22:53:54.117633Z","iopub.execute_input":"2023-04-13T22:53:54.118096Z","iopub.status.idle":"2023-04-13T22:53:54.123601Z","shell.execute_reply.started":"2023-04-13T22:53:54.118046Z","shell.execute_reply":"2023-04-13T22:53:54.12245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### As shown, we have 5 folds models","metadata":{}},{"cell_type":"code","source":"print(f\"we have {len(config_tf_efficientnet_b0.ckpt_path)} folds models to ensemble\")","metadata":{"execution":{"iopub.status.busy":"2023-04-13T22:55:01.160613Z","iopub.execute_input":"2023-04-13T22:55:01.161054Z","iopub.status.idle":"2023-04-13T22:55:01.168828Z","shell.execute_reply.started":"2023-04-13T22:55:01.161017Z","shell.execute_reply":"2023-04-13T22:55:01.166826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config_enesemble = [\n    config_tf_efficientnet_b0,\n]","metadata":{"execution":{"iopub.status.busy":"2023-04-13T22:53:55.941318Z","iopub.execute_input":"2023-04-13T22:53:55.941786Z","iopub.status.idle":"2023-04-13T22:53:55.947091Z","shell.execute_reply.started":"2023-04-13T22:53:55.941743Z","shell.execute_reply":"2023-04-13T22:53:55.945898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def compute_melspec(y, params):\n    \"\"\"\n    Computes a mel-spectrogram and puts it at decibel scale\n    Arguments:\n        y {np array} -- signal\n        params {AudioParams} -- Parameters to use for the spectrogram. Expected to have the attributes sr, n_mels, f_min, f_max\n    Returns:\n        np array -- Mel-spectrogram\n    \"\"\"\n    melspec = librosa.feature.melspectrogram(\n        y=y, sr=params.sr, n_mels=params.n_mels, n_fft=params.n_fft, hop_length=params.hop_length, fmin=params.fmin, fmax=params.fmax,\n    )\n\n    return melspec\n\n\ndef mono_to_color(X, eps=1e-6, mean=None, std=None):\n    \"\"\"\n    Converts a one channel array in [0, 255]\n    Arguments:\n        X {numpy array [H x W]} -- 2D array to convert\n    Keyword Arguments:\n        eps {float} -- To avoid dividing by 0 (default: {1e-6})\n        mean {None or np array} -- Mean for normalization (default: {None})\n        std {None or np array} -- Std for normalization (default: {None})\n    Returns:\n        numpy array [1 x H x W] -- RGB numpy array\n    \"\"\"\n    # X = np.stack([X, X, X], axis=-1)\n    X = np.expand_dims(X, axis=-1)\n\n    # Standardize\n    mean = mean or X.mean()\n    std = std or X.std()\n    X = (X - mean) / (std + eps)\n\n    # Normalize to [0, 255]\n    _min, _max = X.min(), X.max()\n\n    if (_max - _min) > eps:\n        V = np.clip(X, _min, _max)\n        V = 255 * (V - _min) / (_max - _min)\n        V = V.astype(np.uint8)\n    else:\n        V = np.zeros_like(X, dtype=np.uint8)\n\n    return V\n\ndef mono_to_color3(X, eps=1e-6, mean=None, std=None):\n    \"\"\"\n    Converts a one channel array to a 3 channel one in [0, 255]\n    Arguments:\n        X {numpy array [H x W]} -- 2D array to convert\n    Keyword Arguments:\n        eps {float} -- To avoid dividing by 0 (default: {1e-6})\n        mean {None or np array} -- Mean for normalization (default: {None})\n        std {None or np array} -- Std for normalization (default: {None})\n    Returns:\n        numpy array [3 x H x W] -- RGB numpy array\n    \"\"\"\n    X = np.stack([X, X, X], axis=-1)\n\n    # Standardize\n    mean = mean or X.mean()\n    std = std or X.std()\n    X = (X - mean) / (std + eps)\n\n    # Normalize to [0, 255]\n    _min, _max = X.min(), X.max()\n\n    if (_max - _min) > eps:\n        V = np.clip(X, _min, _max)\n        V = 255 * (V - _min) / (_max - _min)\n        V = V.astype(np.uint8)\n    else:\n        V = np.zeros_like(X, dtype=np.uint8)\n\n    return V","metadata":{"execution":{"iopub.status.busy":"2023-04-13T22:55:11.377Z","iopub.execute_input":"2023-04-13T22:55:11.377497Z","iopub.status.idle":"2023-04-13T22:55:11.393892Z","shell.execute_reply.started":"2023-04-13T22:55:11.377454Z","shell.execute_reply":"2023-04-13T22:55:11.392019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TestDataset(torchdata.Dataset):\n    def __init__(self, \n                 df: pd.DataFrame, \n                 clip: np.ndarray,\n                 config=None,\n                ):\n        \n        self.df = df\n        self.clip = clip\n        self.config = config\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx: int):\n\n        sample = self.df.loc[idx, :]\n        row_id = sample.row_id\n\n        end_seconds = int(sample.seconds)\n        start_seconds = int(end_seconds - 5)\n        \n        y = self.clip[self.config.sr * start_seconds : self.config.sr * end_seconds].astype(np.float32)\n        \n        image = compute_melspec(y, self.config)\n        image = librosa.power_to_db(image.astype(np.float32), ref=np.max)\n        \n        if config.in_channels == 3:\n            image = mono_to_color3(image)\n            image = image.astype(np.uint8)\n            image = albu_transforms2['valid'](image=image)['image'].T\n        else:\n            image = mono_to_color(image)\n            image = image.astype(np.uint8)\n            image = albu_transforms['valid'](image=image)['image'].T\n            \n        return {\n            \"image\": image,\n            \"row_id\": row_id,\n        }","metadata":{"execution":{"iopub.status.busy":"2023-04-13T22:55:12.279443Z","iopub.execute_input":"2023-04-13T22:55:12.279899Z","iopub.status.idle":"2023-04-13T22:55:12.291627Z","shell.execute_reply.started":"2023-04-13T22:55:12.279858Z","shell.execute_reply":"2023-04-13T22:55:12.28988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models_ensemble = []\n\nfor config in config_enesemble:\n    \n    config_models = []\n    \n    for ckpt_path in config.ckpt_path:\n    \n        model = TimmSED(\n            base_model_name=config.base_model_name,\n            config=config,\n            pretrained=config.pretrained,\n            num_classes=config.num_classes,\n            in_channels=config.in_channels\n        )\n\n        model.load_state_dict(torch.load(ckpt_path, map_location=device))\n        model.eval()\n        \n        config_models.append(model)\n        \n    models_ensemble.append((config, config_models))","metadata":{"execution":{"iopub.status.busy":"2023-04-13T22:55:21.614998Z","iopub.execute_input":"2023-04-13T22:55:21.61569Z","iopub.status.idle":"2023-04-13T22:55:25.437994Z","shell.execute_reply.started":"2023-04-13T22:55:21.615639Z","shell.execute_reply":"2023-04-13T22:55:25.43666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Version2 is used for submission, so we don't need to multiply test audio path to 15 to simulate the hiddent test dataset size","metadata":{}},{"cell_type":"code","source":"all_audios = list(Path(\"../input/birdclef-2023/test_soundscapes/\").glob(\"*.ogg\"))#*15","metadata":{"execution":{"iopub.status.busy":"2023-04-13T22:57:02.053937Z","iopub.execute_input":"2023-04-13T22:57:02.054485Z","iopub.status.idle":"2023-04-13T22:57:02.066337Z","shell.execute_reply.started":"2023-04-13T22:57:02.054435Z","shell.execute_reply":"2023-04-13T22:57:02.064565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seconds = [i for i in range(5, 605, 5)]\n\ndef prediction_for_clip(\n    audio_path\n):\n    \n    device = torch.device(\"cpu\")\n    \n    # inference\n    prediction_dict = {}\n    \n    global models_ensemble\n\n    clip, _ = librosa.load(audio_path, sr=32000)\n    name_ = \"_\".join(audio_path.name.split(\".\")[:-1])\n    row_ids = [name_+f\"_{second}\" for second in seconds]\n\n    test_df = pd.DataFrame({\n        \"row_id\": row_ids,\n        \"seconds\": seconds\n    })\n    \n    for config_models in models_ensemble:\n        \n        config, models = config_models[0], config_models[1]\n        \n        dataset = TestDataset(\n            df=test_df, \n            clip=clip,\n            config=config,\n        )\n        \n        loader = torchdata.DataLoader(\n            dataset,\n            batch_size=config.batch_size, \n            num_workers=config.num_workers,\n            drop_last=False,\n            shuffle=False,\n            pin_memory=True\n        )\n        \n        for data in loader:\n            \n            row_ids = data['row_id']\n            \n            for row_id in row_ids:\n                if row_id not in prediction_dict:\n                    prediction_dict[str(row_id)] = []\n            \n            image = data['image']#.to(device)\n                \n            probas = []\n            \n            for model in models:\n\n                with torch.no_grad():\n                    output = model(image)\n#                     \n                for row_id_idx, row_id in enumerate(row_ids):\n                    prediction_dict[str(row_id)].append(output['clipwise_output'][[row_id_idx]].numpy().reshape(-1))\n                                                        \n    for row_id in list(prediction_dict.keys()):\n                \n        logits = np.array(prediction_dict[row_id]).mean(0)\n        prediction_dict[row_id] = {}\n        for label in range(len(config.target_columns)):\n            prediction_dict[row_id][config.target_columns[label]] = logits[label]\n\n    return prediction_dict","metadata":{"execution":{"iopub.status.busy":"2023-04-13T22:57:40.502738Z","iopub.execute_input":"2023-04-13T22:57:40.503234Z","iopub.status.idle":"2023-04-13T22:57:40.519921Z","shell.execute_reply.started":"2023-04-13T22:57:40.503185Z","shell.execute_reply":"2023-04-13T22:57:40.518599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_or_pad(y, length, sr, train=True, probs=None):\n    \"\"\"\n    Crops an array to a chosen length\n    Arguments:\n        y {1D np array} -- Array to crop\n        length {int} -- Length of the crop\n        sr {int} -- Sampling rate\n    Keyword Arguments:\n        train {bool} -- Whether we are at train time. If so, crop randomly, else return the beginning of y (default: {True})\n        probs {None or numpy array} -- Probabilities to use to chose where to crop (default: {None})\n    Returns:\n        1D np array -- Cropped array\n    \"\"\"\n    if len(y) <= length:\n        y = np.concatenate([y, np.zeros(length - len(y))])\n    else:\n        if not train:\n            start = 0\n        elif probs is None:\n            start = np.random.randint(len(y) - length)\n        else:\n            start = (\n                    np.random.choice(np.arange(len(probs)), p=probs) + np.random.random()\n            )\n            start = int(sr * (start))\n\n        y = y[start: start + length]\n\n    return y.astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T22:57:40.849632Z","iopub.execute_input":"2023-04-13T22:57:40.850088Z","iopub.status.idle":"2023-04-13T22:57:40.859835Z","shell.execute_reply.started":"2023-04-13T22:57:40.850047Z","shell.execute_reply":"2023-04-13T22:57:40.858593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Using for loop will result in submission timeout error, so just comment it.","metadata":{}},{"cell_type":"code","source":"# %%time\n# start = time.time()\n# dicts = []\n# for audio_path in all_audios:\n#     dicts.append(prediction_for_clip(audio_path))\n# print(f\"Regular for loop costs {time.time()-start} for processing 15 audios\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start = time.time()\nwith concurrent.futures.ThreadPoolExecutor(max_workers=4) as executor:\n    dicts = list(executor.map(prediction_for_clip, all_audios))\nprint(f\"With concurrent ThreadPoolExecutor, time cost reduced to {time.time()-start} for processing 15 audios\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_dicts = {}\nfor d in dicts:\n    prediction_dicts.update(d)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame.from_dict(prediction_dicts, \"index\").rename_axis(\"row_id\").reset_index()\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}