{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Baseline for Pytorch Lightning based submission \n\n**Step 1: For generating spectrograms :** https://www.kaggle.com/code/nischaydnk/split-creating-melspecs-stage-1\n\n**Step 2: Training Notebook with Pytorch Lightning:** https://www.kaggle.com/code/nischaydnk/birdclef-2023-pytorch-lightning-training-w-cmap\n\nFeel free to reach out in comments incase you find bugs or have doubts!!","metadata":{}},{"cell_type":"code","source":"!export OMP_NUM_THREADS=N\n\n!export OMP_SCHEDULE=STATIC\n!export OMP_PROC_BIND=CLOSE\n!export GOMP_CPU_AFFINITY=\"N-M\"","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:17:02.685080Z","iopub.execute_input":"2023-06-02T05:17:02.685519Z","iopub.status.idle":"2023-06-02T05:17:06.779148Z","shell.execute_reply.started":"2023-06-02T05:17:02.685485Z","shell.execute_reply":"2023-06-02T05:17:06.777633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/input/openvino-wheels/openvino-2022.3.0-9052-cp37-cp37m-manylinux_2_17_x86_64.whl --no-index --find-links /kaggle/input/openvino-wheels","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:17:06.781637Z","iopub.execute_input":"2023-06-02T05:17:06.782105Z","iopub.status.idle":"2023-06-02T05:17:21.812180Z","shell.execute_reply.started":"2023-06-02T05:17:06.782051Z","shell.execute_reply":"2023-06-02T05:17:21.810380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport warnings\nimport joblib\nimport torch","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:17:21.814357Z","iopub.execute_input":"2023-06-02T05:17:21.814924Z","iopub.status.idle":"2023-06-02T05:17:23.036344Z","shell.execute_reply.started":"2023-06-02T05:17:21.814864Z","shell.execute_reply":"2023-06-02T05:17:23.035097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    num_classes = 264\n \n    DEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')    \n\n    data_root = \"/kaggle/input/birdclef-2023/\"\n    train_path = \"/kaggle/input/bc2023-train-val-df/train.csv\"\n    valid_path = \"/kaggle/input/bc2023-train-val-df/valid.csv\"\n    \n    train_path = \"/kaggle/input/bc2023-train-val-df/train.csv\"\n    valid_path = \"/kaggle/input/bc2023-train-val-df/valid.csv\"\n    test_path = '/kaggle/input/birdclef-2023/test_soundscapes/'\n\n    SR = 32000\n    DURATION = 5\n\n    \n    infer_duration=5\n    \n    train_duration=10\n    \n    # Sed model\n    model_ckpt = [\n        '/kaggle/input/birdclef-openvino-comp/sed_v2s_final_30s_finetune/sed3_120.xml', #v2s\n        '/kaggle/input/birdclef-openvino-comp/sed_se_half_ce/sed_se_120.xml', #seresnext26t\n        '/kaggle/input/birdclef-openvino-comp/sed_b3ns_30s_finetune/sed3_b3ns_120.xml', #b3ns\n    ]\n    \n    # CNN model\n    re_model_ckpt = [\n        '/kaggle/input/birdclef-openvino-comp/openvino_models_comp_half/re_120.xml', #resnet34d\n        '/kaggle/input/birdclef-openvino-comp/re_b3ns_ce/re_b3ns_120.xml', #b3ns\n        '/kaggle/input/birdclef-openvino-comp/re_v2s_30s_finetune/re_v2s_120.xml', #v2s\n        '/kaggle/input/birdclef-openvino-comp/re_b0ns_final/re_b0ns_120.xml', #b0ns\n    ]\n    ","metadata":{"papermill":{"duration":0.099568,"end_time":"2022-04-22T06:00:18.542447","exception":false,"start_time":"2022-04-22T06:00:18.442879","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-02T05:19:38.941810Z","iopub.execute_input":"2023-06-02T05:19:38.942262Z","iopub.status.idle":"2023-06-02T05:19:38.951873Z","shell.execute_reply.started":"2023-06-02T05:19:38.942217Z","shell.execute_reply":"2023-06-02T05:19:38.950213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(Config.train_path)\nConfig.num_classes = len(df_train.primary_label.unique())","metadata":{"papermill":{"duration":0.035353,"end_time":"2022-04-22T06:01:17.283888","exception":false,"start_time":"2022-04-22T06:01:17.248535","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-02T05:19:39.265767Z","iopub.execute_input":"2023-06-02T05:19:39.266206Z","iopub.status.idle":"2023-06-02T05:19:39.364003Z","shell.execute_reply.started":"2023-06-02T05:19:39.266164Z","shell.execute_reply":"2023-06-02T05:19:39.362638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\ndef sigmoid(a):\n    return 1 / (1 + np.exp(-a))\ndef odds(p):\n    return p / (1 - p)\ndef logit(p):\n    return np.log(odds(p))\n'''","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:19:39.494737Z","iopub.execute_input":"2023-06-02T05:19:39.495167Z","iopub.status.idle":"2023-06-02T05:19:39.502532Z","shell.execute_reply.started":"2023-06-02T05:19:39.495128Z","shell.execute_reply":"2023-06-02T05:19:39.501112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pred(df_test,num_workers=1,sleep=0,batch_size=1):\n    import openvino.runtime as ov\n    core = ov.Core()\n    \n    import numpy as np\n    import pandas as pd\n    import torch\n    import os\n    from torch.utils.data import Dataset, DataLoader\n    import warnings\n\n    warnings.filterwarnings('ignore')\n    import torch.nn as nn\n    import timm\n    import librosa as lb\n    import soundfile as sf\n    from  soundfile import SoundFile \n    import torchaudio\n\n    import torch.nn as nn\n    import time\n    from torch.nn import functional as F\n    from torch.distributions import Beta\n    from torch.nn.parameter import Parameter\n    from joblib.externals.loky.backend.context import get_context\n    #torch.jit.enable_onednn_fusion(True)\n\n\n    class BirdDatasetSED(torch.utils.data.Dataset):\n\n        def __init__(self, df, sr = Config.SR,n_mels=128, fmin=0, fmax=None, step=None, res_type=\"kaiser_fast\",resample=True, duration = Config.DURATION, train = True):\n\n            self.df = df\n            self.sr = sr \n            self.n_mels = n_mels\n            self.fmin = fmin\n            self.fmax = fmax or self.sr//2\n\n            self.train = train\n            self.duration = duration\n\n            self.audio_length = self.duration*self.sr\n            self.step = step or self.audio_length\n\n            self.res_type = res_type\n            self.resample = resample   \n\n        def __len__(self):\n            return len(self.df)\n\n        def read_file(self, filepath):\n            #audio, orig_sr = torchaudio.load(filepath)\n            #if orig_sr != self.sr:\n            #    # sinc_interpolation\n            #    resample_transform = torchaudio.transforms.Resample(orig_sr, self.sr, resampling_method=\"kaiser_window\")\n            #    audio = resample_transform(audio)\n\n            audio, orig_sr = sf.read(filepath, dtype=\"float32\")\n\n            if self.resample and orig_sr != self.sr:\n                audio = lb.resample(audio, orig_sr, self.sr, res_type=self.res_type)\n\n            seconds = []\n            for i in range(self.audio_length, len(audio) + self.step, self.step):\n                start = max(0, i - self.audio_length)\n                end = start + self.audio_length\n                if end > len(audio):\n                    pass\n                else:\n                    seconds.append(int(end/self.sr))\n\n            audio = np.concatenate([audio,audio,audio])\n            audios = []\n            for i,second in enumerate(seconds):\n                end_seconds = int(second)\n                start_seconds = int(end_seconds - Config.DURATION)\n\n                end_index = int(self.sr * (end_seconds + (Config.train_duration - Config.DURATION) / 2) ) + len(audio) // 3\n                start_index = int(self.sr * (start_seconds - (Config.train_duration - Config.DURATION) / 2) ) + len(audio) // 3\n                end_pad = int(self.sr * (Config.train_duration - Config.DURATION) / 2) \n                start_pad = int(self.sr * (Config.train_duration - Config.DURATION) / 2) \n                y = audio[start_index:end_index].astype(np.float32)\n                if i==0:\n                    y[:start_pad] = 0\n                elif i==(len(seconds)-1):\n                    y[-end_pad:] = 0\n                audios.append(y)\n            audios = np.stack(audios)\n            audios = torch.tensor(audios).float().unsqueeze(1)\n            spec384,spec256,spec300_another,spec_rev2s=transform_to_spec(audios,train=False)\n            return spec384,spec256,spec300_another,spec_rev2s\n\n        def __getitem__(self, idx):\n\n            return self.read_file(self.df.loc[idx, \"path\"])\n        \n\n    hop_length384 = Config.infer_duration*Config.SR // (384-1)\n    melspec_transform = torchaudio.transforms.MelSpectrogram(sample_rate=Config.SR, hop_length=hop_length384, n_mels=128, f_min=0, f_max=Config.SR//2, n_fft=2048, center=True, pad_mode='constant',norm='slaney',onesided=True,mel_scale='slaney')\n    hop_length256 = Config.infer_duration*Config.SR // (256-1)\n    melspec_transform256 = torchaudio.transforms.MelSpectrogram(sample_rate=Config.SR, hop_length=hop_length256, n_mels=128, f_min=0, f_max=Config.SR//2, n_fft=2048, center=True, pad_mode='constant',norm='slaney',onesided=True,mel_scale='slaney')\n    #hop_length224 = Config.infer_duration*Config.SR // (224-1)\n    #melspec_transform224 = torchaudio.transforms.MelSpectrogram(sample_rate=Config.SR, hop_length=hop_length224, n_mels=128, f_min=0, f_max=Config.SR//2, n_fft=2048, center=True, pad_mode='constant',norm='slaney',onesided=True,mel_scale='slaney')\n    hop_length300 = Config.infer_duration*Config.SR // (300-1)\n    melspec_transform300 = torchaudio.transforms.MelSpectrogram(sample_rate=Config.SR, hop_length=hop_length300, n_mels=128, f_min=50, f_max=14000, n_fft=1024, center=True, pad_mode='constant',norm='slaney',onesided=True,mel_scale='slaney')\n    melspec_transform_rev2s = torchaudio.transforms.MelSpectrogram(sample_rate=Config.SR, hop_length=320, n_mels=64, f_min=50, f_max=14000, n_fft=1024, center=True, pad_mode='constant',norm='slaney',onesided=True,mel_scale='slaney')\n    \n    db_transform = torchaudio.transforms.AmplitudeToDB(stype='power',top_db=80)\n\n    def transform_to_spec(audio,train=True):\n        import math\n        amin=1e-10\n        ref_value=1.0\n        db_multiplier = math.log10(max(amin, ref_value))\n        spec = melspec_transform(audio)     \n        #spec = torchaudio.functional.amplitude_to_DB(spec,multiplier=10,amin=amin,db_multiplier=db_multiplier,top_db=80)\n        spec = db_transform(spec)\n        spec256 = melspec_transform256(audio)\n        spec256 = db_transform(spec256)\n        \n        #spec224 = melspec_transform224(audio)\n        #spec224 = db_transform(spec224)\n        \n        spec300_another = melspec_transform300(audio)\n        spec300_another = db_transform(spec300_another)\n        \n        spec_rev2s = melspec_transform_rev2s(audio)\n        spec_rev2s = db_transform(spec_rev2s)\n        \n        spec384 = (spec+80)/80\n        spec256 = spec256/255\n        #spec224 = spec224/255\n        spec300_another = spec300_another/255\n        spec_rev2s = (spec_rev2s+80)/80\n        return spec384,spec256,spec300_another,spec_rev2s\n\n    \n    \n    def openvino_infer(model,data,tta):\n        outputs = model.infer(inputs=[data,tta])\n        outputs = torch.tensor(outputs[list(outputs.keys())[0]])\n        return outputs\n    \n    def openvino_infer_re(model,data):\n        outputs = model.infer(inputs=[data])\n        outputs = torch.tensor(outputs[list(outputs.keys())[0]])\n        return outputs\n    \n    def compute_deltas(\n            specgram: torch.Tensor,\n            win_length: int = 5,\n            mode: str = \"replicate\"\n    ) -> torch.Tensor:\n        r\"\"\"Compute delta coefficients of a tensor, usually a spectrogram:\n\n        .. math::\n           d_t = \\frac{\\sum_{n=1}^{\\text{N}} n (c_{t+n} - c_{t-n})}{2 \\sum_{n=1}^{\\text{N}} n^2}\n\n        where :math:`d_t` is the deltas at time :math:`t`,\n        :math:`c_t` is the spectrogram coeffcients at time :math:`t`,\n        :math:`N` is ``(win_length-1)//2``.\n\n        Args:\n            specgram (Tensor): Tensor of audio of dimension (..., freq, time)\n            win_length (int, optional): The window length used for computing delta (Default: ``5``)\n            mode (str, optional): Mode parameter passed to padding (Default: ``\"replicate\"``)\n\n        Returns:\n            Tensor: Tensor of deltas of dimension (..., freq, time)\n\n        Example\n            >>> specgram = torch.randn(1, 40, 1000)\n            >>> delta = compute_deltas(specgram)\n            >>> delta2 = compute_deltas(delta)\n        \"\"\"\n        device = specgram.device\n        dtype = specgram.dtype\n\n        # pack batch\n        shape = specgram.size()\n        specgram = specgram.reshape(1, -1, shape[-1])\n\n        assert win_length >= 3\n\n        n = (win_length - 1) // 2\n\n        # twice sum of integer squared\n        denom = n * (n + 1) * (2 * n + 1) / 3\n\n        specgram = torch.nn.functional.pad(specgram, (n, n), mode=mode)\n\n        kernel = torch.arange(-n, n + 1, 1, device=device, dtype=dtype).repeat(specgram.shape[1], 1, 1)\n\n        output = torch.nn.functional.conv1d(specgram, kernel, groups=specgram.shape[1]) / denom\n\n        # unpack batch\n        output = output.reshape(shape)\n\n        return output\n\n    def make_delta(\n        input_tensor: torch.Tensor\n    ):\n        input_tensor = input_tensor.transpose(3,2)\n        input_tensor = compute_deltas(input_tensor)\n        input_tensor = input_tensor.transpose(3,2)\n        return input_tensor\n\n\n    def image_delta(x):\n        delta_1 = make_delta(x)\n        delta_2 = make_delta(delta_1)\n        x = torch.cat([x,delta_1,delta_2], dim=1)\n        return x\n    \n    def reshp(images):\n        bs,clip_len,channel_num,mel_num,time_len = images.size()\n        images=images.reshape((bs*clip_len,channel_num,mel_num,time_len))\n        return images\n    \n    def predict(data_loader, models,re_models):   \n        predictions = []\n        pred_binary = []\n        dl_test = DataLoader(ds_test, batch_size=batch_size,num_workers = num_workers, multiprocessing_context=get_context('loky'))\n        \n        for spec384,spec256,spec300_another,spec_rev2s in dl_test:\n            spec384 = reshp(spec384)\n            spec256 = reshp(spec256)\n            spec300_another = reshp(spec300_another)\n            spec300_80 = (spec300_another*255+80)/80\n            spec_rev2s = reshp(spec_rev2s)\n            \n            out = []\n            for i,model in enumerate(models):\n                if i==0:\n                    images2_3chan = image_delta(spec384).numpy()\n\n                    if images2_3chan.shape[0]>120:\n                        output1 = openvino_infer(model,images2_3chan[:120,:,:,:],3)\n                        output2 = openvino_infer(model,images2_3chan[120:240,:,:,:],3)\n                        outputs = torch.cat([output1,output2],dim=0)\n                    else:\n                        outputs = openvino_infer(model,images2_3chan,3)\n                elif i==2:\n                    images_3chan = image_delta(spec300_another).numpy()\n                    if images_3chan.shape[0]>120:\n                        output1 = openvino_infer(model,images_3chan[:120,:,:,:],3)\n                        output2 = openvino_infer(model,images_3chan[120:240,:,:,:],3)\n                        outputs = torch.cat([output1,output2],dim=0)\n                    else:\n                        outputs = openvino_infer(model,images_3chan,3)\n                else:\n                    image_res = spec256.numpy()\n\n                    if image_res.shape[0]>120:\n                        output1 = openvino_infer(model,image_res[:120,:,:,:],2)\n                        output2 = openvino_infer(model,image_res[120:240,:,:,:],2)\n                        outputs = torch.cat([output1,output2],dim=0)\n                    else:\n                        outputs = openvino_infer(model,image_res,2)\n\n                out.append(outputs)\n            for i,model in enumerate(re_models):\n                if (i==0):\n                    images_center_resize1 = image_delta(spec256)[:,:,:,128:384].numpy()\n                    if images_center_resize1.shape[0]>120:\n                        output1 = openvino_infer_re(model,images_center_resize1[:120,:,:,:])\n                        output2 = openvino_infer_re(model,images_center_resize1[120:240,:,:,:])\n                        outputs = torch.cat([output1,output2],dim=0)\n                    else:\n                        outputs = openvino_infer_re(model,images_center_resize1)\n                elif (i==1):\n                    images_center_resize2 = image_delta(spec300_80)[:,:,:,150:450].numpy()\n                    if images_center_resize2.shape[0]>120:\n                        output1 = openvino_infer_re(model,images_center_resize2[:120,:,:,:])\n                        output2 = openvino_infer_re(model,images_center_resize2[120:240,:,:,:])\n                        outputs = torch.cat([output1,output2],dim=0)\n                    else:\n                        outputs = openvino_infer_re(model,images_center_resize2)\n                elif (i==2):\n                    images_re_v2s = image_delta(spec_rev2s)[:,:,:,250:750].numpy()\n                    if images_re_v2s.shape[0]>120:\n                        output1 = openvino_infer_re(model,images_re_v2s[:120,:,:,:])\n                        output2 = openvino_infer_re(model,images_re_v2s[120:240,:,:,:])\n                        outputs = torch.cat([output1,output2],dim=0)\n                    else:\n                        outputs = openvino_infer_re(model,images_re_v2s)\n                elif (i==3):\n                    image_b0ns = spec256[:,:,:,128:384].numpy()\n                    if image_b0ns.shape[0]>120:\n                        output1 = openvino_infer_re(model,image_b0ns[:120,:,:,:])\n                        output2 = openvino_infer_re(model,image_b0ns[120:240,:,:,:])\n                        outputs = torch.cat([output1,output2],dim=0)\n                    else:\n                        outputs = openvino_infer_re(model,image_b0ns)    \n                else:\n                    outputs = model(images_center_resize3)\n    \n                out.append(outputs)\n                \n            predictions.append(out)\n        return predictions\n\n    import gc\n\n    print(f\"Create Dataloader...\")\n\n    ds_test = BirdDatasetSED(\n        df_test, \n        sr = Config.SR,\n        duration = Config.DURATION,\n        train = False\n    )\n\n    \n    #print(\"Model Creation\")\n    models = []\n    for i,ckpt in enumerate(Config.model_ckpt):\n        #if i==0:\n        #    model = load_mdl(name,ckpt,size,sed_3chan=True)\n        #else:\n        #    model = load_mdl(name,ckpt,size)\n\n        model = core.read_model(model=ckpt)\n        model = core.compile_model(model, device_name=\"CPU\")\n        model = model.create_infer_request()\n        models.append(model)\n        \n    re_models = []\n    for i,ckpt in enumerate(Config.re_model_ckpt):\n\n        model = core.read_model(model=ckpt)\n        model = core.compile_model(model, device_name=\"CPU\")\n        model = model.create_infer_request()\n        re_models.append(model)\n\n    print(\"Running Inference..\")\n    time.sleep(sleep)\n    preds = predict(ds_test, models,re_models)   \n\n    return preds","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:22:50.220764Z","iopub.execute_input":"2023-06-02T05:22:50.221202Z","iopub.status.idle":"2023-06-02T05:22:50.288299Z","shell.execute_reply.started":"2023-06-02T05:22:50.221159Z","shell.execute_reply":"2023-06-02T05:22:50.286317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom pathlib import Path\ndf_test = pd.DataFrame(\n     [(path.stem, *path.stem.split(\"_\"), path) for path in Path(Config.test_path).glob(\"*.ogg\")],\n    columns = [\"filename\", \"name\" ,\"id\", \"path\"]\n)\nprint(df_test.shape)\ndf_test.head()","metadata":{"papermill":{"duration":0.039034,"end_time":"2022-04-22T06:01:17.350173","exception":false,"start_time":"2022-04-22T06:01:17.311139","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-02T05:22:50.710408Z","iopub.execute_input":"2023-06-02T05:22:50.710853Z","iopub.status.idle":"2023-06-02T05:22:50.731519Z","shell.execute_reply.started":"2023-06-02T05:22:50.710810Z","shell.execute_reply":"2023-06-02T05:22:50.730064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df_test = pd.concat([df_test]*200,axis=0).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:22:51.016522Z","iopub.execute_input":"2023-06-02T05:22:51.016940Z","iopub.status.idle":"2023-06-02T05:22:51.021857Z","shell.execute_reply.started":"2023-06-02T05:22:51.016902Z","shell.execute_reply":"2023-06-02T05:22:51.020462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cpu_num=2","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:22:51.371562Z","iopub.execute_input":"2023-06-02T05:22:51.372010Z","iopub.status.idle":"2023-06-02T05:22:51.376734Z","shell.execute_reply.started":"2023-06-02T05:22:51.371969Z","shell.execute_reply":"2023-06-02T05:22:51.375658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_job = min([cpu_num,len(df_test)])\nsplit = len(df_test)//num_job\nnum_job,split","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:22:51.614312Z","iopub.execute_input":"2023-06-02T05:22:51.615144Z","iopub.status.idle":"2023-06-02T05:22:51.622956Z","shell.execute_reply.started":"2023-06-02T05:22:51.615098Z","shell.execute_reply":"2023-06-02T05:22:51.621698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfs_test = []\ndf_test_left = None\nfor i in range(num_job):\n    df_test_split = df_test.iloc[i*split:(i+1)*split].reset_index(drop=True)\n    dfs_test.append(df_test_split)\n    if i==num_job-1:\n        df_test_left = df_test.iloc[(i+1)*split:].reset_index(drop=True)\nlen(dfs_test),len(df_test_left)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:22:51.791600Z","iopub.execute_input":"2023-06-02T05:22:51.792081Z","iopub.status.idle":"2023-06-02T05:22:51.803975Z","shell.execute_reply.started":"2023-06-02T05:22:51.792028Z","shell.execute_reply":"2023-06-02T05:22:51.802328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nt1=time.time()\n#results1 = joblib.Parallel(n_jobs=num_job, backend='loky')(joblib.delayed(pred)(df_test) for df_test in dfs_test)\nresults1 = joblib.Parallel(n_jobs=num_job, backend='loky')(joblib.delayed(pred)(df_test,num_workers,sl,batch_size) for df_test,num_workers,sl,batch_size in zip(dfs_test,[2,2],[0,5],[2,2]))\nt2=time.time()\nprint(t2-t1)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:22:52.741076Z","iopub.execute_input":"2023-06-02T05:22:52.741490Z","iopub.status.idle":"2023-06-02T05:23:55.920890Z","shell.execute_reply.started":"2023-06-02T05:22:52.741448Z","shell.execute_reply":"2023-06-02T05:23:55.919235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t1=time.time()\nresults2 = []\nif len(df_test_left)>0:\n    results2 = joblib.Parallel(n_jobs=num_job, backend='loky')(joblib.delayed(pred)(df_test_left.iloc[i:i+1].reset_index(drop=True)) for i in range(len(df_test_left)))\nt2=time.time()\nprint(t2-t1)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-06-02T05:24:01.369312Z","iopub.execute_input":"2023-06-02T05:24:01.369794Z","iopub.status.idle":"2023-06-02T05:24:01.378181Z","shell.execute_reply.started":"2023-06-02T05:24:01.369735Z","shell.execute_reply":"2023-06-02T05:24:01.376578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = results1+results2","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:24:08.845133Z","iopub.execute_input":"2023-06-02T05:24:08.845933Z","iopub.status.idle":"2023-06-02T05:24:08.851458Z","shell.execute_reply.started":"2023-06-02T05:24:08.845889Z","shell.execute_reply":"2023-06-02T05:24:08.850039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds=[]\nfor r in results:\n    preds+=r\nlen(preds)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:24:09.547125Z","iopub.execute_input":"2023-06-02T05:24:09.547573Z","iopub.status.idle":"2023-06-02T05:24:09.556504Z","shell.execute_reply.started":"2023-06-02T05:24:09.547531Z","shell.execute_reply":"2023-06-02T05:24:09.555060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds1=[]\npreds2=[]\npreds3=[]\npreds4=[]\npreds5=[]\npreds6=[]\npreds7=[]\nfor r1,r2,r3,r4,r5,r6,r7 in preds:\n    preds1.append(r1)\n    preds2.append(r2)\n    preds3.append(r3)\n    preds4.append(r4)\n    preds5.append(r5)\n    preds6.append(r6)\n    preds7.append(r7)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:24:11.011075Z","iopub.execute_input":"2023-06-02T05:24:11.012199Z","iopub.status.idle":"2023-06-02T05:24:11.019560Z","shell.execute_reply.started":"2023-06-02T05:24:11.012152Z","shell.execute_reply":"2023-06-02T05:24:11.018216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filenames = df_test.filename.values.tolist()\n\nbird_cols = list(pd.get_dummies(df_train['primary_label']).columns)\nsub_df = pd.DataFrame(columns=['row_id']+bird_cols)","metadata":{"papermill":{"duration":0.052364,"end_time":"2022-04-22T06:01:22.708806","exception":false,"start_time":"2022-04-22T06:01:22.656442","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-02T05:24:12.504910Z","iopub.execute_input":"2023-06-02T05:24:12.505325Z","iopub.status.idle":"2023-06-02T05:24:12.531782Z","shell.execute_reply.started":"2023-06-02T05:24:12.505289Z","shell.execute_reply":"2023-06-02T05:24:12.530529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:24:13.542677Z","iopub.execute_input":"2023-06-02T05:24:13.543395Z","iopub.status.idle":"2023-06-02T05:24:13.561432Z","shell.execute_reply.started":"2023-06-02T05:24:13.543353Z","shell.execute_reply":"2023-06-02T05:24:13.560242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generate Submission csv","metadata":{}},{"cell_type":"code","source":"def make_row_ids(file):\n    num_rows = 120\n    row_ids = np.array([f'{file}_{(i+1)*5}' for i in range(num_rows)])\n    return row_ids","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:24:14.810178Z","iopub.execute_input":"2023-06-02T05:24:14.810948Z","iopub.status.idle":"2023-06-02T05:24:14.817851Z","shell.execute_reply.started":"2023-06-02T05:24:14.810898Z","shell.execute_reply":"2023-06-02T05:24:14.816116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#row_ids = joblib.Parallel(n_jobs=4, backend='loky')(joblib.delayed(make_row_ids)(preds[i],file) for i, file in enumerate(filenames))\nrow_ids = joblib.Parallel(n_jobs=4, backend='loky')(joblib.delayed(make_row_ids)(file) for i, file in enumerate(filenames))\nrow_ids = np.concatenate(row_ids,axis=0)\n#data = np.concatenate(preds,axis=0)\ndata1 = torch.cat(preds1,dim=0).logit()\ndata2 = torch.cat(preds2,dim=0).logit()\ndata3 = torch.cat(preds3,dim=0).logit()\ndata4 = torch.cat(preds4,dim=0)\ndata5 = torch.cat(preds5,dim=0)\ndata6 = torch.cat(preds6,dim=0)\ndata7 = torch.cat(preds7,dim=0)\n#data_binary = np.concatenate(preds_binary,axis=0)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:24:16.031238Z","iopub.execute_input":"2023-06-02T05:24:16.031649Z","iopub.status.idle":"2023-06-02T05:24:16.577658Z","shell.execute_reply.started":"2023-06-02T05:24:16.031612Z","shell.execute_reply":"2023-06-02T05:24:16.575804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ensemble(sed_pred,sed_pred2,sed_pred3,re_pred,re_pred2,re_pred3,re_pred4):\n    \n    sed_pred[:,:] = 0.25*sed_pred[:,:] + 0.1*sed_pred2 + 0.21*sed_pred3 + 0.1*re_pred[:,:] + 0.15*re_pred2[:,:] + 0.15*re_pred3[:,:] + 0.04*re_pred4[:,:]\n    \n    return sed_pred","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:24:38.196392Z","iopub.execute_input":"2023-06-02T05:24:38.196843Z","iopub.status.idle":"2023-06-02T05:24:38.203538Z","shell.execute_reply.started":"2023-06-02T05:24:38.196802Z","shell.execute_reply":"2023-06-02T05:24:38.202113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = ensemble(data1,data2,data3,data4,data5,data6,data7).sigmoid().numpy()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:24:41.188683Z","iopub.execute_input":"2023-06-02T05:24:41.189159Z","iopub.status.idle":"2023-06-02T05:24:41.197959Z","shell.execute_reply.started":"2023-06-02T05:24:41.189116Z","shell.execute_reply":"2023-06-02T05:24:41.196602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df['row_id'] = row_ids\nsub_df[bird_cols] = data\n#sub_df = pd.concat(dfs).reset_index(drop=True)\nsub_df","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:24:56.916191Z","iopub.execute_input":"2023-06-02T05:24:56.917507Z","iopub.status.idle":"2023-06-02T05:24:57.042321Z","shell.execute_reply.started":"2023-06-02T05:24:56.917452Z","shell.execute_reply":"2023-06-02T05:24:57.041059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T05:24:57.184218Z","iopub.execute_input":"2023-06-02T05:24:57.185542Z","iopub.status.idle":"2023-06-02T05:24:57.245040Z","shell.execute_reply.started":"2023-06-02T05:24:57.185491Z","shell.execute_reply":"2023-06-02T05:24:57.244001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}