{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":19596,"databundleVersionId":1292430,"sourceType":"competition"},{"sourceId":7388083,"sourceType":"datasetVersion","datasetId":4294482},{"sourceId":7389964,"sourceType":"datasetVersion","datasetId":4295827}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install librosa==0.9.2\nimport cv2 #image processing\nimport audioread #reading and processing audio files\nimport logging \nimport os \nimport random \nimport time \nimport warnings \nimport librosa\nimport numpy as np\nimport pandas as pd\nimport soundfile as sf \nimport torch \nimport torch.nn as nn \nimport torch.nn. functional as F #for element-wise functions\nimport torch.utils.data as data #for loading and batching data during training\nfrom contextlib import contextmanager \nfrom pathlib import Path \nfrom typing import Optional \nfrom fastprogress import progress_bar\nfrom sklearn.metrics import f1_score\nfrom torchvision import models \n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-12T22:27:41.105393Z","iopub.execute_input":"2024-01-12T22:27:41.106486Z","iopub.status.idle":"2024-01-12T22:27:54.492593Z","shell.execute_reply.started":"2024-01-12T22:27:41.106429Z","shell.execute_reply":"2024-01-12T22:27:54.491392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed (seed: int = 42): #sets random seed for different libraries\n    random.seed (seed)\n    np.random.seed (seed)\n    os.environ[ \"PYTHONHASHSEED\"] = str(seed) #for hash functions\n    torch.manual_seed (seed) #random of pytorch I\n    torch.cuda.manual_seed (seed) #type: ignore #same but for GPU\n    torch.backends.cudnn.deterministic = True #type: ignore #deterministic mode\n    torch.backends. cudnn.benchmark = True #type: ignore #optimize performance\ndef get_logger (out_file=None):\n    logger =logging.getLogger()\n    formatter = logging. Formatter(\"%(asctime)s - %(levelname)s - %(message)s\") #timestamp, log level, message\n    logger.handlers = [] #object that transfers log\n    logger.setLevel(logging. INFO) #capture only INFO level\n    handler =logging.StreamHandler() #to the console\n    handler.setFormatter (formatter)\n    handler.setLevel(logging. INFO)\n    logger.addHandler (handler)\n    if out_file is not None:\n        fh= logging. FileHandler(out_file)\n        fh. setFormatter (formatter)\n        fh.setLevel (logging. INFO)\n        logger.addHandler (fh)\n    logger.info(\"logger set up\")\n    return logger\n@contextmanager #measure execution time\ndef timer (name: str, logger: Optional[logging.Logger] = None) :\n    to = time.time() #current time\n    msg =f\"[{name}] start\"\n    if logger is None:\n        print (msg)\n    else:\n        logger.info(msg)\n    yield\n    msg = f\"[{name}] done in {time.time() - to:.2f} s\"\n    if logger is None:\n        print (msg)\n    else:\n        logger.info(msg)","metadata":{"execution":{"iopub.status.busy":"2024-01-12T22:27:54.495606Z","iopub.execute_input":"2024-01-12T22:27:54.495956Z","iopub.status.idle":"2024-01-12T22:27:54.507643Z","shell.execute_reply.started":"2024-01-12T22:27:54.495926Z","shell.execute_reply":"2024-01-12T22:27:54.506783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logger=get_logger(\"main.log\")\nset_seed(1213)","metadata":{"execution":{"iopub.status.busy":"2024-01-12T22:27:54.509149Z","iopub.execute_input":"2024-01-12T22:27:54.509451Z","iopub.status.idle":"2024-01-12T22:27:54.532739Z","shell.execute_reply.started":"2024-01-12T22:27:54.509425Z","shell.execute_reply":"2024-01-12T22:27:54.531570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGET_SR=32000\ntest=pd.read_csv(\"/kaggle/input/birdcall-check/test.csv\")\ntest_audio=\"/kaggle/input/birdcall-check/test_audio\"\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-12T22:27:54.535648Z","iopub.execute_input":"2024-01-12T22:27:54.535961Z","iopub.status.idle":"2024-01-12T22:27:54.557270Z","shell.execute_reply.started":"2024-01-12T22:27:54.535933Z","shell.execute_reply":"2024-01-12T22:27:54.556229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ResNet (nn. Module): #base class\n    def __init__(self, base_model_name: str, pretrained=False, #constructor method #weights\n    num_classes=264):\n        super (). __init__() #initializing the base class\n        base_model=models.__getattribute__(base_model_name) (\n        pretrained-pretrained)\n        layers =list (base_model.children())[:-2] #except pooling and dense\n        layers.append(nn. AdaptiveMaxPool2d (1))\n        self.encoder = nn.Sequential (*layers)\n        in_features = base_model.fc.in_features #number of input features\n        self.classifier = nn.Sequential(\n        nn. Linear (in_features, 1024), nn. ReLU(), nn. Dropout (p=0.2),\n        nn. Linear (1024, 1024), nn. ReLU(), nn. Dropout (p=0.2),\n        nn. Linear (1024, num_classes))\n    def forward(self,x):\n        batch_size = x.size(0) #input tensor\n        x = self.encoder (x). view (batch_size, -1) #1D tensor\n        x = self.classifier (x)\n        multiclass_proba = F.softmax(x, dim=1)\n        multilabel_proba = F.sigmoid(x)\n        return {\n            \"logits\": x,\n            \"multiclass_proba\": multiclass_proba,\n            \"multilabel_proba\": multilabel_proba\n        }\n    ","metadata":{"execution":{"iopub.status.busy":"2024-01-12T22:27:54.558714Z","iopub.execute_input":"2024-01-12T22:27:54.559018Z","iopub.status.idle":"2024-01-12T22:27:54.569103Z","shell.execute_reply.started":"2024-01-12T22:27:54.558992Z","shell.execute_reply":"2024-01-12T22:27:54.568083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_config = {\n    \"base_model_name\": \"resnet50\",\n    \"pretrained\": False,\n    \"num_classes\": 264 \n}\nmelspectrogram_parameters = {\n    \"n_mels\": 128, #number of Mel bins\n    \"fmin\": 20,\n    \"fmax\": 16999\n}\nweights_path = \"../input/birdcall-resnet50-init-weights/best.pth\"","metadata":{"execution":{"iopub.status.busy":"2024-01-12T22:27:54.570706Z","iopub.execute_input":"2024-01-12T22:27:54.571294Z","iopub.status.idle":"2024-01-12T22:27:54.581253Z","shell.execute_reply.started":"2024-01-12T22:27:54.571266Z","shell.execute_reply":"2024-01-12T22:27:54.580260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\ndf = pd. read_csv (\"/kaggle/input/birdsong-recognition/train.csv\")\nunique_bird_names = df['ebird_code' ].unique() \nlabel_encoder=LabelEncoder ()\nencoded_labels = label_encoder.fit_transform(unique_bird_names)\nBIRD_CODE = dict(zip(unique_bird_names, encoded_labels))\n# for bird_name, label in BIRD_CODE.items():\n# print (f\"{bird_name}: {label}\")\nINV_BIRD_CODE = {v: k for k, v in BIRD_CODE.items()}\nfor bird_name, label in INV_BIRD_CODE.items():\n print(f\" {bird_name}: {label}\")","metadata":{"execution":{"iopub.status.busy":"2024-01-12T22:27:54.582438Z","iopub.execute_input":"2024-01-12T22:27:54.583154Z","iopub.status.idle":"2024-01-12T22:27:54.956228Z","shell.execute_reply.started":"2024-01-12T22:27:54.583124Z","shell.execute_reply":"2024-01-12T22:27:54.955109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mono_to_color (X: np.ndarray,\nmean=None,\nstd=None,\nnorm_max=None,\nnorm_min=None,\neps=1e-6):\n    X = np.stack([X, X, X], axis=-1)\n    # Standardize\n\n    mean =mean or X.mean()\n    X = X -mean\n    std= std or X.std()\n    Xstd =X /(std + eps)\n    _min, _max = Xstd.min(), Xstd.max()\n    norm_max = norm_max or _max\n    norm_min= norm_min or _min\n    if (_max-_min) > eps:\n        # Normalize to [0, 255]\n        V = Xstd\n        V[V <norm_min] = norm_min\n        V[V> norm_max] = norm_max\n        V = 255 * (V - norm_min) / (norm_max - norm_min)\n        V = V.astype (np. uint8)\n    else:\n        V = np.zeros_like (Xstd, dtype=np. uint8)\n    return V\nclass TestDataset (data. Dataset):\n    def __init__(self, df: pd.DataFrame, clip: np.ndarray,\n    img_size=224, melspectrogram_parameters={}):\n        self.df =df\n        self.clip= clip\n        self.img_size = img_size\n        self.melspectrogram_parameters = melspectrogram_parameters\n    def __len__(self):\n        return len(self.df) #number of samples\n    def __getitem__(self, idx: int):\n        SR = 32000\n        sample= self.df.loc [idx, :] #return row\n        site =sample.site\n        row_id = sample.row_id\n        if site == \"site_3\":\n            y = self.clip.astype (np.float32)\n            len_y= len(y)\n            start = 0\n            end= SR* 5\n            images = []\n            while len_y> start:\n                y_batch = y[start:end].astype (np. float32)\n                if len(y_batch) != (SR* 5):\n                    break\n                start=end\n                end =end + SR * 5\n                melspec = librosa.feature.melspectrogram(y=y_batch,\n                sr=SR,\n                **self.melspectrogram_parameters)\n                melspec = librosa.power_to_db(melspec).astype (np. float32)\n                image =mono_to_color (melspec)\n                height, width, _ = image.shape\n                image = cv2.resize(image, (int(width * self.img_size/ height), self.img_size))\n                image = np.moveaxis (image, 2, 0) #color channel axis to the first dimension\n                image = (image/255.0).astype(np. float32)\n                images.append(image)\n            images = np.asarray(images)\n            return images, row_id, site\n        else: \n            end_seconds = int(sample.seconds)\n            start_seconds = int(end_seconds - 5)\n            start_index = SR*start_seconds\n            end_index = SR* end_seconds\n            y = self.clip [start_index:end_index].astype(np.float32)\n            melspec = librosa.feature.melspectrogram (y=y, sr=SR, **self.melspectrogram_parameters)\n            melspec = librosa.power_to_db (melspec).astype (np. float32)\n            image =mono_to_color (melspec)\n            height ,width,_ = image.shape\n            image =cv2.resize(image, (int(width * self.img_size/ height), self.img_size))\n            image = np.moveaxis (image, 2, 0) #color channel axis to the first dimension\n            image = (image/255.0).astype (np. float32)\n            return image, row_id, site","metadata":{"execution":{"iopub.status.busy":"2024-01-12T22:27:54.957729Z","iopub.execute_input":"2024-01-12T22:27:54.958058Z","iopub.status.idle":"2024-01-12T22:27:54.981379Z","shell.execute_reply.started":"2024-01-12T22:27:54.958018Z","shell.execute_reply":"2024-01-12T22:27:54.980127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model(config: dict, weights_path: str):\n    model =ResNet (**config)\n    checkpoint = torch.load (weights_path,map_location=torch.device('cpu')) #pretrained weights\n    model.load_state_dict(checkpoint [ \"model_state_dict\"]) #initializing learned parameters of the\n    device = torch.device(\"cpu\") \n    model.to (device) \n    model.eval()\n    return model\n\ndef prediction_for_clip(test_df: pd.DataFrame,\nclip: np.ndarray,\nmodel: ResNet,\nmel_params: dict,\nthreshold=0.5):\n    dataset =TestDataset (df=test_df,\n    clip=clip,\n    img_size=224,\n    melspectrogram_parameters =mel_params)\n    loader = data. DataLoader (dataset, batch_size=1, shuffle=False)\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    model.eval()\n    prediction_dict = {}\n    for image, row_id, site in progress_bar (loader):\n        site = site [0] \n        row_id = row_id[0]\n        if site in {\"site_1\", \"site_2\"}:\n            image =image.to (device)\n            with torch.no_grad():\n                prediction = model(image)\n                proba =prediction [ \"multilabel_proba\"].detach().cpu ().numpy().reshape(-1)\n            events =proba >= threshold\n            labels = np.argwhere (events).reshape(-1).tolist()\n        else:\n            # to avoid prediction on large batch\n            image =image. squeeze (0)\n            batch_size = 16\n            whole_size=image.size (0)\n            if whole_size % batch_size == 0:\n                n_iter =whole_size // batch_size\n            else:\n                n_iter = whole_size // batch_size + 1\n            all_events = set()\n            for batch_i in range(n_iter):\n                batch =image[batch_i* batch_size: (batch_i + 1) * batch_size]\n                if batch.ndim == 3:\n                    batch =batch.unsqueeze(0)\n                batch =batch. to (device)\n                with torch.no_grad():\n                    prediction =model (batch)\n                    proba =prediction [ \"multilabel_proba\"].detach().cpu ().numpy ()\n                events =proba >= threshold\n                for i in range (len (events)):\n                    event =events [i, :]\n                    labels = np.argwhere (event) . reshape(-1).tolist()\n                    for label in labels:\n                        all_events.add(label)\n            labels =list (all_events)\n        if len (labels) == 0:\n            prediction_dict [row_id] = \"nocall\"\n        else:\n            labels_str_list = list (map (lambda x: INV_BIRD_CODE [x], labels))\n            label_string = \" \" .join(labels_str_list)\n            prediction_dict [row_id] = label_string\n    return prediction_dict","metadata":{"execution":{"iopub.status.busy":"2024-01-12T22:27:54.983116Z","iopub.execute_input":"2024-01-12T22:27:54.983485Z","iopub.status.idle":"2024-01-12T22:27:55.003068Z","shell.execute_reply.started":"2024-01-12T22:27:54.983457Z","shell.execute_reply":"2024-01-12T22:27:55.002107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prediction (test_df: pd. DataFrame,\ntest_audio: Path,\nmodel_config: dict,\nmel_params: dict,\nweights_path: str,\nthreshold=0.5):\n    model = get_model(model_config, weights_path)\n    unique_audio_id = test_df.audio_id.unique ()\n    warnings.filterwarnings (\"ignore\")\n    prediction_dfs = []\n    for audio_id in unique_audio_id:\n        with timer (f\"Loading {audio_id}\", logger):\n            clip,_= librosa.load(test_audio + \"/\" + (audio_id + \".mp3\"),\n            sr=TARGET_SR,\n            mono=True,\n            res_type=\"scipy\")\n        test_df_for_audio_id = test_df.query(\n            f\"audio_id == '{audio_id}'\").reset_index (drop=True)\n        with timer (f\"Prediction on {audio_id}\", logger):\n            prediction_dict = prediction_for_clip(test_df_for_audio_id,\n            clip=clip,\n            model=model,\n            mel_params=mel_params,\n            threshold=threshold)\n        row_id = list (prediction_dict.keys())\n        birds = list (prediction_dict.values())\n        prediction_df = pd.DataFrame({\n        \"row_id\": row_id,\n        \"birds\": birds\n        })\n        prediction_dfs.append(prediction_df)\n    prediction_df = pd.concat(prediction_dfs, axis=0, sort=False).reset_index (drop=True)\n    return prediction_df","metadata":{"execution":{"iopub.status.busy":"2024-01-12T22:27:55.006344Z","iopub.execute_input":"2024-01-12T22:27:55.006707Z","iopub.status.idle":"2024-01-12T22:27:55.018744Z","shell.execute_reply.started":"2024-01-12T22:27:55.006665Z","shell.execute_reply":"2024-01-12T22:27:55.017521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = prediction (test_df=test,\n    test_audio=test_audio,\n    model_config=model_config,\n    mel_params=melspectrogram_parameters,\n    weights_path=weights_path,\n    threshold= 0.8 )\nsubmission.to_csv (\"submission.csv\", index=False) ","metadata":{"execution":{"iopub.status.busy":"2024-01-12T22:27:55.020006Z","iopub.execute_input":"2024-01-12T22:27:55.020402Z","iopub.status.idle":"2024-01-12T22:28:48.140311Z","shell.execute_reply.started":"2024-01-12T22:27:55.020373Z","shell.execute_reply":"2024-01-12T22:28:48.139327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2024-01-12T22:28:48.141958Z","iopub.execute_input":"2024-01-12T22:28:48.142328Z","iopub.status.idle":"2024-01-12T22:28:48.153655Z","shell.execute_reply.started":"2024-01-12T22:28:48.142297Z","shell.execute_reply":"2024-01-12T22:28:48.152831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import statistics\nunique_audio=test['audio_id'].unique()\npredictions=[]\nfinal_pred=[]\nfor audio in unique_audio:\n    for j in range(len(submission['row_id'])):\n        if audio in submission['row_id'][j]:\n            predictions.append(submission['birds'][j])\n    final_pred.append(statistics.mode(predictions))\nfinal_submission=pd.DataFrame({'audio_id':unique_audio,'predicted_birds':final_pred})\nfinal_submission.to_csv(\"final_submission.csv\", index=False)\nfinal_submission","metadata":{"execution":{"iopub.status.busy":"2024-01-12T22:28:48.155028Z","iopub.execute_input":"2024-01-12T22:28:48.155445Z","iopub.status.idle":"2024-01-12T22:28:48.185580Z","shell.execute_reply.started":"2024-01-12T22:28:48.155418Z","shell.execute_reply":"2024-01-12T22:28:48.184465Z"},"trusted":true},"execution_count":null,"outputs":[]}]}