{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Summary of this notebook\n\nI'm affraid that this notebook has become so long.\nHere I summarize what we are gonna do.\n\n1. Import & Configration\n\nHere we import libraries and set some configurations as usual.\n\n2. Make Candidate\n\nHere we define functions which extract bird candidates for lightgbm stage training. \nThe output of melspectrogram multiclass classifier is used for that.\n\n3. Add features\n\nHere we define functions which make features for lightgbm stage training.\n\n4. Calculate birdcall probabilities (397dims per clip) from melspectrograms\n\nHere we define functions which calculate birdcall probabilities from melspectrograms.\nThey are used for training_soundscapes audios in this notebook.\nSpeaking of train_short_audio, we have already prepare birdcall probabilities data in csv format.\n(Check '../input/metadata-probability-v0525-2100')\n\n5. Training (Lightgbm)\n\nHere we define functions which train lightgbm models.\n\n6. Optimize Threshold\n\nHere we define functions which optimize the thresholds.\n\n7. Make Submission\n\nHere we define functions which make submission for Kaggle BirdCLEF 2021 Competition.\n\n8. Main\n\nHere we run the functions we have defined.\nTo take a quick look this notebook, this part is good place to start.\n\n# input & output of this notebook\n\n[input]\n\nbirdclef-2021 (original data)\n\nmelspectrogram multiclassifier models (Ⅰ)\n\nhttps://www.kaggle.com/namakemono/birdclef-groupby-author-05221040-728258\n\nhttps://www.kaggle.com/kami634/clefmodel\n\ntrain_short_audio birdcall probabilities calculated by melspectrogram multiclassifier models (Ⅰ)\n\nhttps://www.kaggle.com/namakemono/metadata-probability-v0525-2100\n\nresnest library\n\nhttps://www.kaggle.com/ttahara/resnest50-fast-package\n\nsklearn library (To use StratifiedGroupKfold, we have to install scikit-learn 1.0.dev0)\n\nhttps://www.kaggle.com/namakemono/scikit-learn-10dev0\n\n[output]\n\nsubmission.csv","metadata":{"papermill":{"duration":0.02392,"end_time":"2021-06-03T14:28:53.64268","exception":false,"start_time":"2021-06-03T14:28:53.61876","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Import & Configuration","metadata":{"papermill":{"duration":0.022238,"end_time":"2021-06-03T14:28:53.6874","exception":false,"start_time":"2021-06-03T14:28:53.665162","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# !cp -r \"../input/scikit-learn-10dev0/\" ./\n# !pip install \"/kaggle/working/scikit-learn-10dev0/scikit_learn-1.0.dev0-cp37-cp37m-manylinux2010_x86_64.whl\"\n# !rm -r '/kaggle/working/resnest'\n!cp -r \"../input/resnest50-fast-package/resnest-0.0.6b20200701/resnest\" ./\n!pip install -q \"/kaggle/working/resnest\"\n# !pip install -U scikit-learn\n# !pip install resnest\n# !pip install -q \"../input/resnest-package/resnest-0.0.6b20200701/resnest\"","metadata":{"papermill":{"duration":19.268619,"end_time":"2021-06-03T14:29:12.979037","exception":false,"start_time":"2021-06-03T14:28:53.710418","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:55:02.393210Z","iopub.execute_input":"2023-07-23T18:55:02.393571Z","iopub.status.idle":"2023-07-23T18:55:37.917406Z","shell.execute_reply.started":"2023-07-23T18:55:02.393539Z","shell.execute_reply":"2023-07-23T18:55:37.916248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install resampy\n# import resampy","metadata":{"execution":{"iopub.status.busy":"2023-07-23T18:55:37.919901Z","iopub.execute_input":"2023-07-23T18:55:37.920267Z","iopub.status.idle":"2023-07-23T18:55:37.927173Z","shell.execute_reply.started":"2023-07-23T18:55:37.920232Z","shell.execute_reply":"2023-07-23T18:55:37.925474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"papermill":{"duration":0.030206,"end_time":"2021-06-03T14:29:13.034071","exception":false,"start_time":"2021-06-03T14:29:13.003865","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:55:37.928813Z","iopub.execute_input":"2023-07-23T18:55:37.929392Z","iopub.status.idle":"2023-07-23T18:55:37.939411Z","shell.execute_reply.started":"2023-07-23T18:55:37.929359Z","shell.execute_reply":"2023-07-23T18:55:37.938498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc","metadata":{"execution":{"iopub.status.busy":"2023-07-23T18:55:37.942340Z","iopub.execute_input":"2023-07-23T18:55:37.942950Z","iopub.status.idle":"2023-07-23T18:55:37.950331Z","shell.execute_reply.started":"2023-07-23T18:55:37.942918Z","shell.execute_reply":"2023-07-23T18:55:37.949432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\n\nimport re\nimport time\n\nimport pickle\nfrom typing import List\nfrom tqdm.notebook import tqdm\n\n# sound\nimport librosa as lb\nimport soundfile as sf\n\n# pytorch\nimport torch\nfrom torch import nn\nfrom  torch.utils.data import Dataset, DataLoader\nfrom resnest.torch import resnest50\n\nimport tensorflow as tf\n\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.metrics import accuracy_score, f1_score, recall_score, precision_score\nimport xgboost as xgb\nimport pickle\nfrom catboost import CatBoostClassifier\nfrom catboost import Pool\nfrom imblearn.over_sampling import RandomOverSampler\nimport lightgbm as lgb\nimport random\n\nfrom scipy.optimize import minimize","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":10.323138,"end_time":"2021-06-03T14:29:23.380753","exception":false,"start_time":"2021-06-03T14:29:13.057615","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:55:37.951935Z","iopub.execute_input":"2023-07-23T18:55:37.952557Z","iopub.status.idle":"2023-07-23T18:55:50.855106Z","shell.execute_reply.started":"2023-07-23T18:55:37.952525Z","shell.execute_reply":"2023-07-23T18:55:50.854110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BIRD_LIST = sorted(os.listdir('../input/birdclef-2021/train_short_audio'))\nBIRD2IDX = {bird:idx for idx, bird in enumerate(BIRD_LIST)}\nBIRD2IDX['nocall'] = -1\nIDX2BIRD = {idx:bird for bird, idx in BIRD2IDX.items()}","metadata":{"papermill":{"duration":0.078401,"end_time":"2021-06-03T14:29:23.484177","exception":false,"start_time":"2021-06-03T14:29:23.405776","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:55:50.856535Z","iopub.execute_input":"2023-07-23T18:55:50.857786Z","iopub.status.idle":"2023-07-23T18:55:50.966664Z","shell.execute_reply.started":"2023-07-23T18:55:50.857760Z","shell.execute_reply":"2023-07-23T18:55:50.965775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# IDX2BIRD","metadata":{"execution":{"iopub.status.busy":"2023-07-23T18:55:50.970212Z","iopub.execute_input":"2023-07-23T18:55:50.971130Z","iopub.status.idle":"2023-07-23T18:55:50.975897Z","shell.execute_reply.started":"2023-07-23T18:55:50.971096Z","shell.execute_reply":"2023-07-23T18:55:50.974838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 10 frame version\nfilepath_list = [\n     \"../input/metadata-probability-v0525-2100/birdclef_resnest50_fold1_epoch_34_f1_val_04757_20210524185455.csv\"\n]\nprob_df = pd.concat([pd.read_csv(_) for _ in filepath_list])\nprob_df","metadata":{"papermill":{"duration":9.211879,"end_time":"2021-06-03T14:29:32.720537","exception":false,"start_time":"2021-06-03T14:29:23.508658","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:55:50.977753Z","iopub.execute_input":"2023-07-23T18:55:50.978101Z","iopub.status.idle":"2023-07-23T18:56:01.625241Z","shell.execute_reply.started":"2023-07-23T18:55:50.978067Z","shell.execute_reply":"2023-07-23T18:56:01.624294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TrainingConfig:\n    def __init__(self):\n        self.nocall_threshold:float=0.5\n        self.num_kfolds:int = 5\n        self.num_spieces:int = 397\n        self.num_candidates:int = 5\n        self.max_distance:int = 15 # 20\n        self.sampling_strategy:float = None # 1.0\n        self.random_state:int=777\n        self.num_prob:int = 6\n        self.use_to_birds=True\n        self.weights_filepath_dict = {\n            'lgbm':[f\"./lgbm_{kfold_index}.pkl\" for kfold_index in range(self.num_kfolds)],\n        }\n        \ntraining_config = TrainingConfig()","metadata":{"papermill":{"duration":0.033456,"end_time":"2021-06-03T14:29:32.779213","exception":false,"start_time":"2021-06-03T14:29:32.745757","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:56:01.626804Z","iopub.execute_input":"2023-07-23T18:56:01.627417Z","iopub.status.idle":"2023-07-23T18:56:01.634631Z","shell.execute_reply.started":"2023-07-23T18:56:01.627379Z","shell.execute_reply":"2023-07-23T18:56:01.633659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    def __init__(self):\n        self.num_kfolds:int = training_config.num_kfolds\n        self.num_spieces:int = training_config.num_spieces\n        self.num_candidates:int = training_config.num_candidates\n        self.max_distance:int = training_config.max_distance\n        self.nocall_threshold:float = training_config.nocall_threshold\n        self.num_prob:int = training_config.num_prob\n        # check F1 score without 3rd stage(table competition) \n        self.check_baseline:bool = True\n        # List of file paths of the models which are used when determining if the bird is acceptable.\n        self.weights_filepath_dict = training_config.weights_filepath_dict\n        # Weights for the models to predict the probability of each bird singing for each frame.\n        self.checkpoint_paths = [ \n            Path(\"../input/clefmodel/birdclef_resnest50_fold0_epoch_27_f1_val_05179_20210520120053.pth\"), # id36\n            Path(\"../input/clefmodel/birdclef_resnest50_fold0_epoch_13_f1_val_03502_20210522050604.pth\"), # id51\n            Path(\"../input/birdclef-groupby-author-05221040-728258/birdclef_resnest50_fold0_epoch_33_f1_val_03859_20210524151554.pth\"), # id58\n            Path(\"../input/birdclef-groupby-author-05221040-728258/birdclef_resnest50_fold1_epoch_34_f1_val_04757_20210524185455.pth\"), # id59\n            Path(\"../input/birdclef-groupby-author-05221040-728258/birdclef_resnest50_fold2_epoch_34_f1_val_05027_20210524223209.pth\"), # id60\n            Path(\"../input/birdclef-groupby-author-05221040-728258/birdclef_resnest50_fold3_epoch_20_f1_val_04299_20210525010703.pth\"), # id61\n            Path(\"../input/birdclef-groupby-author-05221040-728258/birdclef_resnest50_fold4_epoch_34_f1_val_05140_20210525074929.pth\"), # id62\n            Path(\"../input/clefmodel/resnest50_sr32000_d7_miixup-5.0_2ndlw-0.6_grouped-by-auther/birdclef_resnest50_fold0_epoch_78_f1_val_03658_20210528221629.pth\"), # id97\n            Path(\"../input/clefmodel/resnest50_sr32000_d7_miixup-5.0_2ndlw-0.6_grouped-by-auther/birdclef_resnest50_fold0_epoch_84_f1_val_03689_20210528225810.pth\"), # id97\n            Path(\"../input/clefmodel/resnest50_sr32000_d7_miixup-5.0_2ndlw-0.6_grouped-by-auther/birdclef_resnest50_fold1_epoch_27_f1_val_03942_20210529062427.pth\"), # id98\n        ]\n        # call probability of each bird for each sample used for candidate extraction (cache)\n        self.pred_filepath_list = [\n            self.get_prob_filepath_from_checkpoint(path) for path in self.checkpoint_paths\n        ]\n        \n    def get_prob_filepath_from_checkpoint(self, checkpoint_path:Path) -> str:\n        filename = f\"train_soundscape_labels_probabilitiy_%s.csv\" % checkpoint_path.stem\n        return filename\n\nconfig = Config()","metadata":{"papermill":{"duration":0.034411,"end_time":"2021-06-03T14:29:32.837976","exception":false,"start_time":"2021-06-03T14:29:32.803565","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:56:01.640277Z","iopub.execute_input":"2023-07-23T18:56:01.640570Z","iopub.status.idle":"2023-07-23T18:56:01.652959Z","shell.execute_reply.started":"2023-07-23T18:56:01.640524Z","shell.execute_reply":"2023-07-23T18:56:01.652045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make Candidate\nhelper functions for making candidates","metadata":{"papermill":{"duration":0.023979,"end_time":"2021-06-03T14:29:32.886126","exception":false,"start_time":"2021-06-03T14:29:32.862147","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def get_locations():\n    return [{\n        \"site\": \"COL\",\n        \"latitude\": 5.57,\n        \"longitude\": -75.85\n    }, {\n        \"site\": \"COR\",\n        \"latitude\": 10.12,\n        \"longitude\": -84.51\n    }, {\n        \"site\": \"SNE\",\n        \"latitude\": 38.49,\n        \"longitude\": -119.95\n    }, {\n        \"site\": \"SSW\",\n        \"latitude\": 42.47,\n        \"longitude\": -76.45\n    }]\n\n\ndef to_site(row, max_distance:int):\n    best = max_distance\n    answer = \"Other\"\n    for location in get_locations():\n        x = (row[\"latitude\"] - location[\"latitude\"])\n        y = (row[\"longitude\"] - location[\"longitude\"])\n        dist = (x**2 + y**2) ** 0.5\n        if dist < best:\n            best = dist\n            answer = location[\"site\"]\n    return answer\n\n\ndef to_latitude(site:str) -> str:\n    for location in get_locations():\n        if site == location[\"site\"]:\n            return location[\"latitude\"]\n    return -10000\n\n\ndef to_longitude(site:str) -> str:\n    for location in get_locations():\n        if site == location[\"site\"]:\n            return location[\"longitude\"]\n    return -10000\n\n\ndef to_birds(row, th:float) -> str:\n    if row[\"call_prob\"] < th:\n        return \"nocall\"\n    res = [row[\"primary_label\"]] + eval(row[\"secondary_labels\"])\n    return \" \".join(res)","metadata":{"papermill":{"duration":0.036791,"end_time":"2021-06-03T14:29:32.947315","exception":false,"start_time":"2021-06-03T14:29:32.910524","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:56:01.654300Z","iopub.execute_input":"2023-07-23T18:56:01.654876Z","iopub.status.idle":"2023-07-23T18:56:01.668325Z","shell.execute_reply.started":"2023-07-23T18:56:01.654842Z","shell.execute_reply":"2023-07-23T18:56:01.667202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"making candidates main func","metadata":{"papermill":{"duration":0.024216,"end_time":"2021-06-03T14:29:32.996454","exception":false,"start_time":"2021-06-03T14:29:32.972238","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def make_candidates(\n    prob_df:pd.DataFrame,\n    num_spieces:int,\n    num_candidates:int,\n    max_distance:int,\n    num_prob:int=6, # number of frames to be allocated for front and rear (if 3, then 3 for front, 3 for rear)\n    nocall_threshold:float=0.5,\n):\n    if \"author\" in prob_df.columns: # meta data (train_short_audio)\n        prob_df[\"birds\"] = prob_df.apply(\n            lambda row: to_birds(row, th=nocall_threshold),\n            axis=1\n        )\n        print(\"Candidate nocall ratio: %.4f\" % (prob_df[\"birds\"] == \"nocall\").mean())\n        prob_df[\"audio_id\"] = prob_df[\"filename\"].apply(\n            lambda _: int(_.replace(\"XC\", \"\").replace(\".ogg\", \"\"))\n        )\n        prob_df[\"row_id\"] = prob_df.apply(\n            lambda row: \"%s_%s_%s\" % (row[\"audio_id\"], row['site'], row[\"seconds\"]),\n            axis=1\n        )\n        prob_df[\"year\"] = prob_df[\"date\"].apply(lambda _: int(_.split(\"-\")[0]))\n        prob_df[\"month\"] = prob_df[\"date\"].apply(lambda _: int(_.split(\"-\")[1]))\n        prob_df[\"site\"] = prob_df.apply(\n            lambda row: to_site(row, max_distance),\n            axis=1\n        )\n    else:\n        prob_df[\"year\"] = prob_df[\"date\"].apply(lambda _: int(str(_)[:4]))\n        prob_df[\"month\"] = prob_df[\"date\"].apply(lambda _: int(str(_)[4:6]))\n        prob_df[\"latitude\"] = prob_df[\"site\"].apply(to_latitude)\n        prob_df[\"longitude\"] = prob_df[\"site\"].apply(to_longitude)\n        prob_df[\"row_id\"] = prob_df.apply(\n            lambda row: \"%s_%s_%s\" % (row[\"audio_id\"], row['site'], row[\"seconds\"]),\n            axis=1\n        )\n        \n    sum_prob_list = prob_df[BIRD_LIST].sum(axis=1).tolist()\n    mean_prob_list = prob_df[BIRD_LIST].mean(axis=1).tolist()\n    std_prob_list = prob_df[BIRD_LIST].std(axis=1).tolist()\n    max_prob_list = prob_df[BIRD_LIST].max(axis=1).tolist()\n    min_prob_list = prob_df[BIRD_LIST].min(axis=1).tolist()\n    skew_prob_list = prob_df[BIRD_LIST].skew(axis=1).tolist()\n    kurt_prob_list = prob_df[BIRD_LIST].kurt(axis=1).tolist()\n    \n    X = prob_df[BIRD_LIST].values\n    bird_ids_list = np.argsort(-X)[:,:num_candidates]\n    row_ids = prob_df[\"row_id\"].tolist()\n    rows = [i//num_candidates for i in range(len(bird_ids_list.flatten()))]\n    cols = bird_ids_list.flatten()\n    # What number?\n    ranks = [i%num_candidates for i in range(len(rows))]\n    probs_list = X[rows, cols]\n    D = {\n        \"row_id\": [row_ids[i] for i in rows],\n        \"rank\": ranks,\n        \"bird_id\": bird_ids_list.flatten(),\n        \"prob\": probs_list.flatten(),\n        \"sum_prob\": [sum_prob_list[i//num_candidates] for i in range(num_candidates*len(mean_prob_list))],\n        \"mean_prob\": [mean_prob_list[i//num_candidates] for i in range(num_candidates*len(mean_prob_list))],\n        \"std_prob\": [std_prob_list[i//num_candidates] for i in range(num_candidates*len(std_prob_list))],\n        \"max_prob\": [max_prob_list[i//num_candidates] for i in range(num_candidates*len(max_prob_list))],\n        \"min_prob\": [min_prob_list[i//num_candidates] for i in range(num_candidates*len(min_prob_list))],\n        \"skew_prob\": [skew_prob_list[i//num_candidates] for i in range(num_candidates*len(skew_prob_list))],\n        \"kurt_prob\": [kurt_prob_list[i//num_candidates] for i in range(num_candidates*len(kurt_prob_list))],\n    }\n    audio_ids = prob_df[\"audio_id\"].values[rows]\n    for diff in range(-num_prob, num_prob+1):\n        if diff == 0:\n            continue\n        neighbor_audio_ids = prob_df[\"audio_id\"].shift(diff).values[rows]\n        Y = prob_df[BIRD_LIST].shift(diff).values\n        c = f\"next{abs(diff)}_prob\" if diff < 0 else f\"prev{diff}_prob\"\n        c = c.replace(\"1_prob\", \"_prob\") # Fix next1_prob to next_prob\n        v = Y[rows, cols].flatten()\n        v[audio_ids != neighbor_audio_ids] = np.nan\n        D[c] = v\n\n    candidate_df = pd.DataFrame(D)\n    columns = [\n        \"row_id\",\n        \"site\",\n        \"year\",\n        \"month\",\n        \"audio_id\",\n        \"seconds\",\n        \"birds\",\n    ]\n    candidate_df = pd.merge(\n        candidate_df,\n        prob_df[columns],\n        how=\"left\",\n        on=\"row_id\"\n    )\n    candidate_df[\"target\"] = candidate_df.apply(\n        lambda row: IDX2BIRD[row[\"bird_id\"]] in set(str(row[\"birds\"]).split()),\n        axis=1\n    )\n    candidate_df[\"label\"] = candidate_df[\"bird_id\"].map(IDX2BIRD)\n    return candidate_df","metadata":{"papermill":{"duration":0.049883,"end_time":"2021-06-03T14:29:33.070891","exception":false,"start_time":"2021-06-03T14:29:33.021008","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:56:01.669860Z","iopub.execute_input":"2023-07-23T18:56:01.670194Z","iopub.status.idle":"2023-07-23T18:56:01.697219Z","shell.execute_reply.started":"2023-07-23T18:56:01.670162Z","shell.execute_reply":"2023-07-23T18:56:01.696284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Add Features\nhelper functions for adding features","metadata":{"papermill":{"duration":0.024856,"end_time":"2021-06-03T14:29:33.120328","exception":false,"start_time":"2021-06-03T14:29:33.095472","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def load_metadata():\n    meta_df = pd.read_csv(\"../input/birdclef-2021/train_metadata.csv\")\n    meta_df[\"id\"] = meta_df.index + 1\n    meta_df[\"year\"] = meta_df[\"date\"].apply(lambda _: _.split(\"-\")[0]).astype(int)\n    meta_df[\"month\"] = meta_df[\"date\"].apply(lambda _: _.split(\"-\")[1]).astype(int)\n    return meta_df\n\n\ndef to_zscore(row):\n    x = row[\"prob\"]\n    mu = row[\"prob_avg_in_same_audio\"]\n    sigma = row[\"prob_var_in_same_audio\"] ** 0.5\n    if sigma < 1e-6:\n        return 0\n    else:\n        return (x - mu) / sigma\n\n\ndef add_same_audio_features(\n    candidate_df:pd.DataFrame,\n    df:pd.DataFrame\n):\n    # Average probability per bird in the same audio\n    _gdf = df.groupby([\"audio_id\"], as_index=False).mean()[[\"audio_id\"] + BIRD_LIST]\n    _df = pd.melt(\n        _gdf,\n        id_vars=[\"audio_id\"]\n    ).rename(columns={\n        \"variable\": \"label\",\n        \"value\": \"prob_avg_in_same_audio\"\n    })\n    candidate_df = pd.merge(candidate_df, _df, how=\"left\", on=[\"audio_id\", \"label\"])\n    # Maximum value for each bird in the same audio\n    _gdf = df.groupby([\"audio_id\"], as_index=False).max()[[\"audio_id\"] + BIRD_LIST]\n    _df = pd.melt(\n        _gdf,\n        id_vars=[\"audio_id\"]\n    ).rename(columns={\n        \"variable\": \"label\",\n        \"value\": \"prob_max_in_same_audio\"\n    })\n    candidate_df = pd.merge(candidate_df, _df, how=\"left\", on=[\"audio_id\", \"label\"])\n    # Variance of each bird in the same audio\n    _gdf = df.groupby([\"audio_id\"], as_index=False).var()[[\"audio_id\"] + BIRD_LIST]\n    _df = pd.melt(\n        _gdf,\n        id_vars=[\"audio_id\"]\n    ).rename(columns={\n        \"variable\": \"label\",\n        \"value\": \"prob_var_in_same_audio\"\n    })\n    candidate_df = pd.merge(candidate_df, _df, how=\"left\", on=[\"audio_id\", \"label\"])\n    candidate_df[\"zscore_in_same_audio\"] = candidate_df.apply(to_zscore, axis=1)\n    return candidate_df","metadata":{"papermill":{"duration":0.039197,"end_time":"2021-06-03T14:29:33.18412","exception":false,"start_time":"2021-06-03T14:29:33.144923","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:56:01.699058Z","iopub.execute_input":"2023-07-23T18:56:01.699925Z","iopub.status.idle":"2023-07-23T18:56:01.713980Z","shell.execute_reply.started":"2023-07-23T18:56:01.699892Z","shell.execute_reply":"2023-07-23T18:56:01.713036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"making candidates main func","metadata":{}},{"cell_type":"code","source":"def add_features(\n    candidate_df:pd.DataFrame,\n    df:pd.DataFrame,\n    max_distance:int,\n):\n    meta_df = load_metadata()\n    # latitude & longitude\n    if not \"latitude\" in candidate_df.columns:\n        candidate_df[\"latitude\"] = candidate_df[\"site\"].apply(to_latitude)\n    if not \"longitude\" in candidate_df.columns:\n        candidate_df[\"longitude\"] = candidate_df[\"site\"].apply(to_longitude)\n    # Number of Appearances\n    candidate_df[\"num_appear\"] = candidate_df[\"label\"].map(\n        meta_df[\"primary_label\"].value_counts()\n    )\n    meta_df[\"site\"] = meta_df.apply(\n        lambda row: to_site(\n            row,\n            max_distance=max_distance\n        ),\n        axis=1\n    )\n\n    # Number of occurrences by region\n    _df = meta_df.groupby(\n        [\"primary_label\", \"site\"],\n        as_index=False\n    )[\"id\"].count().rename(\n        columns={\n            \"primary_label\": \"label\",\n            \"id\": \"site_num_appear\"\n        }\n    )\n    candidate_df = pd.merge(\n        candidate_df,\n        _df,\n        how=\"left\",\n        on=[\"label\", \"site\"]\n    )\n    candidate_df[\"site_appear_ratio\"] = candidate_df[\"site_num_appear\"] / candidate_df[\"num_appear\"]\n    # Seasonal statistics\n    _df = meta_df.groupby(\n        [\"primary_label\", \"month\"],\n        as_index=False\n    )[\"id\"].count().rename(\n        columns={\n            \"primary_label\": \"label\",\n            \"id\": \"month_num_appear\"\n        }\n    )\n    candidate_df = pd.merge(candidate_df, _df, how=\"left\", on=[\"label\", \"month\"])\n    candidate_df[\"month_appear_ratio\"] = candidate_df[\"month_num_appear\"] / candidate_df[\"num_appear\"]\n\n    candidate_df = add_same_audio_features(candidate_df, df)\n\n    # Correction of probability (all down)\n    candidate_df[\"prob / num_appear\"] = candidate_df[\"prob\"] / (candidate_df[\"num_appear\"].fillna(0) + 1)\n    candidate_df[\"prob / site_num_appear\"] = candidate_df[\"prob\"] / (candidate_df[\"site_num_appear\"].fillna(0) + 1)\n    candidate_df[\"prob * site_appear_ratio\"] = candidate_df[\"prob\"] * (candidate_df[\"site_appear_ratio\"].fillna(0) + 0.001)\n\n    # Amount of change from the previous and following frames\n    candidate_df[\"prob_avg\"] = candidate_df[[\"prev_prob\", \"prob\", \"next_prob\"]].mean(axis=1)\n    candidate_df[\"prob_diff\"] = candidate_df[\"prob\"] - candidate_df[\"prob_avg\"]\n    candidate_df[\"prob - prob_max_in_same_audio\"] = candidate_df[\"prob\"] - candidate_df[\"prob_max_in_same_audio\"]\n\n    # Average of back and forward frames\n\n    return candidate_df","metadata":{"papermill":{"duration":0.039005,"end_time":"2021-06-03T14:29:33.247442","exception":false,"start_time":"2021-06-03T14:29:33.208437","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:56:01.715092Z","iopub.execute_input":"2023-07-23T18:56:01.715356Z","iopub.status.idle":"2023-07-23T18:56:01.732108Z","shell.execute_reply.started":"2023-07-23T18:56:01.715333Z","shell.execute_reply":"2023-07-23T18:56:01.731498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Calculate birdcall probabilities (397dims per clip) from melspectrograms\nhelper functions for calculating birdcall probabilities","metadata":{"papermill":{"duration":0.024251,"end_time":"2021-06-03T14:29:33.296327","exception":false,"start_time":"2021-06-03T14:29:33.272076","status":"completed"},"tags":[]}},{"cell_type":"code","source":"DEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(\"DEVICE:\", DEVICE)\n\n\nclass MelSpecComputer:\n    def __init__(self, sr, n_mels, fmin, fmax, **kwargs):\n        self.sr = sr\n        self.n_mels = n_mels\n        self.fmin = fmin\n        self.fmax = fmax\n        kwargs[\"n_fft\"] = kwargs.get(\"n_fft\", self.sr//10)\n        kwargs[\"hop_length\"] = kwargs.get(\"hop_length\", self.sr//(10*4))\n        self.kwargs = kwargs\n\n    def __call__(self, y):\n\n        melspec = lb.feature.melspectrogram(\n            y=y, sr=self.sr, n_mels=self.n_mels, fmin=self.fmin, fmax=self.fmax, **self.kwargs,\n        )\n\n        melspec = lb.power_to_db(melspec).astype(np.float32)\n        return melspec\n\n    \ndef mono_to_color(X, eps=1e-6, mean=None, std=None):\n    mean = mean or X.mean()\n    std = std or X.std()\n    X = (X - mean) / (std + eps)\n\n    _min, _max = X.min(), X.max()\n\n    if (_max - _min) > eps:\n        V = np.clip(X, _min, _max)\n        V = 255 * (V - _min) / (_max - _min)\n        V = V.astype(np.uint8)\n    else:\n        V = np.zeros_like(X, dtype=np.uint8)\n\n    return V\n\n\ndef crop_or_pad(y, length):\n    if len(y) < length:\n        y = np.concatenate([y, length - np.zeros(len(y))])\n    elif len(y) > length:\n        y = y[:length]\n    return y\n\n\nclass BirdCLEFDataset(Dataset):\n    def __init__(self, data, sr=32_000, n_mels=128, fmin=0, fmax=None, duration=5, step=None, res_type=\"kaiser_fast\", resample=True):\n\n        self.data = data\n\n        self.sr = sr\n        self.n_mels = n_mels\n        self.fmin = fmin\n        self.fmax = fmax or self.sr//2\n\n        self.duration = duration\n        self.audio_length = self.duration*self.sr\n        self.step = step or self.audio_length\n\n        self.res_type = res_type\n        self.resample = resample\n\n        self.mel_spec_computer = MelSpecComputer(\n            sr=self.sr,\n            n_mels=self.n_mels,\n            fmin=self.fmin,\n            fmax=self.fmax\n        )\n        self.npy_save_root = Path(\"./data\")\n        \n        os.makedirs(self.npy_save_root, exist_ok=True)\n\n    def __len__(self):\n        return len(self.data)\n\n    @staticmethod\n    def normalize(image):\n        image = image.astype(\"float32\", copy=False) / 255.0\n        image = np.stack([image, image, image])\n        return image\n\n    def audio_to_image(self, audio):\n        melspec = self.mel_spec_computer(audio)\n        image = mono_to_color(melspec)\n        image = self.normalize(image)\n        return image\n\n    def read_file(self, filepath):\n#         print(\"path: \", filepath)\n        filename = filepath.stem\n        npy_path = self.npy_save_root / f\"{filename}.npy\"\n        \n        if not os.path.exists(npy_path):\n            audio, orig_sr = sf.read(filepath, dtype=\"float32\")\n\n            if self.resample and orig_sr != self.sr:\n                print(\"LB: \", audio, orig_sr, self.sr, self.res_type)\n                audio = lb.resample(y=audio, orig_sr = orig_sr, target_sr = self.sr,res_type =  self.res_type)\n\n            audios = []\n            for i in range(self.audio_length, len(audio) + self.step, self.step):\n#                 print(\"in loop: \", i)\n                start = max(0, i - self.audio_length)\n                end = start + self.audio_length\n                audios.append(audio[start:end])\n\n#             print(\"done loop\")\n            if len(audios[-1]) < self.audio_length:\n                audios = audios[:-1]\n\n#             print(\"adding to images\")\n#             print(\"audios size: \", len(audios))\n#             images = [self.audio_to_image(audio) for audio in audios]\n\n            images = []\n            for audio in audios:\n                x = self.audio_to_image(audio)\n#                 print(len(x))\n                images.append(x)\n            \n#             print(\"len: \", len(images))\n            images = np.stack(images)\n            \n#             print(\"saving images\")\n            np.save(str(npy_path), images)\n        return np.load(npy_path)\n\n    def __getitem__(self, idx):\n        return self.read_file(self.data.loc[idx, \"filepath\"])\n\n    \ndef load_net(checkpoint_path, num_classes=397):\n    net = resnest50(pretrained=False)\n    net.fc = nn.Linear(net.fc.in_features, num_classes)\n    dummy_device = torch.device(\"cpu\")\n    d = torch.load(checkpoint_path, map_location=dummy_device)\n    for key in list(d.keys()):\n        d[key.replace(\"model.\", \"\")] = d.pop(key)\n    net.load_state_dict(d)\n    net = net.to(DEVICE)\n    net = net.eval()\n    return net\n\n\n@torch.no_grad()\ndef get_thresh_preds(out, thresh=None):\n    thresh = thresh or THRESH\n    o = (-out).argsort(1)\n    npreds = (out > thresh).sum(1)\n    preds = []\n    for oo, npred in zip(o, npreds):\n        preds.append(oo[:npred].cpu().numpy().tolist())\n    return preds\n\n\ndef predict(nets, test_data, names=True):\n    preds = []\n    with torch.no_grad():\n        for idx in  tqdm(list(range(len(test_data)))):\n            xb = torch.from_numpy(test_data[idx]).to(DEVICE)\n            pred = 0.\n            for net in nets:\n                o = net(xb)\n                o = torch.sigmoid(o)\n                pred += o\n            pred /= len(nets)\n            if names:\n                pred = BIRD_LIST(get_thresh_preds(pred))\n\n            preds.append(pred)\n    return preds","metadata":{"papermill":{"duration":0.119816,"end_time":"2021-06-03T14:29:33.440692","exception":false,"start_time":"2021-06-03T14:29:33.320876","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:56:01.733660Z","iopub.execute_input":"2023-07-23T18:56:01.734537Z","iopub.status.idle":"2023-07-23T18:56:01.791263Z","shell.execute_reply.started":"2023-07-23T18:56:01.734500Z","shell.execute_reply":"2023-07-23T18:56:01.789350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"calculating birdcall probabilities main func","metadata":{"papermill":{"duration":0.037678,"end_time":"2021-06-03T14:29:33.521001","exception":false,"start_time":"2021-06-03T14:29:33.483323","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def objective(weights, model_outputs, target):\n#     print(model_outputs, weights)\n    blended_output = np.average(model_outputs, axis=0, weights=weights)\n    error = np.mean((blended_output - target) ** 2)\n    \n    return sum(error)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T18:56:01.792958Z","iopub.execute_input":"2023-07-23T18:56:01.793347Z","iopub.status.idle":"2023-07-23T18:56:01.805197Z","shell.execute_reply.started":"2023-07-23T18:56:01.793300Z","shell.execute_reply":"2023-07-23T18:56:01.804203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_prob_df(config, audio_paths, optimal_weights = []):\n    gc.collect()\n#     print('getting prob df for ', audio_paths)\n    data = pd.DataFrame(\n         [(path.stem, *path.stem.split(\"_\"), path) for path in Path(audio_paths).glob(\"*.ogg\")],\n        columns = [\"filename\", \"id\", \"site\", \"date\", \"filepath\"]\n    )\n    test_data = BirdCLEFDataset(data=data)\n#     print(\"get_prob_df, data: \", data)\n\n    for checkpoint_path in config.checkpoint_paths:\n        prob_filepath = config.get_prob_filepath_from_checkpoint(checkpoint_path)\n        if (not os.path.exists(prob_filepath)) or (TARGET_PATH is None):  # Always calculate when no cash is available or when submitting.\n#         if (True):  # Always calculate when no cash is available or when submitting.\n            nets = [load_net(checkpoint_path.as_posix())]\n            pred_probas = predict(nets, test_data, names=False)\n#             print(prob_filepath, 'probas: ', pred_probas)\n            if TARGET_PATH: # local                \n                df = pd.read_csv(TARGET_PATH, usecols=[\"row_id\", \"birds\"])\n            else: # when it is submission\n                if str(audio_paths)==\"../input/birdclef-2021/train_soundscapes\":\n                    print(audio_paths)\n                    df = pd.read_csv(Path(\"../input/birdclef-2021/train_soundscape_labels.csv\"), usecols=[\"row_id\", \"birds\"])\n                else:\n                    print(SAMPLE_SUB_PATH)\n                    df = pd.read_csv(SAMPLE_SUB_PATH, usecols=[\"row_id\", \"birds\"])\n#             print(\"get_prob_df, df: \",df)\n            df[\"audio_id\"] = df[\"row_id\"].apply(lambda _: int(_.split(\"_\")[0]))\n            df[\"site\"] = df[\"row_id\"].apply(lambda _: _.split(\"_\")[1])\n            df[\"seconds\"] = df[\"row_id\"].apply(lambda _: int(_.split(\"_\")[2]))\n            assert len(data) == len(pred_probas)\n            n = len(data)\n#             print(\"get_prob_df, n: \",n)\n            audio_id_to_date = {}\n            audio_id_to_site = {}\n            for filepath in audio_paths.glob(\"*.ogg\"):\n                audio_id, site, date = os.path.basename(filepath).replace(\".ogg\", \"\").split(\"_\")\n                audio_id = int(audio_id)\n                audio_id_to_date[audio_id] = date\n                audio_id_to_site[audio_id] = site\n            dfs = []\n            for i in range(n):\n                row = data.iloc[i]\n                audio_id = int(row[\"id\"])\n                pred = pred_probas[i]\n                _df = pd.DataFrame(pred.to(\"cpu\").numpy())\n                _df.columns = [IDX2BIRD[j] for j in range(_df.shape[1])]\n                _df[\"audio_id\"] = audio_id\n                _df[\"date\"] = audio_id_to_date[audio_id]\n                _df[\"site\"] = audio_id_to_site[audio_id]\n#                 _df[\"seconds\"] = [(j+1)*5 for j in range(len(_df))]\n                _df[\"seconds\"] = [(j+1)*5 for j in range(120)]\n                dfs.append(_df)\n            prob_df = pd.concat(dfs)\n            prob_df = pd.merge(prob_df, df, how=\"left\", on=[\"site\", \"audio_id\", \"seconds\"])\n            print(f\"Save to {prob_filepath}\")\n            prob_df.to_csv(prob_filepath, index=False)\n\n#     if (not weights == []):\n#         predictions = [pd.read_csv(\n#             config.get_prob_filepath_from_checkpoint(config.checkpoint_paths[0])\n#             )[BIRD_LIST]]\n#         if len(config.checkpoint_paths) > 1:\n#             columns = BIRD_LIST\n#         for checkpoint_path in config.checkpoint_paths[1:]:\n#             _df = pd.read_csv(\n#             config.get_prob_filepath_from_checkpoint(checkpoint_path)\n#         )\n#         predictions.append(_df[BIRD_LIST])\n        \n#         final_predictions = predictions[0] * weights[0]\n#         for i in range(len(predictions)):\n#             if i > 1:\n#                 final_predictions = predictions[i] * weights[i]\n            \n#         return final_predictions\n#     # Ensemble\n#     prob_df = pd.read_csv(\n#         config.get_prob_filepath_from_checkpoint(config.checkpoint_paths[0])\n#     )\n#     if len(config.checkpoint_paths) > 1:\n#         columns = BIRD_LIST\n#         for checkpoint_path in config.checkpoint_paths[1:]:\n#             _df = pd.read_csv(\n#                 config.get_prob_filepath_from_checkpoint(checkpoint_path)\n#             )\n#             print(_df.head(5))\n#             prob_df[columns] += _df[columns]\n#         prob_df[columns] /= len(config.checkpoint_paths)\n#     predictions = [pd.read_csv(\n#         config.get_prob_filepath_from_checkpoint(config.checkpoint_paths[0])\n#         )[BIRD_LIST]]\n#     if len(config.checkpoint_paths) > 1:\n#         columns = BIRD_LIST\n#         for checkpoint_path in config.checkpoint_paths[1:]:\n#             _df = pd.read_csv(\n#             config.get_prob_filepath_from_checkpoint(checkpoint_path)\n#         )\n#         predictions.append(_df[BIRD_LIST])\n\n    temp = pd.read_csv(\n        config.get_prob_filepath_from_checkpoint(config.checkpoint_paths[0])\n    )[BIRD_LIST]\n    # temp = temp.reshape(len(temp), 1)\n    predictions = [temp]\n    if len(config.checkpoint_paths) > 1:\n        columns = BIRD_LIST\n        for checkpoint_path in config.checkpoint_paths[1:]:\n            _df = pd.read_csv(\n                config.get_prob_filepath_from_checkpoint(checkpoint_path)\n            )\n    #         _df = _df.reshape(len(_df), 1)\n            predictions.append(_df[columns])\n    \n    # define the outputs of the 10 models\n    model_outputs = predictions\n    print('len: ', len(predictions))\n    \n    if (optimal_weights == []):\n        train_soundscape_label = pd.read_csv('/kaggle/input/birdclef-2021/train_soundscape_labels.csv')\n\n        y_train = train_soundscape_label['birds']\n\n        # Specify the number of rows\n        num_rows = len(y_train)\n\n        # Create an array of zeros\n        data = np.zeros((num_rows, len(BIRD_LIST)))\n\n        # Create a DataFrame\n        df = pd.DataFrame(data, columns=BIRD_LIST)\n\n        for i in range(len(y_train)):\n            if not y_train[i] == 'nocall':\n                lst = y_train[i].split(' ')\n                for col in lst:\n                    df.at[i, col] = 1\n\n#         # define the outputs of the 10 models\n#         model_outputs = predictions\n\n        # define the target variable\n        target = df\n\n        # define the initial weights\n        initial_weights = [0.001] * 10\n\n        # find the optimal weights\n        res = minimize(objective, initial_weights, args=(model_outputs, target), method='SLSQP', bounds=[(-10.0, 10.0)] * 10)\n\n        # get the optimal weights\n        optimal_weights = res.x\n        \n        print(\"optimal_weights calculated\")\n        \n    print(\"optimal_weights being used\")\n\n    # calculate the blended output using the optimal weights\n    blended_output = np.average(model_outputs, axis=0, weights=optimal_weights)\n    \n    prob_df = pd.read_csv(\n        config.get_prob_filepath_from_checkpoint(config.checkpoint_paths[0])\n    )\n    prob_df[BIRD_LIST] = blended_output\n\n    return prob_df","metadata":{"papermill":{"duration":0.044679,"end_time":"2021-06-03T14:29:33.590655","exception":false,"start_time":"2021-06-03T14:29:33.545976","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:56:01.806758Z","iopub.execute_input":"2023-07-23T18:56:01.807150Z","iopub.status.idle":"2023-07-23T18:56:01.833660Z","shell.execute_reply.started":"2023-07-23T18:56:01.807118Z","shell.execute_reply":"2023-07-23T18:56:01.832757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training (LightGBM)\nhelper function for training","metadata":{"papermill":{"duration":0.024599,"end_time":"2021-06-03T14:29:33.640248","exception":false,"start_time":"2021-06-03T14:29:33.615649","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def seed_everything(seed=1234):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True","metadata":{"papermill":{"duration":0.033361,"end_time":"2021-06-03T14:29:33.698423","exception":false,"start_time":"2021-06-03T14:29:33.665062","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:56:01.836441Z","iopub.execute_input":"2023-07-23T18:56:01.837547Z","iopub.status.idle":"2023-07-23T18:56:01.846639Z","shell.execute_reply.started":"2023-07-23T18:56:01.837514Z","shell.execute_reply":"2023-07-23T18:56:01.846016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"training main func","metadata":{"papermill":{"duration":0.024879,"end_time":"2021-06-03T14:29:33.748128","exception":false,"start_time":"2021-06-03T14:29:33.723249","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def train(\n    candidate_df:pd.DataFrame,\n    df:pd.DataFrame,\n    candidate_df_soundscapes:pd.DataFrame,\n    df_soundscapes:pd.DataFrame,\n    num_kfolds:int,\n    num_candidates:int,\n    verbose:bool=False,\n    sampling_strategy:float=1.0,\n    random_state:int=777,\n):\n    \n    seed_everything(random_state)\n    feature_names = get_feature_names()\n    if verbose:\n        print(\"features\", feature_names)\n        \n        \n    # short audio の  k fold\n    groups = candidate_df[\"audio_id\"]\n    kf = StratifiedGroupKFold(n_splits=num_kfolds) # When using lgbm_rank, it is necessary to use the data attached to each group, so don't shuffle them.\n    for kfold_index, (_, valid_index) in enumerate(kf.split(candidate_df[feature_names].values, candidate_df[\"target\"].values, groups)):\n        candidate_df.loc[valid_index, \"fold\"] = kfold_index\n                        \n    X = candidate_df[feature_names].values\n    y = candidate_df[\"target\"].values\n    oofa = np.zeros(len(candidate_df_soundscapes), dtype=np.float32)\n    \n    for kfold_index in range(num_kfolds):\n        print(f\"fold {kfold_index}\")\n        train_index = candidate_df[candidate_df[\"fold\"] != kfold_index].index\n        valid_index = candidate_df[candidate_df[\"fold\"] == kfold_index].index\n        X_train, y_train = X[train_index], y[train_index]\n        #X_valid, y_valid = X[valid_index], y[valid_index]\n        X_valid, y_valid = candidate_df_soundscapes[feature_names].values, candidate_df_soundscapes[\"target\"].values\n        \n        dtrain = lgb.Dataset(X_train, label=y_train)\n        dvalid = lgb.Dataset(X_valid, label=y_valid)\n        params = {\n            'objective': 'binary',\n            'metric': 'binary_logloss',\n            'device':'gpu',\n        }\n        model = lgb.train(\n            params,\n            dtrain,\n            valid_sets=dvalid,\n            num_boost_round=200,\n            early_stopping_rounds=20,\n            verbose_eval=20,\n        )\n        oofa += model.predict(X_valid.astype(np.float32))/num_kfolds\n        pickle.dump(model, open(f\"lgbm_{kfold_index}.pkl\", \"wb\"))\n        \n    def f(th):\n        _df = candidate_df_soundscapes[(oofa > th)]\n        if len(_df) == 0:\n            return 0\n        _gdf = _df.groupby(\n            [\"audio_id\", \"seconds\"],\n            as_index=False\n        )[\"label\"].apply(lambda _: \" \".join(_))\n        df2 = pd.merge(\n            df_soundscapes[[\"audio_id\", \"seconds\", \"birds\"]],\n            _gdf,\n            how=\"left\",\n            on=[\"audio_id\", \"seconds\"]\n        )\n        df2.loc[df2[\"label\"].isnull(), \"label\"] = \"nocall\"\n        return df2.apply(\n            lambda _: get_metrics(_[\"birds\"], _[\"label\"])[\"f1\"],\n            axis=1\n        ).mean()\n\n\n    print(\"-\"*30)\n    print(f\"#sound_scapes (len:{len(candidate_df_soundscapes)}) でのスコア\")\n    lb, ub = 0, 1\n    for k in range(30):\n        th1 = (2*lb + ub) / 3\n        th2 = (lb + 2*ub) / 3\n        if f(th1) < f(th2):\n            lb = th1\n        else:\n            ub = th2\n    th = (lb + ub) / 2\n    print(\"best th: %.4f\" % th)\n    print(\"best F1: %.4f\" % f(th))\n    if verbose:\n        y_soundscapes =  candidate_df_soundscapes[\"target\"].values\n        oof = (oofa > th).astype(int)\n        print(\"[details] Call or No call classirication\")\n        print(\"binary F1: %.4f\" % f1_score(y_soundscapes, oof))\n        print(\"gt positive ratio: %.4f\" % np.mean(y_soundscapes))\n        print(\"oof positive ratio: %.4f\" % np.mean(oof))\n        print(\"Accuracy: %.4f\" % accuracy_score(y_soundscapes, oof))\n        print(\"Recall: %.4f\" % recall_score(y_soundscapes, oof))\n        print(\"Precision: %.4f\" % precision_score(y_soundscapes, oof))\n    print(\"-\"*30)\n    print()","metadata":{"papermill":{"duration":0.046046,"end_time":"2021-06-03T14:29:33.819545","exception":false,"start_time":"2021-06-03T14:29:33.773499","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:56:01.848302Z","iopub.execute_input":"2023-07-23T18:56:01.848775Z","iopub.status.idle":"2023-07-23T18:56:01.869793Z","shell.execute_reply.started":"2023-07-23T18:56:01.848744Z","shell.execute_reply":"2023-07-23T18:56:01.868976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Optimize Threshold","metadata":{"papermill":{"duration":0.024325,"end_time":"2021-06-03T14:29:33.86862","exception":false,"start_time":"2021-06-03T14:29:33.844295","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"helper functions for optimizing threshold","metadata":{"papermill":{"duration":0.024112,"end_time":"2021-06-03T14:29:33.917545","exception":false,"start_time":"2021-06-03T14:29:33.893433","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def get_feature_names() -> List[str]:\n    return [\n        \"year\",\n        \"month\",\n        \"sum_prob\",\n        \"mean_prob\",\n        #\"std_prob\",\n        \"max_prob\",\n        #\"min_prob\",\n        #\"skew_prob\",\n        #\"kurt_prob\",\n        \"prev6_prob\",\n        \"prev5_prob\",\n        \"prev4_prob\",\n        \"prev3_prob\",\n        \"prev2_prob\",\n        \"prev_prob\",\n        \"prob\",\n        \"next_prob\",\n        \"next2_prob\",\n        \"next3_prob\",\n        \"next4_prob\",\n        \"next5_prob\",\n        \"next6_prob\",\n        \"rank\",\n        \"latitude\",\n        \"longitude\",\n        \"bird_id\", # +0.013700\n        \"seconds\", # -0.0050\n        \"num_appear\",\n        \"site_num_appear\",\n        \"site_appear_ratio\",\n        # \"prob / num_appear\", # -0.005\n        # \"prob / site_num_appear\", # -0.0102\n        # \"prob * site_appear_ratio\", # -0.0049\n        # \"prob_avg\", # -0.0155\n        \"prob_diff\", # 0.0082\n        # \"prob_avg_in_same_audio\", # -0.0256\n        # \"prob_max_in_same_audio\", # -0.0142\n        # \"prob_var_in_same_audio\", # -0.0304\n        # \"prob - prob_max_in_same_audio\", # -0.0069\n        # \"zscore_in_same_audio\", # -0.0110\n        # \"month_num_appear\", # 0.0164\n    ]\n\n\ndef get_metrics(s_true, s_pred):\n    s_true = set(s_true.split())\n    s_pred = set(s_pred.split())\n    n, n_true, n_pred = len(s_true.intersection(s_pred)), len(s_true), len(s_pred)\n    prec = n/n_pred\n    rec = n/n_true\n    f1 = 2*prec*rec/(prec + rec) if prec + rec else 0\n    return {\n        \"f1\": f1,\n        \"prec\": prec,\n        \"rec\": rec,\n        \"n_true\": n_true,\n        \"n_pred\": n_pred,\n        \"n\": n\n    }","metadata":{"papermill":{"duration":0.035858,"end_time":"2021-06-03T14:29:33.978091","exception":false,"start_time":"2021-06-03T14:29:33.942233","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:56:01.871415Z","iopub.execute_input":"2023-07-23T18:56:01.871877Z","iopub.status.idle":"2023-07-23T18:56:01.884682Z","shell.execute_reply.started":"2023-07-23T18:56:01.871844Z","shell.execute_reply":"2023-07-23T18:56:01.883804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"optimizing threshold main func","metadata":{"papermill":{"duration":0.02412,"end_time":"2021-06-03T14:29:34.026936","exception":false,"start_time":"2021-06-03T14:29:34.002816","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def optimize(\n    candidate_df:pd.DataFrame,\n    prob_df:pd.DataFrame,\n    num_kfolds:int,\n    weights_filepath_dict:dict,\n):\n    feature_names = get_feature_names()\n    X = candidate_df[feature_names].values\n    y_preda_list = []\n    for mode in weights_filepath_dict.keys():\n        fold_y_preda_list = []\n        for kfold_index in range(num_kfolds):\n            clf = pickle.load(open(weights_filepath_dict[mode][kfold_index], \"rb\"))\n            if mode=='lgbm':\n                y_preda = clf.predict(X.astype(np.float32), num_iteration=clf.best_iteration)\n            elif mode=='lgbm_rank':\n                y_preda = clf.predict(X.astype(np.float32), num_iteration=clf.best_iteration)\n            else:\n                y_preda = clf.predict_proba(X)[:,1]\n            fold_y_preda_list.append(y_preda)\n        mean_preda = np.mean(fold_y_preda_list, axis=0)\n        if mode=='lgbm_rank': # scaling\n            mean_preda = 1/(1 + np.exp(-mean_preda))\n        y_preda_list.append(mean_preda)\n    y_preda = np.mean(y_preda_list, axis=0)\n    candidate_df[\"y_preda\"] = y_preda\n    \n    def f(th):\n        _df = candidate_df[y_preda > th]\n        if len(_df) == 0:\n            return 0\n        _gdf = _df.groupby(\n            [\"audio_id\", \"seconds\"],\n            as_index=False\n        )[\"label\"].apply(\n            lambda _: \" \".join(_)\n        ).rename(columns={\n            \"label\": \"predictions\"\n        })\n        submission_df = pd.merge(\n            prob_df[[\"row_id\", \"audio_id\", \"seconds\", \"birds\"]],\n            _gdf,\n            how=\"left\",\n            on=[\"audio_id\", \"seconds\"]\n        )\n        submission_df.loc[submission_df[\"predictions\"].isnull(), \"predictions\"] = \"nocall\"\n#         print(f\"sub_df in f({th}): {submission_df}\")\n        return submission_df.apply(\n            lambda row: get_metrics(str(row[\"birds\"]), str(row[\"predictions\"]))[\"f1\"],\n            axis=1\n        ).mean()\n    \n    lb, ub = 0, 1\n    for k in range(30):\n        th1 = (lb * 2 + ub) / 3\n        th2 = (lb + ub * 2) / 3\n        if f(th1) < f(th2):\n            lb = th1\n        else:\n            ub = th2\n    th = (lb + ub) / 2\n    print(\"-\" * 30)\n    print(\"📌best threshold: %f\" % th)\n    print(\"best F1: %f\" % f(th))\n    \n    # nocall injection\n    _df = candidate_df[y_preda > th]\n    if len(_df) == 0:\n        return 0\n    _gdf = _df.groupby(\n            [\"audio_id\", \"seconds\"],\n            as_index=False\n    )[\"label\"].apply(\n        lambda _: \" \".join(_)\n    ).rename(columns={\n        \"label\": \"predictions\"\n    })\n    submission_df = pd.merge(\n            prob_df[[\"row_id\", \"audio_id\", \"seconds\", \"birds\"]],\n            _gdf,\n            how=\"left\",\n            on=[\"audio_id\", \"seconds\"]\n        )\n    submission_df.loc[submission_df[\"predictions\"].isnull(), \"predictions\"] = \"nocall\"\n\n    \n    _gdf2 = _df.groupby(\n            [\"audio_id\", \"seconds\"],\n            as_index=False\n    )[\"y_preda\"].sum()\n    submission_df = pd.merge(\n            submission_df,\n            _gdf2,\n            how=\"left\",\n            on=[\"audio_id\", \"seconds\"]\n        )\n    def f_nocall(nocall_th):\n        submission_df_with_nocall = submission_df.copy()\n        submission_df_with_nocall.loc[(submission_df_with_nocall[\"y_preda\"]<nocall_th) \n                                      & (submission_df_with_nocall[\"predictions\"]!=\"nocall\"), \"predictions\"] += \" nocall\"\n        return submission_df_with_nocall.apply(\n            lambda row: get_metrics(str(row[\"birds\"]), str(row[\"predictions\"]))[\"f1\"],\n\n            axis=1\n        ).mean()\n    lb, ub = 0, 1\n    for k in range(30):\n        th1 = (lb * 2 + ub) / 3\n        th2 = (lb + ub * 2) / 3\n        if f_nocall(th1) < f_nocall(th2):\n            lb = th1\n        else:\n            ub = th2\n    nocall_th = (lb + ub) / 2\n    print(\"-\" * 30)\n    print(\"## nocall injection\")\n    print(\"📌best nocall threshold: %f\" % nocall_th)\n    print(\"best F1: %f\" % f_nocall(nocall_th))\n    \n    return th, nocall_th","metadata":{"papermill":{"duration":0.047718,"end_time":"2021-06-03T14:29:34.098972","exception":false,"start_time":"2021-06-03T14:29:34.051254","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:56:01.886393Z","iopub.execute_input":"2023-07-23T18:56:01.887232Z","iopub.status.idle":"2023-07-23T18:56:01.911280Z","shell.execute_reply.started":"2023-07-23T18:56:01.887207Z","shell.execute_reply":"2023-07-23T18:56:01.910158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make Submission\nhelper functions for making submission","metadata":{"papermill":{"duration":0.024158,"end_time":"2021-06-03T14:29:34.14762","exception":false,"start_time":"2021-06-03T14:29:34.123462","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def calc_baseline(prob_df:pd.DataFrame):\n    \"\"\"Calculate the optimal value of F1 score simply based on the threshold alone (without 3rd stage)\"\"\"\n    columns = BIRD_LIST\n    X = prob_df[columns].values\n    def f(th):\n        n = X.shape[0]\n        pred_labels = [[] for i in range(n)]\n        I, J = np.where(X > th)\n        for i, j in zip(I, J):\n            pred_labels[i].append(IDX2BIRD[j])\n        for i in range(n):\n            if len(pred_labels[i]) == 0:\n                pred_labels[i] = \"nocall\"\n            else:\n                pred_labels[i] = \" \".join(pred_labels[i])\n        prob_df[\"pred_labels\"] = pred_labels\n        return prob_df.apply(\n            lambda _: get_metrics(str(_[\"birds\"]), str(_[\"pred_labels\"]))[\"f1\"],\n\n            axis=1\n        ).mean()\n\n    lb, ub = 0, 1\n    for k in range(30):\n        th1 = (2*lb + ub) / 3\n        th2 = (lb + 2*ub) / 3\n        if f(th1) < f(th2):\n            lb = th1\n        else:\n            ub = th2\n    th = (lb + ub) / 2\n    print(\"best th: %.4f\" % th)\n    print(\"best F1: %.4f\" % f(th))\n    return th","metadata":{"papermill":{"duration":0.036143,"end_time":"2021-06-03T14:29:34.208136","exception":false,"start_time":"2021-06-03T14:29:34.171993","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:56:01.913843Z","iopub.execute_input":"2023-07-23T18:56:01.914100Z","iopub.status.idle":"2023-07-23T18:56:01.927044Z","shell.execute_reply.started":"2023-07-23T18:56:01.914078Z","shell.execute_reply":"2023-07-23T18:56:01.926091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"make_submission main func","metadata":{"papermill":{"duration":0.024182,"end_time":"2021-06-03T14:29:34.256677","exception":false,"start_time":"2021-06-03T14:29:34.232495","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def make_submission(\n    candidate_df:pd.DataFrame,\n    prob_df:pd.DataFrame,\n    num_kfolds:int,\n    th:float,\n    nocall_th:float,\n    weights_filepath_dict:dict,\n    max_distance:int\n):\n    feature_names = get_feature_names()\n    X = candidate_df[feature_names].values\n    y_preda_list = []\n    for mode in weights_filepath_dict.keys():\n        fold_y_preda_list = []\n        for kfold_index in range(num_kfolds):\n            clf = pickle.load(open(weights_filepath_dict[mode][kfold_index], \"rb\"))\n            if mode=='lgbm':\n                y_preda = clf.predict(X.astype(np.float32), num_iteration=clf.best_iteration)\n            elif mode=='lgbm_rank':\n                y_preda = clf.predict(X.astype(np.float32), num_iteration=clf.best_iteration)\n            else:\n                y_preda = clf.predict_proba(X)[:,1]\n            fold_y_preda_list.append(y_preda)\n        mean_preda = np.mean(fold_y_preda_list, axis=0)\n        if mode=='lgbm_rank':  # scaling\n            mean_preda = 1/(1 + np.exp(-mean_preda))\n        y_preda_list.append(mean_preda)\n    y_preda = np.mean(y_preda_list, axis=0)\n    candidate_df[\"y_preda\"] = y_preda\n#     print(\"candidate_df1: \", candidate_df)\n    \n    _df = candidate_df[y_preda > th]\n    _gdf = _df.groupby(\n        [\"audio_id\", \"seconds\"],\n        as_index=False\n    )[\"label\"].apply(\n        lambda _: \" \".join(_)\n    ).rename(columns={\n        \"label\": \"predictions\"\n    })\n    submission_df = pd.merge(\n            prob_df[[\"row_id\", \"audio_id\", \"seconds\", \"birds\"]],\n            _gdf,\n            how=\"left\",\n            on=[\"audio_id\", \"seconds\"]\n        )\n    submission_df.loc[submission_df[\"predictions\"].isnull(), \"predictions\"] = \"nocall\"\n    print(\"submission_df: \", submission_df)\n    if TARGET_PATH:\n        score_df = pd.DataFrame(\n            submission_df.apply(\n                lambda row: get_metrics(row[\"birds\"], row[\"predictions\"]),\n                axis=1\n            ).tolist()\n        )\n        print(\"-\" * 30)\n        print(\"BEFORE nocall injection\")\n        print(\"CV score on a trained model with train_short_audio (to check the model behavior)\")\n        print(\"F1: %.4f\" % score_df[\"f1\"].mean())\n        print(\"Recall: %.4f\" % score_df[\"rec\"].mean())\n        print(\"Precision: %.4f\" % score_df[\"prec\"].mean())\n        \n    # nocall injection\n    _gdf2 = _df.groupby(\n            [\"audio_id\", \"seconds\"],\n            as_index=False\n    )[\"y_preda\"].sum()\n    submission_df = pd.merge(\n            submission_df,\n            _gdf2,\n            how=\"left\",\n            on=[\"audio_id\", \"seconds\"]\n        )\n    submission_df.loc[(submission_df[\"y_preda\"] < nocall_th) \n                     & (submission_df[\"predictions\"]!=\"nocall\"), \"predictions\"] += \" nocall\"\n    print(\"submission_df: \", submission_df)\n    if TARGET_PATH:\n        score_df = pd.DataFrame(\n            submission_df.apply(\n                lambda row: get_metrics(row[\"birds\"], row[\"predictions\"]),\n                axis=1\n            ).tolist()\n        )\n        print(\"-\" * 30)\n        print(\"AFTER nocall injection\")\n        print(\"CV score on a trained model with train_short_audio (to check the model behavior)\")\n        print(\"F1: %.4f\" % score_df[\"f1\"].mean())\n        print(\"Recall: %.4f\" % score_df[\"rec\"].mean())\n        print(\"Precision: %.4f\" % score_df[\"prec\"].mean())\n                \n    return submission_df[[\"row_id\", \"predictions\"]].rename(columns={\n        \"predictions\": \"birds\"\n    })","metadata":{"papermill":{"duration":0.043061,"end_time":"2021-06-03T14:29:34.324188","exception":false,"start_time":"2021-06-03T14:29:34.281127","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:56:01.928740Z","iopub.execute_input":"2023-07-23T18:56:01.929107Z","iopub.status.idle":"2023-07-23T18:56:01.948527Z","shell.execute_reply.started":"2023-07-23T18:56:01.929077Z","shell.execute_reply":"2023-07-23T18:56:01.947636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Main","metadata":{"papermill":{"duration":0.024431,"end_time":"2021-06-03T14:29:34.372753","exception":false,"start_time":"2021-06-03T14:29:34.348322","status":"completed"},"tags":[]}},{"cell_type":"code","source":"####################################################\n# Train the model for the table competition part.\n####################################################\n\nTEST_AUDIO_ROOT = Path(\"../input/birdclef-2021/test_soundscapes\")\n# TEST_AUDIO_ROOT = Path(\"/kaggle/input/50inputtest\")\nSAMPLE_SUB_PATH = \"../input/birdclef-2021/sample_submission.csv\"\nTARGET_PATH = None\n\nif not len(list(TEST_AUDIO_ROOT.glob(\"*.ogg\"))): # If there isn't any sound source for testing, call for train_soundscapes\n    TEST_AUDIO_ROOT = Path(\"../input/birdclef-2021/train_soundscapes\")\n    SAMPLE_SUB_PATH = None\n    # SAMPLE_SUB_PATH = \"../input/birdclef-2021/sample_submission.csv\"\n    TARGET_PATH = Path(\"../input/birdclef-2021/train_soundscape_labels.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-07-23T18:56:01.950058Z","iopub.execute_input":"2023-07-23T18:56:01.950802Z","iopub.status.idle":"2023-07-23T18:56:02.011383Z","shell.execute_reply.started":"2023-07-23T18:56:01.950770Z","shell.execute_reply":"2023-07-23T18:56:02.010559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# short audio\n# Exclude items that do not need to be trained\nif not \"site\" in prob_df.columns:\n    prob_df[\"site\"] = prob_df.apply(\n        lambda row: to_site(\n            row,\n            max_distance=training_config.max_distance\n        ),\n        axis=1\n    )\n    print(\"[exclude other]before: %d\" % len(prob_df))\n    prob_df = prob_df[prob_df[\"site\"] != \"Other\"].reset_index(drop=True)\n    print(\"[exclude other]after: %d\" % len(prob_df))\n\n\ncandidate_df = make_candidates(\n    prob_df,\n    num_spieces=training_config.num_spieces,\n    num_candidates=training_config.num_candidates,\n    max_distance=training_config.max_distance\n)\ncandidate_df = add_features(\n    candidate_df,\n    prob_df,\n    max_distance=training_config.max_distance\n)\n   \n","metadata":{"papermill":{"duration":254.814107,"end_time":"2021-06-03T14:33:49.21158","exception":false,"start_time":"2021-06-03T14:29:34.397473","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T18:56:02.013182Z","iopub.execute_input":"2023-07-23T18:56:02.013778Z","iopub.status.idle":"2023-07-23T18:56:53.217709Z","shell.execute_reply.started":"2023-07-23T18:56:02.013745Z","shell.execute_reply":"2023-07-23T18:56:53.216634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# soundscapes\noptimal_weights = []\nprob_df_soundscapes = get_prob_df(config,Path(\"../input/birdclef-2021/train_soundscapes\"), optimal_weights)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-23T18:56:53.219222Z","iopub.execute_input":"2023-07-23T18:56:53.219616Z","iopub.status.idle":"2023-07-23T19:00:00.871272Z","shell.execute_reply.started":"2023-07-23T18:56:53.219565Z","shell.execute_reply":"2023-07-23T19:00:00.870225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prob_df","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:00:00.872864Z","iopub.execute_input":"2023-07-23T19:00:00.873201Z","iopub.status.idle":"2023-07-23T19:00:00.920392Z","shell.execute_reply.started":"2023-07-23T19:00:00.873167Z","shell.execute_reply":"2023-07-23T19:00:00.919349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prob_df_soundscapes","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:00:00.927533Z","iopub.execute_input":"2023-07-23T19:00:00.927832Z","iopub.status.idle":"2023-07-23T19:00:00.962243Z","shell.execute_reply.started":"2023-07-23T19:00:00.927806Z","shell.execute_reply":"2023-07-23T19:00:00.961409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"candidate_df_soundscapes = make_candidates(\n    prob_df_soundscapes,\n    num_spieces=training_config.num_spieces,\n    num_candidates=training_config.num_candidates,\n    max_distance=training_config.max_distance\n)\n\ncandidate_df_soundscapes = add_features(\n    candidate_df_soundscapes,\n    prob_df_soundscapes,\n    max_distance=training_config.max_distance\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:00:00.963434Z","iopub.execute_input":"2023-07-23T19:00:00.964834Z","iopub.status.idle":"2023-07-23T19:00:05.504286Z","shell.execute_reply.started":"2023-07-23T19:00:00.964792Z","shell.execute_reply":"2023-07-23T19:00:05.503283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"candidate_df_soundscapes","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:00:05.510140Z","iopub.execute_input":"2023-07-23T19:00:05.514186Z","iopub.status.idle":"2023-07-23T19:00:05.561256Z","shell.execute_reply.started":"2023-07-23T19:00:05.514140Z","shell.execute_reply":"2023-07-23T19:00:05.560463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for mode in config.weights_filepath_dict.keys():\n    print(f'training of {mode} is going...')\n    train(\n        candidate_df,\n        prob_df,\n        candidate_df_soundscapes,\n        prob_df_soundscapes,\n        num_kfolds=training_config.num_kfolds,\n        num_candidates=training_config.num_candidates,\n        verbose=True,\n    )","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:00:05.566110Z","iopub.execute_input":"2023-07-23T19:00:05.568991Z","iopub.status.idle":"2023-07-23T19:00:32.953712Z","shell.execute_reply.started":"2023-07-23T19:00:05.568957Z","shell.execute_reply":"2023-07-23T19:00:32.952765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_th, best_nocall_th = optimize(\n    candidate_df,\n    prob_df,\n    num_kfolds=config.num_kfolds,\n    weights_filepath_dict=config.weights_filepath_dict,\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:00:32.955268Z","iopub.execute_input":"2023-07-23T19:00:32.955564Z","iopub.status.idle":"2023-07-23T19:04:12.887882Z","shell.execute_reply.started":"2023-07-23T19:00:32.955540Z","shell.execute_reply":"2023-07-23T19:04:12.886817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# train ensemble model","metadata":{}},{"cell_type":"code","source":"# # Step 2: Create an empty ensemble dataframe\n# ensemble_df = pd.DataFrame()\n\n# # Step 3: Define the weight selection model\n# weight_model = LinearRegression()\n\n# # Step 4: Prepare the training data\n# X = pd.concat([pred_model1, pred_model2, ...])  # Combine all predictions\n# y = pd.read_csv('true_values.csv')  # True values of the target variable\n# X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2)\n\n# # Step 5: Train the weight selection model\n# weight_model.fit(X_train, y_train)\n\n# # Step 6: Evaluate the weight selection model\n# score = weight_model.score(X_val, y_val)\n# print(\"Weight selection model score:\", score)\n\n# # Step 7: Obtain the weights\n# weights = weight_model.predict(X)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:04:12.889521Z","iopub.execute_input":"2023-07-23T19:04:12.890853Z","iopub.status.idle":"2023-07-23T19:04:12.896496Z","shell.execute_reply.started":"2023-07-23T19:04:12.890818Z","shell.execute_reply":"2023-07-23T19:04:12.895333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:04:12.898009Z","iopub.execute_input":"2023-07-23T19:04:12.898371Z","iopub.status.idle":"2023-07-23T19:04:12.910467Z","shell.execute_reply.started":"2023-07-23T19:04:12.898337Z","shell.execute_reply":"2023-07-23T19:04:12.909529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = pd.read_csv(\n    config.get_prob_filepath_from_checkpoint(config.checkpoint_paths[0])\n)[BIRD_LIST]\n# temp = temp.reshape(len(temp), 1)\npredictions = [temp]\nif len(config.checkpoint_paths) > 1:\n    columns = BIRD_LIST\n    for checkpoint_path in config.checkpoint_paths[1:]:\n        _df = pd.read_csv(\n            config.get_prob_filepath_from_checkpoint(checkpoint_path)\n        )\n#         _df = _df.reshape(len(_df), 1)\n        predictions.append(_df[columns])","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:04:12.911756Z","iopub.execute_input":"2023-07-23T19:04:12.912184Z","iopub.status.idle":"2023-07-23T19:04:14.854086Z","shell.execute_reply.started":"2023-07-23T19:04:12.912125Z","shell.execute_reply":"2023-07-23T19:04:14.853099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions[0]","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:04:14.855368Z","iopub.execute_input":"2023-07-23T19:04:14.855731Z","iopub.status.idle":"2023-07-23T19:04:14.887702Z","shell.execute_reply.started":"2023-07-23T19:04:14.855699Z","shell.execute_reply":"2023-07-23T19:04:14.886655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_soundscape_label = pd.read_csv('/kaggle/input/birdclef-2021/train_soundscape_labels.csv')\n\ny_train = train_soundscape_label['birds']\n\n# Specify the number of rows\nnum_rows = len(y_train)\n\n# Create an array of zeros\ndata = np.zeros((num_rows, len(BIRD_LIST)))\n\n# Create a DataFrame\ndf = pd.DataFrame(data, columns=BIRD_LIST)\n\nfor i in range(len(y_train)):\n    if not y_train[i] == 'nocall':\n        lst = y_train[i].split(' ')\n        for col in lst:\n            df.at[i, col] = 1\n","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:04:14.889180Z","iopub.execute_input":"2023-07-23T19:04:14.889523Z","iopub.status.idle":"2023-07-23T19:04:14.950182Z","shell.execute_reply.started":"2023-07-23T19:04:14.889491Z","shell.execute_reply":"2023-07-23T19:04:14.949339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare data for training the weight learning model\nX_blend = np.column_stack(predictions)\ny_blend = df","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:04:14.951421Z","iopub.execute_input":"2023-07-23T19:04:14.951840Z","iopub.status.idle":"2023-07-23T19:04:14.979968Z","shell.execute_reply.started":"2023-07-23T19:04:14.951806Z","shell.execute_reply":"2023-07-23T19:04:14.978979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_blend","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:04:14.981402Z","iopub.execute_input":"2023-07-23T19:04:14.981974Z","iopub.status.idle":"2023-07-23T19:04:15.027061Z","shell.execute_reply.started":"2023-07-23T19:04:14.981937Z","shell.execute_reply":"2023-07-23T19:04:15.026021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.optimize import minimize\n\n# define the objective function\ndef objective(weights, model_outputs, target):\n    blended_output = np.average(model_outputs, axis=0, weights=weights)\n    error = np.mean((blended_output - target) ** 2)\n    \n    return sum(error)\n\n# define the outputs of the 10 models\nmodel_outputs = predictions\n\n# define the target variable\ntarget = y_blend\n\n# define the initial weights\ninitial_weights = [0.001] * 10\n\n# find the optimal weights 1\nres = minimize(objective, initial_weights, args=(model_outputs, target), method='SLSQP', bounds=[(0.0, 10.0)] * 10)\n\n# get the optimal weights\noptimal_weights = res.x\n\n# calculate the blended output using the optimal weights\nblended_output = np.average(model_outputs, axis=0, weights=optimal_weights)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:04:15.028376Z","iopub.execute_input":"2023-07-23T19:04:15.028780Z","iopub.status.idle":"2023-07-23T19:04:29.237439Z","shell.execute_reply.started":"2023-07-23T19:04:15.028747Z","shell.execute_reply":"2023-07-23T19:04:29.236384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"blended_output.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:04:29.239128Z","iopub.execute_input":"2023-07-23T19:04:29.239528Z","iopub.status.idle":"2023-07-23T19:04:29.247982Z","shell.execute_reply.started":"2023-07-23T19:04:29.239490Z","shell.execute_reply":"2023-07-23T19:04:29.246977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# # Train weight learning model\n# weight_model = LinearRegression()\n# weight_model.fit(X_blend, y_blend)\n\n# # Obtain weights from the weight learning model\n# weights = weight_model.coef_\n\n# # Combine predictions using learned weights\n# # final_predictions = predictions1 * weights[0] + predictions2 * weights[1]\n","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:04:29.249577Z","iopub.execute_input":"2023-07-23T19:04:29.250703Z","iopub.status.idle":"2023-07-23T19:04:29.260314Z","shell.execute_reply.started":"2023-07-23T19:04:29.250664Z","shell.execute_reply":"2023-07-23T19:04:29.259259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# weights","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:04:29.261860Z","iopub.execute_input":"2023-07-23T19:04:29.263046Z","iopub.status.idle":"2023-07-23T19:04:29.271241Z","shell.execute_reply.started":"2023-07-23T19:04:29.263007Z","shell.execute_reply":"2023-07-23T19:04:29.270141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"######################################################\n# for submission\n######################################################\nprob_df = get_prob_df(config, TEST_AUDIO_ROOT, optimal_weights)\n# prob_df = get_prob_df(config, '/kaggle/input/new100inputtest')\n\n# prob_df","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:04:29.272833Z","iopub.execute_input":"2023-07-23T19:04:29.273728Z","iopub.status.idle":"2023-07-23T19:04:45.210616Z","shell.execute_reply.started":"2023-07-23T19:04:29.273690Z","shell.execute_reply":"2023-07-23T19:04:45.209263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# candidate extraction\ncandidate_df = make_candidates(\n    prob_df,\n    num_spieces=config.num_spieces,\n    num_candidates=config.num_candidates,\n    max_distance=config.max_distance,\n    num_prob=config.num_prob,\n    nocall_threshold=config.nocall_threshold\n)\n# add features\ncandidate_df = add_features(\n    candidate_df,\n    prob_df,\n    max_distance=config.max_distance\n)\n    \n# best_th = None\n# best_nocall_th = None\n\n\n\nif config.check_baseline:\n    print(\"-\" * 30)\n    print(\"check F1 score without 3rd stage(table competition)\")\n    calc_baseline(prob_df)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:04:45.211737Z","iopub.status.idle":"2023-07-23T19:04:45.212720Z","shell.execute_reply.started":"2023-07-23T19:04:45.212431Z","shell.execute_reply":"2023-07-23T19:04:45.212463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prob_df","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:04:45.214322Z","iopub.status.idle":"2023-07-23T19:04:45.214805Z","shell.execute_reply.started":"2023-07-23T19:04:45.214555Z","shell.execute_reply":"2023-07-23T19:04:45.214577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# candidate_df","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:04:45.216214Z","iopub.status.idle":"2023-07-23T19:04:45.217024Z","shell.execute_reply.started":"2023-07-23T19:04:45.216783Z","shell.execute_reply":"2023-07-23T19:04:45.216808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# best_th, best_nocall_th","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:04:45.218300Z","iopub.status.idle":"2023-07-23T19:04:45.219501Z","shell.execute_reply.started":"2023-07-23T19:04:45.219254Z","shell.execute_reply":"2023-07-23T19:04:45.219279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = make_submission(\n    candidate_df,\n    prob_df,\n    num_kfolds=config.num_kfolds,\n    th=best_th,\n    nocall_th=best_nocall_th,\n    weights_filepath_dict=config.weights_filepath_dict,\n    max_distance=config.max_distance\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T19:04:45.220977Z","iopub.status.idle":"2023-07-23T19:04:45.221802Z","shell.execute_reply.started":"2023-07-23T19:04:45.221526Z","shell.execute_reply":"2023-07-23T19:04:45.221550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv(\"submission.csv\", index=False)","metadata":{"papermill":{"duration":0.056387,"end_time":"2021-06-03T14:33:49.310563","exception":false,"start_time":"2021-06-03T14:33:49.254176","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T19:04:45.223262Z","iopub.status.idle":"2023-07-23T19:04:45.224098Z","shell.execute_reply.started":"2023-07-23T19:04:45.223851Z","shell.execute_reply":"2023-07-23T19:04:45.223876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !cp '/kaggle/input/birdclef-2021/sample_submission.csv' ./","metadata":{"papermill":{"duration":0.042051,"end_time":"2021-06-03T14:33:49.394309","exception":false,"start_time":"2021-06-03T14:33:49.352258","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-07-23T19:04:45.225552Z","iopub.status.idle":"2023-07-23T19:04:45.226390Z","shell.execute_reply.started":"2023-07-23T19:04:45.226144Z","shell.execute_reply":"2023-07-23T19:04:45.226169Z"},"trusted":true},"execution_count":null,"outputs":[]}]}