{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\n\n# Checking if we work with public or private test set\ntestdir = os.listdir('../input/birdclef-2021/test_soundscapes/')\npath = '../input/birdclef-2021/train_soundscapes/'\nfor t in testdir:\n    if \".ogg\" in t:\n        path = '../input/birdclef-2021/test_soundscapes/'\n\n# Private test set has non-sound files, filter them out\nfiles = os.listdir(path)\nfiles2 = []\nfor f in files:\n    if \".ogg\" in f:\n        files2.append(f)\n\n# For time (360 days) or longitude (360 degrees) distance\ndef long_dist(a, b):\n    if np.abs(a - b) > 180:\n        return 360 - np.abs(a - b)\n    else:\n        return np.abs(a - b)\n\n# Reading train metadata\ndf = pd.read_csv(\"../input/birdclef-2021/train_metadata.csv\")\ndf1 = df[['primary_label', 'latitude', 'longitude', 'date']]\ndf1[['latitude', 'longitude']] = df1[['latitude', 'longitude']].apply(pd.to_numeric)\n\ngeotime_dict = {}\nfor i in range(0, df1.shape[0]):\n    row = df1.iloc[i]\n    name = row[0]\n    lat = row[1]\n    long = row[2]\n    date = row[3]\n    day_of_year = int(str.split(date, \"-\")[1]) * 30 + int(str.split(date, \"-\")[2])\n\n    if name not in geotime_dict:\n        geotime_dict[name] = []\n\n    point_loc = [lat, long, day_of_year]\n    geotime_dict[name].append(point_loc)\n\n# Soft lock - no hard threshold, just declining probabilities\ndef soft_lock(geotime, lock_begin, lock_end):\n    if geotime < lock_begin:\n        geotime = 1\n    elif geotime < lock_end:\n        geotime = (lock_end - geotime) / (lock_end - lock_begin)\n    else:\n        geotime = 0\n    geotime = np.round(geotime, 3)\n    return geotime\n\ndef get_birds_possibility(day, latitude, longitude, percent_cases, day_to_degree, geotime_lock_begin, geotime_lock_end):\n    geotime_array = []\n    for bird in geotime_dict:\n        # Populate array with all train recordings for a bird\n        geotime = []\n        for loc in geotime_dict[bird]:\n            lat_diff = np.abs(loc[0] - latitude)\n            long_diff = long_dist(loc[1], longitude)\n            time_diff = long_dist(loc[2], day) * day_to_degree\n\n            # 3d distance with scaled time (~1.8 degrees/month by default) as third dimension\n            geotime_diff = np.sqrt(lat_diff * lat_diff + long_diff * long_diff + time_diff * time_diff)\n            geotime.append(geotime_diff)\n\n        geotime = sorted(geotime)\n\n        # Get distance of n% closest bird\n        dist_len = int(np.ceil(len(geotime) * percent_cases / 100))\n        geotime = geotime[0:dist_len]\n        geotime = geotime[-1]\n\n        geotime = soft_lock(geotime, geotime_lock_begin, geotime_lock_end)\n        geotime_array.append(geotime)\n\n    return geotime_array\n\n# Get recording site and date\ndef get_lat_long_time(f):\n    fdate = f.split(\"_\")[2]\n    fdate = fdate.split(\".\")[0]\n    month = int(fdate[4:6])\n    day = int(fdate[6:8])\n    day_of_year = month * 30 + day\n\n    point_loc = {'COL': [5.57, -75.85], 'COR': [10.12, -84.51], 'SNE': [38.49, -119.95], 'SSW': [42.47, -76.45]}\n    scape_lat = 0\n    scape_long = 0\n    if \"COL\" in f:\n        scape_lat = point_loc['COL'][0]\n        scape_long = point_loc['COL'][1]\n    if \"COR\" in f:\n        scape_lat = point_loc['COR'][0]\n        scape_long = point_loc['COR'][1]\n    if \"SNE\" in f:\n        scape_lat = point_loc['SNE'][0]\n        scape_long = point_loc['SNE'][1]\n    if \"SSW\" in f:\n        scape_lat = point_loc['SSW'][0]\n        scape_long = point_loc['SSW'][1]\n\n    return day_of_year, scape_lat, scape_long\n\ngeotimelock_perfile = {}\nfor f in files2:\n    day, scape_lat, scape_long = get_lat_long_time(f)\n    \n    # Adjusted for my models by validating on train soundscapes;\n    geotime_lock_array = get_birds_possibility(day, scape_lat, scape_long, 4, 0.06, 5, 7)\n    \n    # Optional adjustment that might depend on your model:\n    # Another post-processing step that i performed was to check if any bird had a lot of false positives, by validating on short audio clips\n    # Majority of them were okay (~30% ratio of false positive/true positive avg), but grhowl (and only grhowl) had more false positives than true positives\n    # Any sufficiently noisy recording was predicted as grhowl. By deleting grhowl CV improved by 0.016\n    # grhowl is evil. Woo-Hoo. Hoo. Hoo.\n    \n    # geotime_lock_array[164] = 0\n    \n    geotimelock_perfile[f] = geotime_lock_array\n    print(\"For soundscape \" + f + \" geotime lock is: \" + str(geotime_lock_array))\n\n# Usage instructions\n\n# ...Somewhere later in actual predicting code of your submission\n# Sigmoid and 0..1 values are expected for answers\n# answers is 397-long array of your model/ensemble prediction on a 5-second soundscape slice or equivalent\n\n# answers = answers * geotimelock_perfile[current_file]\n\n# Warning: Using such post-processing might drastically change optimal threshold.\n# Peak validation F1 on train soundscapes on me:\n#   No geotime lock and no grhowl deletion: 0.465/0.3, 0.625/0.5, 0.668/0.7, peak 0.67/0.74\n#   No geotime lock:                        0.605/0.3, 0.686/0.5, 0.671/0.7, peak 0.686/0.5\n#   With geotime lock:                      0.738/0.3, 0.713/0.5, 0.682/0.7, peak 0.741/0.32\n# Make sure that you adjust your threshold based on train soundscapes validation before submitting!","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-02T03:58:45.525480Z","iopub.execute_input":"2021-06-02T03:58:45.525831Z","iopub.status.idle":"2021-06-02T03:59:21.386704Z","shell.execute_reply.started":"2021-06-02T03:58:45.525801Z","shell.execute_reply":"2021-06-02T03:59:21.385716Z"},"trusted":true},"execution_count":null,"outputs":[]}]}