{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import datetime as dt\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.graph_objects as go\nimport plotly.express as px","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Summarize Missings and Unexpected Values","execution_count":null},{"metadata":{"scrolled":true,"trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/birdsong-recognition/train.csv')\nprint('{} observations in training dataset.'.format(train.shape[0]))\nprint(' ')\nprint('{} features contain missing values in training dataset:'.format(train.columns[train.isnull().sum() > 0].shape[0]))\nprint('{}'.format(train.columns[train.isnull().sum() > 0].tolist()))\nprint(' ')\n\nfor col in train.columns[train.isnull().sum() > 0]:\n    print('{}: {} missing values / {} %'\\\n          .format(col, train.isnull().sum()[col], round(train.isnull().sum()[col]/train.shape[0]*100, 2)))\nprint(' ')\n\nprint('{} incomplete date values'.format(train['date'].str.endswith('00').sum()))\nprint('{} observations missing day value.'.format(train['date'].str.endswith('00').sum()))\nprint('{} observations missing month value.'.format((train['date'].str[-5:-3] == '00').sum()))\nprint('{} observations missing year value.'.format((train['date'].str[:4] == '0000').sum()))\nprint(' ')\n\nprint('{} observations without longitude'.format((train['longitude']=='Not specified').sum()))\nprint('{} observations without latitude'.format((train['latitude']=='Not specified').sum()))\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Date Adjustments","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Obs 5048, 5049 labeled with year 1012\ntrain.loc[train['date'].str.startswith('1012'), 'date'] = '2012'+train['date'].str[4:]\n# Sue Riffe was at matching long, lat in July 2019\ntrain.loc[train['date'].str.startswith('0201'), 'date'] = '2019'+train['date'].str[4:]\n\n# Temporarily ignore 34 obs with missing date info, treat 112 as belonging to first day of month\ntrain.drop(train[train['date'].str[-5:-3] == '00'].index, axis=0, inplace=True)\ntrain['date_edit'] = False\ntrain.loc[train['date'].str.endswith('00'), 'date_edit'] = True\ntrain.loc[train['date'].str.endswith('00'), 'date'] = train.loc[train['date'].str.endswith('00'), 'date'].str.replace('00','01')\n\ntrain['date'] = pd.to_datetime(train['date'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Recording Dates","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16,8))\nsns.set()\n\nax1 = plt.subplot(1,3,1)\nplt.hist(train['date'].dt.year, bins=20, alpha=0.8)\nplt.xlabel('Year of Recording')\n\nax2 = plt.subplot(1,3,2, sharey=ax1)\nplt.hist(train['date'].dt.month, bins=12, alpha=0.8)\nplt.xlabel('Month of Recording')\n\nax3 = plt.subplot(1,3,3, sharey=ax1)\nplt.hist(train['date'].dt.day, bins=31, alpha=0.8)\nplt.xlabel('Day of Recording')\n\nplt.suptitle('Distribution of Recording Dates', fontsize=16)\n;","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We observe a relatively large spike in recordings from the spring and summer months to no surprise. We have relatively consistent recording frequency in winter months in the 600 - 900 range.","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"### Lat & Long Spread","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(14,8))\nsns.set()\n\n# Ignore 226 observations with missing coordinates\nax1 = plt.subplot(1,2,1)\nplt.hist(train.loc[train['latitude'] != 'Not specified','latitude'].astype(float), bins=20, alpha=0.8)\nplt.xlabel('Latitude')\n\nax2 = plt.subplot(1,2,2, sharey=ax1)\nplt.hist(train.loc[train['longitude'] != 'Not specified','longitude'].astype(float), bins=20, alpha=0.8)\nplt.xlabel('Longitude')\n\nplt.suptitle('Distribution of Lat & Long', fontsize=16)\n;\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = go.Figure(data=go.Scattergeo(\n        lon = train['longitude'],\n        lat = train['latitude'],\n        text = train['primary_label'],\n        mode = 'markers',\n        marker = dict(\n            size = 3,\n            opacity = 0.5)\n        ))\n\nfig.update_layout(\n        title = 'Bird Recordings Locations',\n    )\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We observe a concentration of recordings from North America and Europe, with sparse entries from South America and Asia. In the US, we can see geographic clusters of recordings in coastal regions and some inland regions.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def make_bird_map(df, birds):\n    '''Plot map of recordings for specified ebird_codes'''\n    if type(birds) != list:\n        birds = birds.tolist() # for single bird specified as str\n    colors  = sns.color_palette(\"hls\", len(birds)).as_hex()\n\n    fig = go.Figure()\n    for i in range(len(birds)):\n        df_sub = df.loc[df['ebird_code']==birds[i],:]\n        df_sub['text'] = 'Bird: ' + df_sub['species'] + \\\n            '<br>Recordist: ' + df_sub['recordist'] + \\\n            '<br>Date: ' + df_sub['date'].astype(str)\n        \n        fig.add_trace(\n            go.Scattergeo(\n                lon = df_sub['longitude'],\n                lat = df_sub['latitude'],\n                text = df_sub['text'],\n                mode = 'markers',\n                marker = dict(\n                    size = 3,\n                    opacity = 0.75,\n                    color = colors[i]),\n                name = birds[i],\n            )\n        )\n\n    fig.update_layout(\n            title = 'Recordings Locations (Selected Birds)',\n        )\n    fig.show()","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"trusted":true},"cell_type":"code","source":"make_bird_map(train, train['ebird_code'].value_counts().index[:10])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Iterating through different birds, we already see clear geographic concentrations of recordings. Incorporating geo cordinates and matching to climate & biome information should improve prediction compared to audio alone and will be explored.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def make_recordist_map(df, recorders):\n    '''Plot map of recordings for specified ebird_code'''\n    if type(recorders) != list:\n        recorders = recorders.tolist()\n    colors  = sns.color_palette(\"hls\", len(recorders)).as_hex()\n\n    fig = go.Figure()\n    for i in range(len(recorders)):\n        df_sub = df.loc[df['recordist']==recorders[i],:]\n        df_sub['text'] = 'Bird: ' + df_sub['species'] + '<br>Date: '+ df_sub['date'].astype(str)\n        \n        fig.add_trace(\n            go.Scattergeo(\n                lon = df_sub['longitude'],\n                lat = df_sub['latitude'],\n                text = df_sub['text'],\n                mode = 'markers',\n                marker = dict(\n                    size = 3,\n                    opacity = 0.5,\n                    color = colors[i]),\n                name = recorders[i],\n            )\n        )\n\n    fig.update_layout(\n            title = 'Recordings Locations (Selected Recordists)',\n        )\n    fig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"make_recordist_map(train, train['recordist'].value_counts().index[:10])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Summarize Geo Info of Birds","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def make_bird_geo_summary(df, birds):\n    if type(birds) != list:\n        birds = birds.tolist()\n    df_sub = train.loc[train['ebird_code'].isin(birds),:]\n    df_sub = df_sub.loc[df_sub['latitude']!='Not specified',:]\n    df_sub['longitude'] = df_sub['longitude'].astype(float)\n    df_sub['latitude'] = df_sub['latitude'].astype(float)\n    \n    make_bird_map(df_sub, birds)\n    \n    plt.figure(figsize=(14,8))\n    ax1 = plt.subplot(1,3,1)\n    sns.countplot(df_sub['date'].dt.month, alpha=0.8, palette='Blues_d')\n    plt.xlabel('Month of Recording')\n    #plt.xticks(ticks = [1,4,7,10], labels=[1,4,7,10])\n    \n    ax2 = plt.subplot(1,3,2)\n    sns.boxplot(x=df_sub['date'].dt.quarter, y=\"longitude\", data=df_sub)\n    plt.xlabel('Quarter', fontsize=14)\n    \n    ax3 = plt.subplot(1,3,3)\n    sns.boxplot(x=df_sub['date'].dt.quarter, y=\"latitude\", data=df_sub)\n    plt.xlabel('Quarter', fontsize=14)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Identify birds with widest geographic appearances\nsub_lat = train.loc[train['longitude']!='Not specified',:]\nsub_lat['latitude'] = sub_lat['latitude'].astype('float')\ntravel_birds = sub_lat.groupby('ebird_code')['latitude'].std().sort_values(ascending=False).index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"make_bird_geo_summary(train, travel_birds[:5])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Here, we observe some birds have recordings from a wide range of geographic locations. The latitude plot suggests some birds in our dataset do migrate seasonally, suggesting time of recording may interact with location in a meaningful way when predicting bird types.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"make_bird_geo_summary(train, travel_birds[-10:])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"In contrast, some birds are highly concentrated in specific areas. Assuming the training dataset is representative of these birds living areas, geographic location may be an effective screening criteria. Further research into biomes, climate, and bird specifies will be pursued to confirm this.","execution_count":null}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}