{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-07T21:50:47.424868Z","iopub.execute_input":"2023-04-07T21:50:47.425255Z","iopub.status.idle":"2023-04-07T21:50:47.849255Z","shell.execute_reply.started":"2023-04-07T21:50:47.425223Z","shell.execute_reply":"2023-04-07T21:50:47.847935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport random\nimport cv2\nimport librosa\nimport folium\n\n\nimport pandas as pd\nimport numpy as np\nfrom scipy.linalg import norm\n\nfrom IPython.display import Audio, display\nfrom scipy.stats import zscore\n\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.graph_objs as go\nfrom collections import Counter\nimport matplotlib.pyplot as plt\nfrom IPython.display import Audio\nimport altair as alt","metadata":{"execution":{"iopub.status.busy":"2023-04-07T21:50:47.851401Z","iopub.execute_input":"2023-04-07T21:50:47.852041Z","iopub.status.idle":"2023-04-07T21:50:47.859282Z","shell.execute_reply.started":"2023-04-07T21:50:47.852005Z","shell.execute_reply":"2023-04-07T21:50:47.857973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defining some helper functions\ndef normalize(v):\n    if norm(v) == 0:\n        return v\n    return norm(v)","metadata":{"execution":{"iopub.status.busy":"2023-04-07T21:50:47.860241Z","iopub.execute_input":"2023-04-07T21:50:47.860587Z","iopub.status.idle":"2023-04-07T21:50:47.873608Z","shell.execute_reply.started":"2023-04-07T21:50:47.860560Z","shell.execute_reply":"2023-04-07T21:50:47.872521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr_meta_df = pd.read_csv(\"/kaggle/input/birdclef-2023/train_metadata.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-04-07T21:50:47.877176Z","iopub.execute_input":"2023-04-07T21:50:47.877447Z","iopub.status.idle":"2023-04-07T21:50:47.936358Z","shell.execute_reply.started":"2023-04-07T21:50:47.877421Z","shell.execute_reply":"2023-04-07T21:50:47.935079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr_meta_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-07T21:50:47.938347Z","iopub.execute_input":"2023-04-07T21:50:47.939067Z","iopub.status.idle":"2023-04-07T21:50:47.948671Z","shell.execute_reply.started":"2023-04-07T21:50:47.938984Z","shell.execute_reply":"2023-04-07T21:50:47.947203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr_meta_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T21:50:47.954490Z","iopub.execute_input":"2023-04-07T21:50:47.954937Z","iopub.status.idle":"2023-04-07T21:50:47.982219Z","shell.execute_reply.started":"2023-04-07T21:50:47.954890Z","shell.execute_reply":"2023-04-07T21:50:47.980455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr_meta_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2023-04-07T21:50:47.984364Z","iopub.execute_input":"2023-04-07T21:50:47.984774Z","iopub.status.idle":"2023-04-07T21:50:48.001305Z","shell.execute_reply.started":"2023-04-07T21:50:47.984735Z","shell.execute_reply":"2023-04-07T21:50:47.999852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.graph_objects as go\nimport pandas as pd\n\n# Calculate percentage of empty values in each column\nempty_pct = (tr_meta_df.isnull().sum() / len(tr_meta_df)) * 100\n\n# Create a bar chart using Plotly\nfig = go.Figure(data=[go.Bar(\n    x=empty_pct.index,  # x-axis values\n    y=empty_pct.values,  # y-axis values\n    text=empty_pct.round(2).astype(str) + '%',  # text label with percentage\n    textposition='auto',  # position of the text label\n    marker=dict(color='green')  # set the color of the bar to green\n)])\n\n# Update the layout of the chart\nfig.update_layout(\n    title='Percentage of Empty Values in tr_meta_df',\n    xaxis=dict(title='Columns'),\n    yaxis=dict(title='Percentage of Empty Values')\n)\n\n# Display the chart\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-07T21:50:48.002857Z","iopub.execute_input":"2023-04-07T21:50:48.003192Z","iopub.status.idle":"2023-04-07T21:50:48.030805Z","shell.execute_reply.started":"2023-04-07T21:50:48.003165Z","shell.execute_reply":"2023-04-07T21:50:48.029388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(tr_meta_df['primary_label'].describe())\nprint(tr_meta_df['primary_label'].unique())","metadata":{"execution":{"iopub.status.busy":"2023-04-07T21:50:48.032413Z","iopub.execute_input":"2023-04-07T21:50:48.033391Z","iopub.status.idle":"2023-04-07T21:50:48.043827Z","shell.execute_reply.started":"2023-04-07T21:50:48.033357Z","shell.execute_reply":"2023-04-07T21:50:48.042667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_distribution(df, column, nbins=50):\n    ordered_values = df[column].value_counts().index.tolist()\n    fig = px.histogram(df, x=column, nbins=nbins,\n                       color_discrete_sequence=['green'])\n    fig.update_layout(template='plotly_white',\n                      title=f'Distribution of {column}',\n                      xaxis_title=column.capitalize(),\n                      yaxis_title='Count')\n    fig.update_xaxes(type='category', categoryorder='array', categoryarray=ordered_values)\n    fig.show()\n    \ndef plot_distribution2(df, column, nbins=50):\n    fig = px.histogram(df, x=column, nbins=nbins,\n                       color_discrete_sequence=['green'])\n    fig.update_layout(template='plotly_white',\n                      title=f'Distribution of {column}',\n                      xaxis_title=column.capitalize(),\n                      yaxis_title='Count')\n    fig.update_xaxes(type='category', categoryorder='array')\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T22:19:12.194625Z","iopub.execute_input":"2023-04-07T22:19:12.194978Z","iopub.status.idle":"2023-04-07T22:19:12.202984Z","shell.execute_reply.started":"2023-04-07T22:19:12.194943Z","shell.execute_reply":"2023-04-07T22:19:12.202117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_distribution(tr_meta_df,'primary_label')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T21:50:48.056386Z","iopub.execute_input":"2023-04-07T21:50:48.057498Z","iopub.status.idle":"2023-04-07T21:50:48.194691Z","shell.execute_reply.started":"2023-04-07T21:50:48.057458Z","shell.execute_reply":"2023-04-07T21:50:48.193217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(tr_meta_df['secondary_labels'].unique())\n# print(tr_meta_df['secondary_labels'].value_counts())\n\nplot_distribution(tr_meta_df,'secondary_labels')\n\n# removing the first one and then printing again\n\ntr_meta_df_filtered = tr_meta_df.loc[tr_meta_df['secondary_labels'].apply(len) > 2]\n\n# Call the plot_distribution function\nprint(\"distribution of secondary_labels modified with removal of []\")\nplot_distribution(tr_meta_df_filtered, 'secondary_labels')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T21:50:48.195874Z","iopub.execute_input":"2023-04-07T21:50:48.196207Z","iopub.status.idle":"2023-04-07T21:50:48.407305Z","shell.execute_reply.started":"2023-04-07T21:50:48.196175Z","shell.execute_reply":"2023-04-07T21:50:48.406355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr_meta_df[\"type\"].head()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T21:50:48.426712Z","iopub.execute_input":"2023-04-07T21:50:48.427525Z","iopub.status.idle":"2023-04-07T21:50:48.437758Z","shell.execute_reply.started":"2023-04-07T21:50:48.427491Z","shell.execute_reply":"2023-04-07T21:50:48.436405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import ast\n# Flatten the list of labels in the \"type\" column\nlabels = [label.strip(\"[]'\") for sublist in trainmeta_df['type'].apply(ast.literal_eval) for label in sublist]\n\n# Count the occurrence of each label\nlabel_counts = Counter(labels)\n\n# Select the top 10 labels\ntop_labels = dict(sorted(label_counts.items(), key=lambda item: item[1], reverse=True)[:10])\n\n# Create a bar plot of the top 10 label counts\nfig = px.bar(x=list(top_labels.keys()), y=list(top_labels.values()), color_discrete_sequence=['green'])\nfig.update_layout(title_text=\"Top 10 Types by Count\")\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-07T22:01:51.376533Z","iopub.execute_input":"2023-04-07T22:01:51.376928Z","iopub.status.idle":"2023-04-07T22:01:51.573248Z","shell.execute_reply.started":"2023-04-07T22:01:51.376887Z","shell.execute_reply":"2023-04-07T22:01:51.572330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.density_mapbox(tr_meta_df, lat='latitude', lon='longitude',\n                        radius=2, center=dict(lat=0, lon=180),\n                        zoom=1, mapbox_style=\"stamen-terrain\", color_continuous_scale='greens')\n\n# set plot title\nfig.update_layout(title='Heatmap')\n\n# show the plot\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T22:08:54.355181Z","iopub.execute_input":"2023-04-07T22:08:54.355538Z","iopub.status.idle":"2023-04-07T22:08:54.421000Z","shell.execute_reply.started":"2023-04-07T22:08:54.355507Z","shell.execute_reply":"2023-04-07T22:08:54.420124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_distribution(tr_meta_df, 'author')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T22:14:19.736680Z","iopub.execute_input":"2023-04-07T22:14:19.737029Z","iopub.status.idle":"2023-04-07T22:14:19.870585Z","shell.execute_reply.started":"2023-04-07T22:14:19.736996Z","shell.execute_reply":"2023-04-07T22:14:19.869596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define your text to visualize\nfrom wordcloud import WordCloud\n\n# Create the WordCloud object with a green color\ntext = ' '.join(tr_meta_df['author'].astype(str).values.tolist())\nwc = WordCloud(background_color='white', width=800, height=400, colormap='Greens').generate(text)\n\n# Display the word cloud\nplt.figure(figsize=(12, 6))\nplt.imshow(wc, interpolation='bilinear')\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T22:13:35.917162Z","iopub.execute_input":"2023-04-07T22:13:35.917508Z","iopub.status.idle":"2023-04-07T22:13:36.767307Z","shell.execute_reply.started":"2023-04-07T22:13:35.917481Z","shell.execute_reply":"2023-04-07T22:13:36.766220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_distribution(tr_meta_df, 'common_name')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T22:15:02.145971Z","iopub.execute_input":"2023-04-07T22:15:02.146365Z","iopub.status.idle":"2023-04-07T22:15:02.280776Z","shell.execute_reply.started":"2023-04-07T22:15:02.146333Z","shell.execute_reply":"2023-04-07T22:15:02.279917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define your text to visualize\nfrom wordcloud import WordCloud\n\n# Create the WordCloud object with a green color\ntext = ' '.join(tr_meta_df['common_name'].astype(str).values.tolist())\nwc = WordCloud(background_color='white', width=800, height=400, colormap='Greens').generate(text)\n\n# Display the word cloud\nplt.figure(figsize=(12, 6))\nplt.imshow(wc, interpolation='bilinear')\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T22:15:43.136013Z","iopub.execute_input":"2023-04-07T22:15:43.136400Z","iopub.status.idle":"2023-04-07T22:15:43.939797Z","shell.execute_reply.started":"2023-04-07T22:15:43.136362Z","shell.execute_reply":"2023-04-07T22:15:43.938839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_distribution(tr_meta_df, 'license')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T22:16:33.437442Z","iopub.execute_input":"2023-04-07T22:16:33.437820Z","iopub.status.idle":"2023-04-07T22:16:33.575811Z","shell.execute_reply.started":"2023-04-07T22:16:33.437785Z","shell.execute_reply":"2023-04-07T22:16:33.575008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define your text to visualize\nfrom wordcloud import WordCloud\n\n# Create the WordCloud object with a green color\ntext = ' '.join(tr_meta_df['license'].astype(str).values.tolist())\nwc = WordCloud(background_color='white', width=800, height=400, colormap='Greens').generate(text)\n\n# Display the word cloud\nplt.figure(figsize=(12, 6))\nplt.imshow(wc, interpolation='bilinear')\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T22:16:49.401159Z","iopub.execute_input":"2023-04-07T22:16:49.401546Z","iopub.status.idle":"2023-04-07T22:16:49.890036Z","shell.execute_reply.started":"2023-04-07T22:16:49.401512Z","shell.execute_reply":"2023-04-07T22:16:49.889057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_distribution2(tr_meta_df,'rating')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T22:19:18.558667Z","iopub.execute_input":"2023-04-07T22:19:18.559091Z","iopub.status.idle":"2023-04-07T22:19:18.635880Z","shell.execute_reply.started":"2023-04-07T22:19:18.559040Z","shell.execute_reply":"2023-04-07T22:19:18.634792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_boxplot(df, y_col, title):\n    fig = px.box(df, y=y_col)\n    fig.update_traces(marker_color='green')\n    fig.update_layout(title=title)\n    fig.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-07T22:22:53.737945Z","iopub.execute_input":"2023-04-07T22:22:53.738332Z","iopub.status.idle":"2023-04-07T22:22:53.744687Z","shell.execute_reply.started":"2023-04-07T22:22:53.738298Z","shell.execute_reply":"2023-04-07T22:22:53.743356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_boxplot(tr_meta_df,'rating','Box Plot for Ratings')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T22:23:32.481078Z","iopub.execute_input":"2023-04-07T22:23:32.481452Z","iopub.status.idle":"2023-04-07T22:23:32.560306Z","shell.execute_reply.started":"2023-04-07T22:23:32.481419Z","shell.execute_reply":"2023-04-07T22:23:32.559225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(tr_meta_df, x=\"rating\", nbins=len(tr_meta_df[\"rating\"].unique()) , color_discrete_sequence=['green'])\nfig.update_layout(title_text=\"Distribution of Ratings\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T22:24:13.780967Z","iopub.execute_input":"2023-04-07T22:24:13.781367Z","iopub.status.idle":"2023-04-07T22:24:13.840321Z","shell.execute_reply.started":"2023-04-07T22:24:13.781330Z","shell.execute_reply":"2023-04-07T22:24:13.839382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Integration with MAPBOX","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}