{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-04T06:59:59.996871Z","iopub.execute_input":"2024-04-04T06:59:59.997352Z","iopub.status.idle":"2024-04-04T07:00:17.723632Z","shell.execute_reply.started":"2024-04-04T06:59:59.997319Z","shell.execute_reply":"2024-04-04T07:00:17.722465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load the dataset","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:00:17.725600Z","iopub.execute_input":"2024-04-04T07:00:17.726318Z","iopub.status.idle":"2024-04-04T07:00:17.967504Z","shell.execute_reply.started":"2024-04-04T07:00:17.726276Z","shell.execute_reply":"2024-04-04T07:00:17.965819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:00:17.969285Z","iopub.execute_input":"2024-04-04T07:00:17.969838Z","iopub.status.idle":"2024-04-04T07:00:18.002697Z","shell.execute_reply.started":"2024-04-04T07:00:17.969792Z","shell.execute_reply":"2024-04-04T07:00:18.001135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.tail()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:01:44.077831Z","iopub.execute_input":"2024-04-04T07:01:44.078617Z","iopub.status.idle":"2024-04-04T07:01:44.098633Z","shell.execute_reply.started":"2024-04-04T07:01:44.078580Z","shell.execute_reply":"2024-04-04T07:01:44.097306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:01:51.887844Z","iopub.execute_input":"2024-04-04T07:01:51.888354Z","iopub.status.idle":"2024-04-04T07:01:51.896937Z","shell.execute_reply.started":"2024-04-04T07:01:51.888311Z","shell.execute_reply":"2024-04-04T07:01:51.895557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explore the columns and data types of the dataset. ","metadata":{}},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:01:57.665927Z","iopub.execute_input":"2024-04-04T07:01:57.666470Z","iopub.status.idle":"2024-04-04T07:01:57.728164Z","shell.execute_reply.started":"2024-04-04T07:01:57.666418Z","shell.execute_reply":"2024-04-04T07:01:57.726366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let's Find the missing values\n","metadata":{}},{"cell_type":"code","source":"df.isnull()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:02:04.528016Z","iopub.execute_input":"2024-04-04T07:02:04.528546Z","iopub.status.idle":"2024-04-04T07:02:04.581123Z","shell.execute_reply.started":"2024-04-04T07:02:04.528506Z","shell.execute_reply":"2024-04-04T07:02:04.579494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:02:10.036203Z","iopub.execute_input":"2024-04-04T07:02:10.037027Z","iopub.status.idle":"2024-04-04T07:02:10.073083Z","shell.execute_reply.started":"2024-04-04T07:02:10.036987Z","shell.execute_reply":"2024-04-04T07:02:10.071701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().sum() / df.shape[0]","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:02:14.541534Z","iopub.execute_input":"2024-04-04T07:02:14.542044Z","iopub.status.idle":"2024-04-04T07:02:14.579004Z","shell.execute_reply.started":"2024-04-04T07:02:14.542010Z","shell.execute_reply":"2024-04-04T07:02:14.577598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.duplicated()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:02:21.008033Z","iopub.execute_input":"2024-04-04T07:02:21.009392Z","iopub.status.idle":"2024-04-04T07:02:21.061593Z","shell.execute_reply.started":"2024-04-04T07:02:21.009327Z","shell.execute_reply":"2024-04-04T07:02:21.060305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:02:27.269947Z","iopub.execute_input":"2024-04-04T07:02:27.270662Z","iopub.status.idle":"2024-04-04T07:02:27.322050Z","shell.execute_reply.started":"2024-04-04T07:02:27.270606Z","shell.execute_reply":"2024-04-04T07:02:27.320703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Statistical Information of the Numeric Columns \n","metadata":{}},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:02:33.634498Z","iopub.execute_input":"2024-04-04T07:02:33.635679Z","iopub.status.idle":"2024-04-04T07:02:33.663487Z","shell.execute_reply.started":"2024-04-04T07:02:33.635624Z","shell.execute_reply":"2024-04-04T07:02:33.662220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# By  Using the Transpose method","metadata":{}},{"cell_type":"code","source":"df.describe().T","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:02:40.186851Z","iopub.execute_input":"2024-04-04T07:02:40.187316Z","iopub.status.idle":"2024-04-04T07:02:40.219244Z","shell.execute_reply.started":"2024-04-04T07:02:40.187283Z","shell.execute_reply":"2024-04-04T07:02:40.217978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let's Explore the Statistical Information of the Categorical Columns\n","metadata":{}},{"cell_type":"code","source":"df.describe(include=['O'])","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:02:46.263297Z","iopub.execute_input":"2024-04-04T07:02:46.263870Z","iopub.status.idle":"2024-04-04T07:02:46.393986Z","shell.execute_reply.started":"2024-04-04T07:02:46.263833Z","shell.execute_reply":"2024-04-04T07:02:46.392494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Summarize the Dataset","metadata":{}},{"cell_type":"code","source":"pip install skimpy","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:03:05.131171Z","iopub.execute_input":"2024-04-04T07:03:05.131663Z","iopub.status.idle":"2024-04-04T07:03:24.935211Z","shell.execute_reply.started":"2024-04-04T07:03:05.131618Z","shell.execute_reply":"2024-04-04T07:03:24.933960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nfrom skimpy import skim  # for summarize datasets\ndf = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\nskim(df)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:03:38.870651Z","iopub.execute_input":"2024-04-04T07:03:38.871158Z","iopub.status.idle":"2024-04-04T07:03:40.450079Z","shell.execute_reply.started":"2024-04-04T07:03:38.871121Z","shell.execute_reply":"2024-04-04T07:03:40.448700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize the Dataset","metadata":{}},{"cell_type":"code","source":"df.hist()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:07:19.225906Z","iopub.execute_input":"2024-04-04T07:07:19.226414Z","iopub.status.idle":"2024-04-04T07:07:20.412117Z","shell.execute_reply.started":"2024-04-04T07:07:19.226371Z","shell.execute_reply":"2024-04-04T07:07:20.410772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#importing the dependencies for visualization\nimport plotly.express as px\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\n\n#setting the style and background\nsns.set_style('darkgrid')\nmatplotlib.rcParams['font.size'] = 14\nmatplotlib.rcParams['figure.figsize'] = (10, 6)\nmatplotlib.rcParams['figure.facecolor'] = '#00000000'","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:07:20.415177Z","iopub.execute_input":"2024-04-04T07:07:20.415749Z","iopub.status.idle":"2024-04-04T07:07:20.425469Z","shell.execute_reply.started":"2024-04-04T07:07:20.415692Z","shell.execute_reply":"2024-04-04T07:07:20.424119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create bar chart using Plotly Express\nfig_bar = px.bar(df, x='primary_label', y='rating', color='primary_label', title='Distribution of Bird Species by Rating')\n\n# Show the bar chart\nfig_bar.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:07:20.427648Z","iopub.execute_input":"2024-04-04T07:07:20.428514Z","iopub.status.idle":"2024-04-04T07:07:21.479390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nsns.countplot(x='rating', data=df)\nplt.title('Distribution of Ratings')\nplt.xlabel('Rating')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:07:21.482037Z","iopub.execute_input":"2024-04-04T07:07:21.482420Z","iopub.status.idle":"2024-04-04T07:07:21.935665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lmplot(data=df , x='longitude',y='latitude',hue='rating')   # Show tips by 'smoker'\n","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:07:21.937832Z","iopub.execute_input":"2024-04-04T07:07:21.938353Z","iopub.status.idle":"2024-04-04T07:07:26.135941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate counts for the top 10 authors\ntop_authors = df['author'].value_counts().head(10)\n\n# Create a pie chart\nplt.figure(figsize=(8, 8))\nplt.pie(top_authors, labels=top_authors.index, autopct='%1.1f%%', startangle=140)\nplt.title('Top 10 Authors by Number of Recordings')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:07:26.137655Z","iopub.execute_input":"2024-04-04T07:07:26.138339Z","iopub.status.idle":"2024-04-04T07:07:26.451527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\n\n# Assuming metadata_df contains the data\n# Create a new DataFrame with the count of each bird species\nspecies_counts = df['primary_label'].value_counts().reset_index()\nspecies_counts.columns = ['Bird Species', 'Count']\n\n# Create a sunburst chart\nfig = px.sunburst(species_counts, path=['Bird Species'], values='Count', title='Geographical Distribution of Bird Species')\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:07:26.452941Z","iopub.execute_input":"2024-04-04T07:07:26.453731Z","iopub.status.idle":"2024-04-04T07:07:26.553694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\n\n# Assuming metadata_df contains the data\n# Create scatter plot on a map\nfig = px.scatter_mapbox(df, lat='latitude', lon='longitude', color='primary_label', \n                        hover_name='primary_label', hover_data=['latitude', 'longitude'], \n                        title='Geographical Distribution of Bird Species',\n                        zoom=3, height=600)\nfig.update_layout(mapbox_style=\"open-street-map\")\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:07:26.555156Z","iopub.execute_input":"2024-04-04T07:07:26.555524Z","iopub.status.idle":"2024-04-04T07:07:27.536547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub = pd.read_csv(\"/kaggle/input/birdclef-2024/sample_submission.csv\")\ndisplay(df_sub)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:08:11.993153Z","iopub.execute_input":"2024-04-04T07:08:11.994402Z","iopub.status.idle":"2024-04-04T07:08:12.037309Z","shell.execute_reply.started":"2024-04-04T07:08:11.994356Z","shell.execute_reply":"2024-04-04T07:08:12.036028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T07:08:18.020234Z","iopub.execute_input":"2024-04-04T07:08:18.020737Z","iopub.status.idle":"2024-04-04T07:08:18.030802Z","shell.execute_reply.started":"2024-04-04T07:08:18.020703Z","shell.execute_reply":"2024-04-04T07:08:18.029215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}