{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":19596,"databundleVersionId":1292430,"sourceType":"competition"},{"sourceId":1262046,"sourceType":"datasetVersion","datasetId":726424},{"sourceId":1264575,"sourceType":"datasetVersion","datasetId":725893},{"sourceId":7401974,"sourceType":"datasetVersion","datasetId":4304204},{"sourceId":7402018,"sourceType":"datasetVersion","datasetId":4304231},{"sourceId":7402031,"sourceType":"datasetVersion","datasetId":4304243},{"sourceId":7402038,"sourceType":"datasetVersion","datasetId":4304249},{"sourceId":7402070,"sourceType":"datasetVersion","datasetId":4304272},{"sourceId":7402154,"sourceType":"datasetVersion","datasetId":4304325},{"sourceId":7402163,"sourceType":"datasetVersion","datasetId":4304331},{"sourceId":7402168,"sourceType":"datasetVersion","datasetId":4304334},{"sourceId":7402174,"sourceType":"datasetVersion","datasetId":4304339},{"sourceId":7402179,"sourceType":"datasetVersion","datasetId":4304342},{"sourceId":7402188,"sourceType":"datasetVersion","datasetId":4304350},{"sourceId":7402192,"sourceType":"datasetVersion","datasetId":4304353},{"sourceId":7402200,"sourceType":"datasetVersion","datasetId":4304360}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Importing necessary libraries\n\nimport numpy as np              # NumPy for numerical operations\nimport pandas as pd             # Pandas for data manipulation and analysis\nimport matplotlib.pyplot as plt # Matplotlib for creating visualizations\nimport seaborn as sns           # Seaborn for statistical data visualization\nfrom sklearn.model_selection import train_test_split  # Scikit-learn for machine learning, train-test split\nfrom sklearn.preprocessing import LabelEncoder        # Scikit-learn for label encoding categorical variables","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:37.574226Z","iopub.execute_input":"2024-01-14T19:22:37.574830Z","iopub.status.idle":"2024-01-14T19:22:37.580883Z","shell.execute_reply.started":"2024-01-14T19:22:37.574758Z","shell.execute_reply":"2024-01-14T19:22:37.580044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data=pd.read_csv(\"/kaggle/input/birdsong-recognition/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:37.582361Z","iopub.execute_input":"2024-01-14T19:22:37.583277Z","iopub.status.idle":"2024-01-14T19:22:38.027566Z","shell.execute_reply.started":"2024-01-14T19:22:37.583244Z","shell.execute_reply":"2024-01-14T19:22:38.026499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setting Pandas options for displaying DataFrame columns\n\npd.set_option('display.max_columns', None)        # Display all columns when showing a DataFrame\npd.set_option('display.expand_frame_repr', False)  # Disable expanding the DataFrame to fit the width of the console","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.029120Z","iopub.execute_input":"2024-01-14T19:22:38.029459Z","iopub.status.idle":"2024-01-14T19:22:38.034798Z","shell.execute_reply.started":"2024-01-14T19:22:38.029430Z","shell.execute_reply":"2024-01-14T19:22:38.033599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.038082Z","iopub.execute_input":"2024-01-14T19:22:38.038525Z","iopub.status.idle":"2024-01-14T19:22:38.075824Z","shell.execute_reply.started":"2024-01-14T19:22:38.038480Z","shell.execute_reply":"2024-01-14T19:22:38.074582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the shape of the DataFrame (number of rows and columns)\n\ndata.shape","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.077070Z","iopub.execute_input":"2024-01-14T19:22:38.077400Z","iopub.status.idle":"2024-01-14T19:22:38.084266Z","shell.execute_reply.started":"2024-01-14T19:22:38.077371Z","shell.execute_reply":"2024-01-14T19:22:38.083009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Finding columns with null values in the DataFrame\n\ncolumns_with_nulls = data.columns[data.isnull().any()]  # Get columns containing at least one null value\nprint(columns_with_nulls)\n","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.085486Z","iopub.execute_input":"2024-01-14T19:22:38.085837Z","iopub.status.idle":"2024-01-14T19:22:38.168920Z","shell.execute_reply.started":"2024-01-14T19:22:38.085792Z","shell.execute_reply":"2024-01-14T19:22:38.167703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Retrieving the column names of the DataFrame\n\ndata.columns","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.170992Z","iopub.execute_input":"2024-01-14T19:22:38.171487Z","iopub.status.idle":"2024-01-14T19:22:38.182968Z","shell.execute_reply.started":"2024-01-14T19:22:38.171432Z","shell.execute_reply":"2024-01-14T19:22:38.181623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the number of unique values in the 'rating' column\n\ndata['rating'].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.184853Z","iopub.execute_input":"2024-01-14T19:22:38.185195Z","iopub.status.idle":"2024-01-14T19:22:38.194325Z","shell.execute_reply.started":"2024-01-14T19:22:38.185167Z","shell.execute_reply":"2024-01-14T19:22:38.193484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Retrieving the unique values in the 'rating' column\n\ndata['rating'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.198657Z","iopub.execute_input":"2024-01-14T19:22:38.199525Z","iopub.status.idle":"2024-01-14T19:22:38.207334Z","shell.execute_reply.started":"2024-01-14T19:22:38.199490Z","shell.execute_reply":"2024-01-14T19:22:38.206293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the occurrences of each unique value in the 'rating' column\n\ndata['rating'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.208531Z","iopub.execute_input":"2024-01-14T19:22:38.209391Z","iopub.status.idle":"2024-01-14T19:22:38.221068Z","shell.execute_reply.started":"2024-01-14T19:22:38.209359Z","shell.execute_reply":"2024-01-14T19:22:38.220107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Retrieving the unique values in the 'playback_used' column\n\ndata['playback_used'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.222278Z","iopub.execute_input":"2024-01-14T19:22:38.223198Z","iopub.status.idle":"2024-01-14T19:22:38.232877Z","shell.execute_reply.started":"2024-01-14T19:22:38.223165Z","shell.execute_reply":"2024-01-14T19:22:38.231914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the occurrences of each unique value in the 'playback_used' column\n\ndata['playback_used'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.234181Z","iopub.execute_input":"2024-01-14T19:22:38.235220Z","iopub.status.idle":"2024-01-14T19:22:38.247534Z","shell.execute_reply.started":"2024-01-14T19:22:38.235185Z","shell.execute_reply":"2024-01-14T19:22:38.246727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filling missing values in the 'playback_used' column with 'no'\n\ndata['playback_used'].fillna('no', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.249084Z","iopub.execute_input":"2024-01-14T19:22:38.249455Z","iopub.status.idle":"2024-01-14T19:22:38.262421Z","shell.execute_reply.started":"2024-01-14T19:22:38.249417Z","shell.execute_reply":"2024-01-14T19:22:38.261171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the occurrences of each unique value in the 'playback_used' column after handling missing values\n\ndata['playback_used'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.263988Z","iopub.execute_input":"2024-01-14T19:22:38.264343Z","iopub.status.idle":"2024-01-14T19:22:38.279968Z","shell.execute_reply.started":"2024-01-14T19:22:38.264312Z","shell.execute_reply":"2024-01-14T19:22:38.279110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encoding the 'playback_used' column using LabelEncoder and creating a new column 'playback_used_encoded'\n\nle = LabelEncoder()                                  # Creating a LabelEncoder instance\ndata['playback_used_encoded'] = le.fit_transform(data['playback_used'])  # Encoding 'playback_used' and creating a new column\ndata.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.281211Z","iopub.execute_input":"2024-01-14T19:22:38.282185Z","iopub.status.idle":"2024-01-14T19:22:38.329629Z","shell.execute_reply.started":"2024-01-14T19:22:38.282138Z","shell.execute_reply":"2024-01-14T19:22:38.328448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Retrieving the unique values in the 'ebird_code' column\n\ndata['ebird_code'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.331052Z","iopub.execute_input":"2024-01-14T19:22:38.331373Z","iopub.status.idle":"2024-01-14T19:22:38.341283Z","shell.execute_reply.started":"2024-01-14T19:22:38.331347Z","shell.execute_reply":"2024-01-14T19:22:38.340449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the number of unique values in the 'ebird_code' column\n\ndata['ebird_code'].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.342489Z","iopub.execute_input":"2024-01-14T19:22:38.342833Z","iopub.status.idle":"2024-01-14T19:22:38.355487Z","shell.execute_reply.started":"2024-01-14T19:22:38.342793Z","shell.execute_reply":"2024-01-14T19:22:38.354292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Retrieving the unique values in the 'channels' column\n\ndata['channels'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.356871Z","iopub.execute_input":"2024-01-14T19:22:38.357322Z","iopub.status.idle":"2024-01-14T19:22:38.372405Z","shell.execute_reply.started":"2024-01-14T19:22:38.357292Z","shell.execute_reply":"2024-01-14T19:22:38.371227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the occurrences of each unique value in the 'channels' column\n\ndata['channels'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.373462Z","iopub.execute_input":"2024-01-14T19:22:38.373842Z","iopub.status.idle":"2024-01-14T19:22:38.388138Z","shell.execute_reply.started":"2024-01-14T19:22:38.373802Z","shell.execute_reply":"2024-01-14T19:22:38.387048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Converting the 'channels' column to integers by extracting the first character as a string and then converting to int\n\ndata['channels'] = data['channels'].astype(str).str[0].astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.389608Z","iopub.execute_input":"2024-01-14T19:22:38.389962Z","iopub.status.idle":"2024-01-14T19:22:38.421861Z","shell.execute_reply.started":"2024-01-14T19:22:38.389933Z","shell.execute_reply":"2024-01-14T19:22:38.420556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the occurrences of each unique value in the 'channels' column after conversion\n\ndata['channels'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.423213Z","iopub.execute_input":"2024-01-14T19:22:38.423661Z","iopub.status.idle":"2024-01-14T19:22:38.433399Z","shell.execute_reply.started":"2024-01-14T19:22:38.423614Z","shell.execute_reply":"2024-01-14T19:22:38.432225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Retrieving the unique values in the 'date' column\n\ndata['date'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.434856Z","iopub.execute_input":"2024-01-14T19:22:38.435322Z","iopub.status.idle":"2024-01-14T19:22:38.448778Z","shell.execute_reply.started":"2024-01-14T19:22:38.435291Z","shell.execute_reply":"2024-01-14T19:22:38.447644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the number of unique values in the 'date' column\n\ndata['date'].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.450478Z","iopub.execute_input":"2024-01-14T19:22:38.450939Z","iopub.status.idle":"2024-01-14T19:22:38.465608Z","shell.execute_reply.started":"2024-01-14T19:22:38.450898Z","shell.execute_reply":"2024-01-14T19:22:38.464438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extracting the year from the 'date' column and creating a new 'year' column\n\ndata['year'] = data['date'].apply(lambda x: x.split('-')[0]).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.476901Z","iopub.execute_input":"2024-01-14T19:22:38.477475Z","iopub.status.idle":"2024-01-14T19:22:38.503803Z","shell.execute_reply.started":"2024-01-14T19:22:38.477435Z","shell.execute_reply":"2024-01-14T19:22:38.502827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extracting the month from the 'date' column and creating a new 'month' column\n\ndata['month'] = data['date'].apply(lambda x: x.split('-')[1]).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.505045Z","iopub.execute_input":"2024-01-14T19:22:38.505366Z","iopub.status.idle":"2024-01-14T19:22:38.532259Z","shell.execute_reply.started":"2024-01-14T19:22:38.505338Z","shell.execute_reply":"2024-01-14T19:22:38.531027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extracting the day from the 'date' column and creating a new 'day' column\n\ndata['day'] = data['date'].apply(lambda x: x.split('-')[2]).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.533493Z","iopub.execute_input":"2024-01-14T19:22:38.533869Z","iopub.status.idle":"2024-01-14T19:22:38.558374Z","shell.execute_reply.started":"2024-01-14T19:22:38.533836Z","shell.execute_reply":"2024-01-14T19:22:38.557238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dropping the 'date' column from the DataFrame\n\ndata.drop('date', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.559943Z","iopub.execute_input":"2024-01-14T19:22:38.560313Z","iopub.status.idle":"2024-01-14T19:22:38.575049Z","shell.execute_reply.started":"2024-01-14T19:22:38.560278Z","shell.execute_reply":"2024-01-14T19:22:38.574058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.576578Z","iopub.execute_input":"2024-01-14T19:22:38.577110Z","iopub.status.idle":"2024-01-14T19:22:38.607375Z","shell.execute_reply.started":"2024-01-14T19:22:38.577079Z","shell.execute_reply":"2024-01-14T19:22:38.606522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Retrieving the unique values in the 'pitch' column\n\ndata['pitch'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.608828Z","iopub.execute_input":"2024-01-14T19:22:38.609409Z","iopub.status.idle":"2024-01-14T19:22:38.616710Z","shell.execute_reply.started":"2024-01-14T19:22:38.609376Z","shell.execute_reply":"2024-01-14T19:22:38.615958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the occurrences of each unique value in the 'pitch' column\n\ndata['pitch'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.618155Z","iopub.execute_input":"2024-01-14T19:22:38.618864Z","iopub.status.idle":"2024-01-14T19:22:38.637502Z","shell.execute_reply.started":"2024-01-14T19:22:38.618810Z","shell.execute_reply":"2024-01-14T19:22:38.636347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encoding the 'pitch' column using LabelEncoder and creating a new column 'pitch_encoded'\n\nle = LabelEncoder()                              # Creating a LabelEncoder instance\ndata['pitch_encoded'] = le.fit_transform(data['pitch'])  # Encoding 'pitch' and creating a new column\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.639600Z","iopub.execute_input":"2024-01-14T19:22:38.639983Z","iopub.status.idle":"2024-01-14T19:22:38.686393Z","shell.execute_reply.started":"2024-01-14T19:22:38.639914Z","shell.execute_reply":"2024-01-14T19:22:38.685280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dropping the 'pitch' column from the DataFrame\n\ndata.drop('pitch', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.687996Z","iopub.execute_input":"2024-01-14T19:22:38.688867Z","iopub.status.idle":"2024-01-14T19:22:38.705631Z","shell.execute_reply.started":"2024-01-14T19:22:38.688731Z","shell.execute_reply":"2024-01-14T19:22:38.704738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Retrieving the unique values in the 'duration' column\n\ndata['duration'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.707178Z","iopub.execute_input":"2024-01-14T19:22:38.707520Z","iopub.status.idle":"2024-01-14T19:22:38.718619Z","shell.execute_reply.started":"2024-01-14T19:22:38.707492Z","shell.execute_reply":"2024-01-14T19:22:38.717094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the number of unique values in the 'duration' column\n\ndata['duration'].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.719934Z","iopub.execute_input":"2024-01-14T19:22:38.720256Z","iopub.status.idle":"2024-01-14T19:22:38.728348Z","shell.execute_reply.started":"2024-01-14T19:22:38.720228Z","shell.execute_reply":"2024-01-14T19:22:38.727311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying concise information about the DataFrame\n\ndata.info()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.729541Z","iopub.execute_input":"2024-01-14T19:22:38.729880Z","iopub.status.idle":"2024-01-14T19:22:38.811116Z","shell.execute_reply.started":"2024-01-14T19:22:38.729850Z","shell.execute_reply":"2024-01-14T19:22:38.809987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Comment on duration and filename columns\n# The 'duration' column has no null values and its data type is integer, indicating it does not require further modification.\n# The 'filename' column has no null values and is used to get the path of the audio files while iterating through each one of them,\n# so we are not making any changes to this column.","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.813602Z","iopub.execute_input":"2024-01-14T19:22:38.813973Z","iopub.status.idle":"2024-01-14T19:22:38.819097Z","shell.execute_reply.started":"2024-01-14T19:22:38.813931Z","shell.execute_reply":"2024-01-14T19:22:38.817806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the number of unique values in the 'speed' column\n\ndata['speed'].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.820322Z","iopub.execute_input":"2024-01-14T19:22:38.820749Z","iopub.status.idle":"2024-01-14T19:22:38.834522Z","shell.execute_reply.started":"2024-01-14T19:22:38.820700Z","shell.execute_reply":"2024-01-14T19:22:38.833246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Retrieving the unique values in the 'speed' column\n\ndata['speed'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.836367Z","iopub.execute_input":"2024-01-14T19:22:38.836847Z","iopub.status.idle":"2024-01-14T19:22:38.846880Z","shell.execute_reply.started":"2024-01-14T19:22:38.836804Z","shell.execute_reply":"2024-01-14T19:22:38.845564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the occurrences of each unique value in the 'speed' column\n\ndata['speed'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.848707Z","iopub.execute_input":"2024-01-14T19:22:38.849326Z","iopub.status.idle":"2024-01-14T19:22:38.862021Z","shell.execute_reply.started":"2024-01-14T19:22:38.849284Z","shell.execute_reply":"2024-01-14T19:22:38.861032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encoding the 'speed' column using LabelEncoder and creating a new column 'speed_encoded'\n\nle = LabelEncoder()                                # Creating a LabelEncoder instance\ndata['speed_encoded'] = le.fit_transform(data['speed'])  # Encoding 'speed' and creating a new column\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.863087Z","iopub.execute_input":"2024-01-14T19:22:38.863869Z","iopub.status.idle":"2024-01-14T19:22:38.906434Z","shell.execute_reply.started":"2024-01-14T19:22:38.863839Z","shell.execute_reply":"2024-01-14T19:22:38.905670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dropping the 'speed' column from the DataFrame\n\ndata.drop('speed', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.907505Z","iopub.execute_input":"2024-01-14T19:22:38.908390Z","iopub.status.idle":"2024-01-14T19:22:38.927466Z","shell.execute_reply.started":"2024-01-14T19:22:38.908345Z","shell.execute_reply":"2024-01-14T19:22:38.926024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the number of unique values in the 'secondary_labels' column\ndata['secondary_labels'].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.929008Z","iopub.execute_input":"2024-01-14T19:22:38.929381Z","iopub.status.idle":"2024-01-14T19:22:38.945975Z","shell.execute_reply.started":"2024-01-14T19:22:38.929347Z","shell.execute_reply":"2024-01-14T19:22:38.944707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the number of unique values in the 'title' column\ndata['title'].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.947415Z","iopub.execute_input":"2024-01-14T19:22:38.947960Z","iopub.status.idle":"2024-01-14T19:22:38.966240Z","shell.execute_reply.started":"2024-01-14T19:22:38.947928Z","shell.execute_reply":"2024-01-14T19:22:38.965169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the occurrences of each unique value in the 'bird_seen' column\n\ndata['bird_seen'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.967864Z","iopub.execute_input":"2024-01-14T19:22:38.968838Z","iopub.status.idle":"2024-01-14T19:22:38.983446Z","shell.execute_reply.started":"2024-01-14T19:22:38.968792Z","shell.execute_reply":"2024-01-14T19:22:38.982232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filling missing values in the 'bird_seen' column with 'yes'\n\ndata['bird_seen'].fillna('yes', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.985036Z","iopub.execute_input":"2024-01-14T19:22:38.985839Z","iopub.status.idle":"2024-01-14T19:22:38.996860Z","shell.execute_reply.started":"2024-01-14T19:22:38.985793Z","shell.execute_reply":"2024-01-14T19:22:38.996016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encoding the 'bird_seen' column using LabelEncoder and creating a new column 'bird_seen_encoded'\n\nle = LabelEncoder()                                     # Creating a LabelEncoder instance\ndata['bird_seen_encoded'] = le.fit_transform(data['bird_seen'])  # Encoding 'bird_seen' and creating a new column","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:38.998499Z","iopub.execute_input":"2024-01-14T19:22:38.999153Z","iopub.status.idle":"2024-01-14T19:22:39.014472Z","shell.execute_reply.started":"2024-01-14T19:22:38.999109Z","shell.execute_reply":"2024-01-14T19:22:39.013445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the occurrences of each unique value in the 'bird_seen_encoded' column after encoding\n\ndata['bird_seen_encoded'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.016068Z","iopub.execute_input":"2024-01-14T19:22:39.016784Z","iopub.status.idle":"2024-01-14T19:22:39.030894Z","shell.execute_reply.started":"2024-01-14T19:22:39.016720Z","shell.execute_reply":"2024-01-14T19:22:39.029763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dropping the 'bird_seen' column from the DataFrame\n\ndata.drop('bird_seen', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.032947Z","iopub.execute_input":"2024-01-14T19:22:39.033478Z","iopub.status.idle":"2024-01-14T19:22:39.056706Z","shell.execute_reply.started":"2024-01-14T19:22:39.033435Z","shell.execute_reply":"2024-01-14T19:22:39.055565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the occurrences of 'Not specified' in the 'latitude' column\n\ncount_not_specified = (data['latitude'] == 'Not specified').sum()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.058819Z","iopub.execute_input":"2024-01-14T19:22:39.059281Z","iopub.status.idle":"2024-01-14T19:22:39.070286Z","shell.execute_reply.started":"2024-01-14T19:22:39.059236Z","shell.execute_reply":"2024-01-14T19:22:39.069017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_not_specified","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.071683Z","iopub.execute_input":"2024-01-14T19:22:39.072053Z","iopub.status.idle":"2024-01-14T19:22:39.082570Z","shell.execute_reply.started":"2024-01-14T19:22:39.072022Z","shell.execute_reply":"2024-01-14T19:22:39.081475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the occurrences of 'Not specified' in the 'longitude' column\n\ncount_not_specified_longitude = (data['longitude'] == 'Not specified').sum()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.084132Z","iopub.execute_input":"2024-01-14T19:22:39.084448Z","iopub.status.idle":"2024-01-14T19:22:39.096991Z","shell.execute_reply.started":"2024-01-14T19:22:39.084420Z","shell.execute_reply":"2024-01-14T19:22:39.096078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_not_specified_longitude","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.098298Z","iopub.execute_input":"2024-01-14T19:22:39.098639Z","iopub.status.idle":"2024-01-14T19:22:39.110712Z","shell.execute_reply.started":"2024-01-14T19:22:39.098610Z","shell.execute_reply":"2024-01-14T19:22:39.109450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dropping rows where latitude or longitude is 'Not specified'\ndata = data[(data['latitude'] != 'Not specified') & (data['longitude'] != 'Not specified')]","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.112296Z","iopub.execute_input":"2024-01-14T19:22:39.112652Z","iopub.status.idle":"2024-01-14T19:22:39.141357Z","shell.execute_reply.started":"2024-01-14T19:22:39.112623Z","shell.execute_reply":"2024-01-14T19:22:39.140273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the occurrences of 'Not specified' in the 'longitude' column after removing those entries\n\ncount_not_specified_longitude1 = (data['longitude'] == 'Not specified').sum()\ncount_not_specified_longitude1","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.143052Z","iopub.execute_input":"2024-01-14T19:22:39.143373Z","iopub.status.idle":"2024-01-14T19:22:39.155108Z","shell.execute_reply.started":"2024-01-14T19:22:39.143345Z","shell.execute_reply":"2024-01-14T19:22:39.153989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the occurrences of 'Not specified' in the 'latitude' column after removing those entries\n\ncount_not_specified_latitude1 = (data['latitude'] == 'Not specified').sum()\ncount_not_specified_latitude1","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.156515Z","iopub.execute_input":"2024-01-14T19:22:39.156942Z","iopub.status.idle":"2024-01-14T19:22:39.169562Z","shell.execute_reply.started":"2024-01-14T19:22:39.156911Z","shell.execute_reply":"2024-01-14T19:22:39.168726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Converting the 'latitude' and 'longitude' columns to float data type using .loc\n\ndata.loc[:, 'latitude'] = data['latitude'].astype(float)\ndata.loc[:, 'longitude'] = data['longitude'].astype(float)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.170926Z","iopub.execute_input":"2024-01-14T19:22:39.171379Z","iopub.status.idle":"2024-01-14T19:22:39.191375Z","shell.execute_reply.started":"2024-01-14T19:22:39.171246Z","shell.execute_reply":"2024-01-14T19:22:39.190150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Counting the occurrences of each unique value in the 'sampling_rate' column\n\ndata['sampling_rate'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.192831Z","iopub.execute_input":"2024-01-14T19:22:39.193621Z","iopub.status.idle":"2024-01-14T19:22:39.206391Z","shell.execute_reply.started":"2024-01-14T19:22:39.193587Z","shell.execute_reply":"2024-01-14T19:22:39.205322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extracting the numerical part of the 'sampling_rate' column and converting to integers using .loc\n\ndata.loc[:, 'sampling_rate'] = data['sampling_rate'].apply(lambda x: str(x).split(' ')[0]).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.208242Z","iopub.execute_input":"2024-01-14T19:22:39.208593Z","iopub.status.idle":"2024-01-14T19:22:39.241365Z","shell.execute_reply.started":"2024-01-14T19:22:39.208562Z","shell.execute_reply":"2024-01-14T19:22:39.240384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['type'].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.242708Z","iopub.execute_input":"2024-01-14T19:22:39.243252Z","iopub.status.idle":"2024-01-14T19:22:39.256983Z","shell.execute_reply.started":"2024-01-14T19:22:39.243217Z","shell.execute_reply":"2024-01-14T19:22:39.255827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['elevation'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.258634Z","iopub.execute_input":"2024-01-14T19:22:39.259586Z","iopub.status.idle":"2024-01-14T19:22:39.269326Z","shell.execute_reply.started":"2024-01-14T19:22:39.259553Z","shell.execute_reply":"2024-01-14T19:22:39.268336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Replacing specified strings in the 'elevation' column with '9999' using .loc\n\ndata.loc[:, 'elevation'] = data['elevation'].replace(['? m', 'Unknown m', ' m'], '9999')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.270747Z","iopub.execute_input":"2024-01-14T19:22:39.271285Z","iopub.status.idle":"2024-01-14T19:22:39.287276Z","shell.execute_reply.started":"2024-01-14T19:22:39.271253Z","shell.execute_reply":"2024-01-14T19:22:39.286033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.loc[:, 'elevation']=data['elevation'].str.replace(',','')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.289280Z","iopub.execute_input":"2024-01-14T19:22:39.290146Z","iopub.status.idle":"2024-01-14T19:22:39.309623Z","shell.execute_reply.started":"2024-01-14T19:22:39.290110Z","shell.execute_reply":"2024-01-14T19:22:39.308436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.loc[:, 'elevation']=data['elevation'].str.replace('~','')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.311044Z","iopub.execute_input":"2024-01-14T19:22:39.311381Z","iopub.status.idle":"2024-01-14T19:22:39.332136Z","shell.execute_reply.started":"2024-01-14T19:22:39.311351Z","shell.execute_reply":"2024-01-14T19:22:39.331136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.loc[:, 'elevation']=data['elevation'].str.replace('.','')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.333812Z","iopub.execute_input":"2024-01-14T19:22:39.334455Z","iopub.status.idle":"2024-01-14T19:22:39.356145Z","shell.execute_reply.started":"2024-01-14T19:22:39.334404Z","shell.execute_reply":"2024-01-14T19:22:39.355106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.loc[:, 'elevation']=data['elevation'].str.replace('?? m','9999')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.357483Z","iopub.execute_input":"2024-01-14T19:22:39.358393Z","iopub.status.idle":"2024-01-14T19:22:39.380249Z","shell.execute_reply.started":"2024-01-14T19:22:39.358361Z","shell.execute_reply":"2024-01-14T19:22:39.379317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.loc[:, 'elevation']=data['elevation'].str.replace('1650-1900 m','1775')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.381409Z","iopub.execute_input":"2024-01-14T19:22:39.382366Z","iopub.status.idle":"2024-01-14T19:22:39.400134Z","shell.execute_reply.started":"2024-01-14T19:22:39.382332Z","shell.execute_reply":"2024-01-14T19:22:39.399239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.loc[:, 'elevation']=data['elevation'].str.replace('- m','9999')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.401283Z","iopub.execute_input":"2024-01-14T19:22:39.401938Z","iopub.status.idle":"2024-01-14T19:22:39.426482Z","shell.execute_reply.started":"2024-01-14T19:22:39.401906Z","shell.execute_reply":"2024-01-14T19:22:39.425282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.loc[:, 'elevation']=data['elevation'].str.replace('930-990 m','9999')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.428009Z","iopub.execute_input":"2024-01-14T19:22:39.428375Z","iopub.status.idle":"2024-01-14T19:22:39.452367Z","shell.execute_reply.started":"2024-01-14T19:22:39.428344Z","shell.execute_reply":"2024-01-14T19:22:39.451304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.loc[:, 'elevation']=data['elevation'].str.replace('1400m','1400')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.473394Z","iopub.execute_input":"2024-01-14T19:22:39.473925Z","iopub.status.idle":"2024-01-14T19:22:39.494747Z","shell.execute_reply.started":"2024-01-14T19:22:39.473890Z","shell.execute_reply":"2024-01-14T19:22:39.492903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.loc[:, 'elevation']=data['elevation'].str.replace('1900m','1900')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.496748Z","iopub.execute_input":"2024-01-14T19:22:39.497239Z","iopub.status.idle":"2024-01-14T19:22:39.520416Z","shell.execute_reply.started":"2024-01-14T19:22:39.497194Z","shell.execute_reply":"2024-01-14T19:22:39.519407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extracting the numerical part of the 'elevation' column and converting to integers\n\ndata.loc[:, 'elevation'] = data['elevation'].apply(lambda x: str(x).split(\" \")[0]).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.522040Z","iopub.execute_input":"2024-01-14T19:22:39.522493Z","iopub.status.idle":"2024-01-14T19:22:39.554847Z","shell.execute_reply.started":"2024-01-14T19:22:39.522440Z","shell.execute_reply":"2024-01-14T19:22:39.553778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['elevation'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.556389Z","iopub.execute_input":"2024-01-14T19:22:39.556944Z","iopub.status.idle":"2024-01-14T19:22:39.569549Z","shell.execute_reply.started":"2024-01-14T19:22:39.556899Z","shell.execute_reply":"2024-01-14T19:22:39.568326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Taking the absolute value of the 'elevation' column\ndata.loc[:, 'elevation'] = data['elevation'].abs()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.571222Z","iopub.execute_input":"2024-01-14T19:22:39.571609Z","iopub.status.idle":"2024-01-14T19:22:39.580361Z","shell.execute_reply.started":"2024-01-14T19:22:39.571577Z","shell.execute_reply":"2024-01-14T19:22:39.579486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['elevation'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.581291Z","iopub.execute_input":"2024-01-14T19:22:39.581599Z","iopub.status.idle":"2024-01-14T19:22:39.595607Z","shell.execute_reply.started":"2024-01-14T19:22:39.581571Z","shell.execute_reply":"2024-01-14T19:22:39.594114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert 'elevation' column to numeric, treating '9999' as NaN\n\ndata.loc[:, 'elevation']=pd.to_numeric(data['elevation'],errors='coerce')\nsum_elevation=0\ncount_valid_values=0\n\nfor index,row in data.iterrows():\n    if row['elevation']!=9999:\n        sum_elevation+=row['elevation']\n        count_valid_values+=1\nmean_elevation =sum_elevation/count_valid_values\nmean_elevation","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:39.597195Z","iopub.execute_input":"2024-01-14T19:22:39.597539Z","iopub.status.idle":"2024-01-14T19:22:41.193515Z","shell.execute_reply.started":"2024-01-14T19:22:39.597510Z","shell.execute_reply":"2024-01-14T19:22:41.192238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Replace '9999' with 667 in the 'elevation' column\n\ndata.loc[:, 'elevation'] = data['elevation'].replace(['9999'], 667)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.194832Z","iopub.execute_input":"2024-01-14T19:22:41.195141Z","iopub.status.idle":"2024-01-14T19:22:41.217295Z","shell.execute_reply.started":"2024-01-14T19:22:41.195114Z","shell.execute_reply":"2024-01-14T19:22:41.216019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['elevation'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.219065Z","iopub.execute_input":"2024-01-14T19:22:41.219520Z","iopub.status.idle":"2024-01-14T19:22:41.231010Z","shell.execute_reply.started":"2024-01-14T19:22:41.219478Z","shell.execute_reply":"2024-01-14T19:22:41.229686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.drop(\"description\" , axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.232488Z","iopub.execute_input":"2024-01-14T19:22:41.233549Z","iopub.status.idle":"2024-01-14T19:22:41.250465Z","shell.execute_reply.started":"2024-01-14T19:22:41.233517Z","shell.execute_reply":"2024-01-14T19:22:41.249198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['bitrate_of_mp3'].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.252304Z","iopub.execute_input":"2024-01-14T19:22:41.252793Z","iopub.status.idle":"2024-01-14T19:22:41.263535Z","shell.execute_reply.started":"2024-01-14T19:22:41.252731Z","shell.execute_reply":"2024-01-14T19:22:41.262365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['bitrate_of_mp3'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.265080Z","iopub.execute_input":"2024-01-14T19:22:41.265529Z","iopub.status.idle":"2024-01-14T19:22:41.282205Z","shell.execute_reply.started":"2024-01-14T19:22:41.265495Z","shell.execute_reply":"2024-01-14T19:22:41.280988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You can clearly see that the number of samples having a bitrate 128000 is fairly large as compared to the second highest bitrate 320000 so we can replace the null values with a bitrate of 128000.","metadata":{}},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.283599Z","iopub.execute_input":"2024-01-14T19:22:41.283941Z","iopub.status.idle":"2024-01-14T19:22:41.358910Z","shell.execute_reply.started":"2024-01-14T19:22:41.283909Z","shell.execute_reply":"2024-01-14T19:22:41.357825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the number of null values in the 'bitrate_of_mp3' column\n\nnull_count = data['bitrate_of_mp3'].isnull().sum()\nprint(\"Number of null values in 'bitrate_of_mp3' column:\", null_count)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.360330Z","iopub.execute_input":"2024-01-14T19:22:41.360678Z","iopub.status.idle":"2024-01-14T19:22:41.370679Z","shell.execute_reply.started":"2024-01-14T19:22:41.360648Z","shell.execute_reply":"2024-01-14T19:22:41.369567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['bitrate_of_mp3'].fillna('128000', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.372002Z","iopub.execute_input":"2024-01-14T19:22:41.372298Z","iopub.status.idle":"2024-01-14T19:22:41.383005Z","shell.execute_reply.started":"2024-01-14T19:22:41.372272Z","shell.execute_reply":"2024-01-14T19:22:41.382138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['bitrate_of_mp3']=data['bitrate_of_mp3'].apply(lambda x: x.split(\" \")[0]).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.384334Z","iopub.execute_input":"2024-01-14T19:22:41.384677Z","iopub.status.idle":"2024-01-14T19:22:41.411017Z","shell.execute_reply.started":"2024-01-14T19:22:41.384647Z","shell.execute_reply":"2024-01-14T19:22:41.409912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['bitrate_of_mp3'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.412328Z","iopub.execute_input":"2024-01-14T19:22:41.412655Z","iopub.status.idle":"2024-01-14T19:22:41.428889Z","shell.execute_reply.started":"2024-01-14T19:22:41.412626Z","shell.execute_reply":"2024-01-14T19:22:41.427465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.430307Z","iopub.execute_input":"2024-01-14T19:22:41.431184Z","iopub.status.idle":"2024-01-14T19:22:41.501797Z","shell.execute_reply.started":"2024-01-14T19:22:41.431151Z","shell.execute_reply":"2024-01-14T19:22:41.500590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['file_type'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.503279Z","iopub.execute_input":"2024-01-14T19:22:41.503620Z","iopub.status.idle":"2024-01-14T19:22:41.516150Z","shell.execute_reply.started":"2024-01-14T19:22:41.503590Z","shell.execute_reply":"2024-01-14T19:22:41.514964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove rows where 'file_type' is 'wav', 'mp2', or 'aac'\n\ndata = data[~data['file_type'].isin(['wav', 'mp2', 'aac'])]","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.517431Z","iopub.execute_input":"2024-01-14T19:22:41.517740Z","iopub.status.idle":"2024-01-14T19:22:41.544120Z","shell.execute_reply.started":"2024-01-14T19:22:41.517713Z","shell.execute_reply":"2024-01-14T19:22:41.543001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.drop('file_type', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.545625Z","iopub.execute_input":"2024-01-14T19:22:41.546075Z","iopub.status.idle":"2024-01-14T19:22:41.563787Z","shell.execute_reply.started":"2024-01-14T19:22:41.546032Z","shell.execute_reply":"2024-01-14T19:22:41.562587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['volume'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.565192Z","iopub.execute_input":"2024-01-14T19:22:41.565541Z","iopub.status.idle":"2024-01-14T19:22:41.579026Z","shell.execute_reply.started":"2024-01-14T19:22:41.565510Z","shell.execute_reply":"2024-01-14T19:22:41.577746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le=LabelEncoder()\ndata['volume_encoded'] = le.fit_transform(data['volume'])","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.580691Z","iopub.execute_input":"2024-01-14T19:22:41.581106Z","iopub.status.idle":"2024-01-14T19:22:41.594950Z","shell.execute_reply.started":"2024-01-14T19:22:41.581074Z","shell.execute_reply":"2024-01-14T19:22:41.593914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.596472Z","iopub.execute_input":"2024-01-14T19:22:41.596862Z","iopub.status.idle":"2024-01-14T19:22:41.632610Z","shell.execute_reply.started":"2024-01-14T19:22:41.596830Z","shell.execute_reply":"2024-01-14T19:22:41.631524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.drop('volume', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.634245Z","iopub.execute_input":"2024-01-14T19:22:41.634950Z","iopub.status.idle":"2024-01-14T19:22:41.647746Z","shell.execute_reply.started":"2024-01-14T19:22:41.634916Z","shell.execute_reply":"2024-01-14T19:22:41.646442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[\"background\"].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.649345Z","iopub.execute_input":"2024-01-14T19:22:41.650386Z","iopub.status.idle":"2024-01-14T19:22:41.660191Z","shell.execute_reply.started":"2024-01-14T19:22:41.650353Z","shell.execute_reply":"2024-01-14T19:22:41.658978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.drop('background', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.661887Z","iopub.execute_input":"2024-01-14T19:22:41.662284Z","iopub.status.idle":"2024-01-14T19:22:41.680658Z","shell.execute_reply.started":"2024-01-14T19:22:41.662253Z","shell.execute_reply":"2024-01-14T19:22:41.679341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['xc_id'].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.682176Z","iopub.execute_input":"2024-01-14T19:22:41.682506Z","iopub.status.idle":"2024-01-14T19:22:41.691524Z","shell.execute_reply.started":"2024-01-14T19:22:41.682477Z","shell.execute_reply":"2024-01-14T19:22:41.690304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['url'].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.693065Z","iopub.execute_input":"2024-01-14T19:22:41.693508Z","iopub.status.idle":"2024-01-14T19:22:41.710982Z","shell.execute_reply.started":"2024-01-14T19:22:41.693454Z","shell.execute_reply":"2024-01-14T19:22:41.710117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.drop('xc_id', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.712241Z","iopub.execute_input":"2024-01-14T19:22:41.712549Z","iopub.status.idle":"2024-01-14T19:22:41.727343Z","shell.execute_reply.started":"2024-01-14T19:22:41.712522Z","shell.execute_reply":"2024-01-14T19:22:41.726496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.drop('url', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.728802Z","iopub.execute_input":"2024-01-14T19:22:41.729446Z","iopub.status.idle":"2024-01-14T19:22:41.743292Z","shell.execute_reply.started":"2024-01-14T19:22:41.729414Z","shell.execute_reply":"2024-01-14T19:22:41.742276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['country'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.744623Z","iopub.execute_input":"2024-01-14T19:22:41.745438Z","iopub.status.idle":"2024-01-14T19:22:41.754631Z","shell.execute_reply.started":"2024-01-14T19:22:41.745406Z","shell.execute_reply":"2024-01-14T19:22:41.753575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['country'].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.756140Z","iopub.execute_input":"2024-01-14T19:22:41.757141Z","iopub.status.idle":"2024-01-14T19:22:41.773177Z","shell.execute_reply.started":"2024-01-14T19:22:41.757108Z","shell.execute_reply":"2024-01-14T19:22:41.771863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using LabelEncoder to encode the 'country' column for the model\n# Creating a LabelEncoder instance\nle = LabelEncoder()\n\n# Applying fit_transform to convert country names into numerical labels\ndata['country_encoded'] = le.fit_transform(data['country'])","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.775030Z","iopub.execute_input":"2024-01-14T19:22:41.775919Z","iopub.status.idle":"2024-01-14T19:22:41.788127Z","shell.execute_reply.started":"2024-01-14T19:22:41.775887Z","shell.execute_reply":"2024-01-14T19:22:41.787140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['author'].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.789457Z","iopub.execute_input":"2024-01-14T19:22:41.790308Z","iopub.status.idle":"2024-01-14T19:22:41.802703Z","shell.execute_reply.started":"2024-01-14T19:22:41.790274Z","shell.execute_reply":"2024-01-14T19:22:41.801683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['recordist'].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.804412Z","iopub.execute_input":"2024-01-14T19:22:41.805537Z","iopub.status.idle":"2024-01-14T19:22:41.815619Z","shell.execute_reply.started":"2024-01-14T19:22:41.805502Z","shell.execute_reply":"2024-01-14T19:22:41.814361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filtering rows where 'author' is not equal to 'recordist' in the DataFrame\n\nmismatched_records = data[data['author'] != data['recordist']]","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.817401Z","iopub.execute_input":"2024-01-14T19:22:41.818424Z","iopub.status.idle":"2024-01-14T19:22:41.831064Z","shell.execute_reply.started":"2024-01-14T19:22:41.818389Z","shell.execute_reply":"2024-01-14T19:22:41.829985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.drop('author', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.832460Z","iopub.execute_input":"2024-01-14T19:22:41.833688Z","iopub.status.idle":"2024-01-14T19:22:41.850117Z","shell.execute_reply.started":"2024-01-14T19:22:41.833653Z","shell.execute_reply":"2024-01-14T19:22:41.848973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['primary_label'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.851290Z","iopub.execute_input":"2024-01-14T19:22:41.851604Z","iopub.status.idle":"2024-01-14T19:22:41.863889Z","shell.execute_reply.started":"2024-01-14T19:22:41.851576Z","shell.execute_reply":"2024-01-14T19:22:41.863005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dropping the 'primary_label' column from the DataFrame\n\ndata.drop('primary_label', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.864956Z","iopub.execute_input":"2024-01-14T19:22:41.865272Z","iopub.status.idle":"2024-01-14T19:22:41.881430Z","shell.execute_reply.started":"2024-01-14T19:22:41.865244Z","shell.execute_reply":"2024-01-14T19:22:41.880158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['length'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.883109Z","iopub.execute_input":"2024-01-14T19:22:41.883525Z","iopub.status.idle":"2024-01-14T19:22:41.893503Z","shell.execute_reply.started":"2024-01-14T19:22:41.883494Z","shell.execute_reply":"2024-01-14T19:22:41.892263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['length'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.895509Z","iopub.execute_input":"2024-01-14T19:22:41.896003Z","iopub.status.idle":"2024-01-14T19:22:41.916068Z","shell.execute_reply.started":"2024-01-14T19:22:41.895940Z","shell.execute_reply":"2024-01-14T19:22:41.914867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extracting numeric values from the 'length' column and creating a new 'min_value' column\ndata['min_value'] = data['length'].str.extract(r'(\\d+)')\n\n# Converting the 'min_value' column to numeric format, treating non-numeric values as NaN\ndata['min_value'] = pd.to_numeric(data['min_value'], errors='coerce')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.917722Z","iopub.execute_input":"2024-01-14T19:22:41.918456Z","iopub.status.idle":"2024-01-14T19:22:41.980185Z","shell.execute_reply.started":"2024-01-14T19:22:41.918400Z","shell.execute_reply":"2024-01-14T19:22:41.979077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extracting numeric values after '-' from the 'length' column and creating a new 'max_value' column\ndata['max_value'] = data['length'].str.extract(r'-(\\d+)')\n\n# Converting the 'max_value' column to numeric format, treating non-numeric values as NaN\ndata['max_value'] = pd.to_numeric(data['max_value'], errors='coerce')","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:41.981675Z","iopub.execute_input":"2024-01-14T19:22:41.982062Z","iopub.status.idle":"2024-01-14T19:22:42.024238Z","shell.execute_reply.started":"2024-01-14T19:22:41.982029Z","shell.execute_reply":"2024-01-14T19:22:42.023163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.025417Z","iopub.execute_input":"2024-01-14T19:22:42.025739Z","iopub.status.idle":"2024-01-14T19:22:42.058482Z","shell.execute_reply.started":"2024-01-14T19:22:42.025711Z","shell.execute_reply":"2024-01-14T19:22:42.057240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['min_value'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.060107Z","iopub.execute_input":"2024-01-14T19:22:42.060635Z","iopub.status.idle":"2024-01-14T19:22:42.071759Z","shell.execute_reply.started":"2024-01-14T19:22:42.060593Z","shell.execute_reply":"2024-01-14T19:22:42.070572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['max_value'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.073427Z","iopub.execute_input":"2024-01-14T19:22:42.073985Z","iopub.status.idle":"2024-01-14T19:22:42.084971Z","shell.execute_reply.started":"2024-01-14T19:22:42.073933Z","shell.execute_reply":"2024-01-14T19:22:42.083817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.drop('length', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.086691Z","iopub.execute_input":"2024-01-14T19:22:42.087061Z","iopub.status.idle":"2024-01-14T19:22:42.101832Z","shell.execute_reply.started":"2024-01-14T19:22:42.087031Z","shell.execute_reply":"2024-01-14T19:22:42.100846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.drop('min_value', axis=1, inplace=True)\ndata.drop('max_value', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.103593Z","iopub.execute_input":"2024-01-14T19:22:42.103991Z","iopub.status.idle":"2024-01-14T19:22:42.129033Z","shell.execute_reply.started":"2024-01-14T19:22:42.103948Z","shell.execute_reply":"2024-01-14T19:22:42.128131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['time'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.130301Z","iopub.execute_input":"2024-01-14T19:22:42.130852Z","iopub.status.idle":"2024-01-14T19:22:42.140147Z","shell.execute_reply.started":"2024-01-14T19:22:42.130820Z","shell.execute_reply":"2024-01-14T19:22:42.138840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['time'].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.141930Z","iopub.execute_input":"2024-01-14T19:22:42.142419Z","iopub.status.idle":"2024-01-14T19:22:42.153194Z","shell.execute_reply.started":"2024-01-14T19:22:42.142377Z","shell.execute_reply":"2024-01-14T19:22:42.152092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.drop('time', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.154760Z","iopub.execute_input":"2024-01-14T19:22:42.155267Z","iopub.status.idle":"2024-01-14T19:22:42.172027Z","shell.execute_reply.started":"2024-01-14T19:22:42.155224Z","shell.execute_reply":"2024-01-14T19:22:42.170869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['license'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.173521Z","iopub.execute_input":"2024-01-14T19:22:42.173968Z","iopub.status.idle":"2024-01-14T19:22:42.188542Z","shell.execute_reply.started":"2024-01-14T19:22:42.173920Z","shell.execute_reply":"2024-01-14T19:22:42.187288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['license'].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.190713Z","iopub.execute_input":"2024-01-14T19:22:42.191106Z","iopub.status.idle":"2024-01-14T19:22:42.201084Z","shell.execute_reply.started":"2024-01-14T19:22:42.191074Z","shell.execute_reply":"2024-01-14T19:22:42.199830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using LabelEncoder to encode the 'license' column and exploring the encoding results\n# Creating a LabelEncoder instance\nle = LabelEncoder()\n\n# Applying fit_transform to convert license names into numerical labels\ndata['license_encoded'] = le.fit_transform(data['license'])\n\n# Extracting unique values from the 'license_encoded' column\nencoded_unique_values = data['license_encoded'].unique()\n\n# Creating a label mapping dictionary using zip with 'license' and 'license_encoded'\nlabel_mapping = dict(zip(data['license'], data['license_encoded']))\n\n# Displaying the label encoded unique values and label mapping\nprint(\"Label Encoded Unique Values:\", encoded_unique_values)\nprint(\"Label Mapping:\", label_mapping)\n","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.202691Z","iopub.execute_input":"2024-01-14T19:22:42.203077Z","iopub.status.idle":"2024-01-14T19:22:42.228649Z","shell.execute_reply.started":"2024-01-14T19:22:42.203047Z","shell.execute_reply":"2024-01-14T19:22:42.227523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.drop('license', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.230194Z","iopub.execute_input":"2024-01-14T19:22:42.230543Z","iopub.status.idle":"2024-01-14T19:22:42.245633Z","shell.execute_reply.started":"2024-01-14T19:22:42.230513Z","shell.execute_reply":"2024-01-14T19:22:42.244482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.247001Z","iopub.execute_input":"2024-01-14T19:22:42.247331Z","iopub.status.idle":"2024-01-14T19:22:42.274941Z","shell.execute_reply.started":"2024-01-14T19:22:42.247302Z","shell.execute_reply":"2024-01-14T19:22:42.273662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.276525Z","iopub.execute_input":"2024-01-14T19:22:42.276968Z","iopub.status.idle":"2024-01-14T19:22:42.329752Z","shell.execute_reply.started":"2024-01-14T19:22:42.276931Z","shell.execute_reply":"2024-01-14T19:22:42.328580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['type'].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.331629Z","iopub.execute_input":"2024-01-14T19:22:42.331987Z","iopub.status.idle":"2024-01-14T19:22:42.341214Z","shell.execute_reply.started":"2024-01-14T19:22:42.331955Z","shell.execute_reply":"2024-01-14T19:22:42.340030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['type'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.343112Z","iopub.execute_input":"2024-01-14T19:22:42.343527Z","iopub.status.idle":"2024-01-14T19:22:42.357021Z","shell.execute_reply.started":"2024-01-14T19:22:42.343489Z","shell.execute_reply":"2024-01-14T19:22:42.355804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.drop('type', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.358519Z","iopub.execute_input":"2024-01-14T19:22:42.358893Z","iopub.status.idle":"2024-01-14T19:22:42.372502Z","shell.execute_reply.started":"2024-01-14T19:22:42.358850Z","shell.execute_reply":"2024-01-14T19:22:42.371345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['number_of_notes'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.374596Z","iopub.execute_input":"2024-01-14T19:22:42.374957Z","iopub.status.idle":"2024-01-14T19:22:42.383821Z","shell.execute_reply.started":"2024-01-14T19:22:42.374927Z","shell.execute_reply":"2024-01-14T19:22:42.382627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['number_of_notes'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.385681Z","iopub.execute_input":"2024-01-14T19:22:42.386129Z","iopub.status.idle":"2024-01-14T19:22:42.402337Z","shell.execute_reply.started":"2024-01-14T19:22:42.386086Z","shell.execute_reply":"2024-01-14T19:22:42.401093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.drop('number_of_notes', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.403664Z","iopub.execute_input":"2024-01-14T19:22:42.404110Z","iopub.status.idle":"2024-01-14T19:22:42.419271Z","shell.execute_reply.started":"2024-01-14T19:22:42.404077Z","shell.execute_reply":"2024-01-14T19:22:42.418122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.to_csv('/kaggle/working/output_file.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.420917Z","iopub.execute_input":"2024-01-14T19:22:42.421657Z","iopub.status.idle":"2024-01-14T19:22:42.917426Z","shell.execute_reply.started":"2024-01-14T19:22:42.421615Z","shell.execute_reply":"2024-01-14T19:22:42.916488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a count plot to visualize the distribution of records across different months\n# Using seaborn's countplot with 'month' on the x-axis and data from the DataFrame 'data'\nsns.countplot(x='month', data=data)\n\n# Displaying the count plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:42.918637Z","iopub.execute_input":"2024-01-14T19:22:42.919006Z","iopub.status.idle":"2024-01-14T19:22:43.310938Z","shell.execute_reply.started":"2024-01-14T19:22:42.918973Z","shell.execute_reply":"2024-01-14T19:22:43.309810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The majority of Bird Recordings are typically captured during the months of May and June.","metadata":{}},{"cell_type":"code","source":"# Creating a pie chart to visualize the distribution of records across the top 10 countries\n# Using groupby to count the number of records for each country and selecting the top 10\ntop_countries = data.groupby('country').size().sort_values(ascending=False).head(10)\n\n# Creating a pie chart with the specified parameters\nplt.pie(top_countries, labels=None, autopct=None, startangle=90, colors=sns.color_palette('viridis'))\n\n# Adding legend with country names and their respective percentages\nplt.legend(labels=top_countries.index + ' (' + top_countries.map(lambda x: f'{x/sum(top_countries)*100:.1f}%') + ')', loc='upper left', bbox_to_anchor=(1, 1))\n\n# Adding title to the pie chart\nplt.title('Top 10 Countries with Highest Records')\n\n# Displaying the pie chart\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:43.312671Z","iopub.execute_input":"2024-01-14T19:22:43.313170Z","iopub.status.idle":"2024-01-14T19:22:43.644736Z","shell.execute_reply.started":"2024-01-14T19:22:43.313125Z","shell.execute_reply":"2024-01-14T19:22:43.643940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It is evident that a significant portion of the records pertains to birds from the United States.","metadata":{}},{"cell_type":"code","source":"# Setting the Seaborn theme to 'whitegrid'\nsns.set_theme(style='whitegrid')\n\n# Assuming 'latitude' is the column you want to plot\nplt.figure(figsize=(10, 6))\n\n# Create the histogram with a filled color\nsns.histplot(data['latitude'], bins=20, kde=False, color='#3498db', edgecolor='#2980b9', linewidth=0.5)\n\n# Adding labels and title with a touch of style\nplt.xlabel('Latitude', fontsize=14, labelpad=10, color='#34495e')\nplt.ylabel('Frequency', fontsize=14, labelpad=10, color='#34495e')\nplt.title('Histogram of Latitude', fontsize=16, pad=20, fontweight='bold', color='#34495e')\n\n# Set background color\nplt.gca().set_facecolor('#ecf0f1')\n\n# Adding grid lines for better readability\nplt.grid(axis='y', linestyle='--', alpha=0.7)\n\n# Removing the top and right spines for aesthetics\nsns.despine()\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:43.646118Z","iopub.execute_input":"2024-01-14T19:22:43.647200Z","iopub.status.idle":"2024-01-14T19:22:44.172731Z","shell.execute_reply.started":"2024-01-14T19:22:43.647153Z","shell.execute_reply":"2024-01-14T19:22:44.171840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The plot indicates that the majority of birds are concentrated within the latitude region of 30-60 degrees.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\n\n# Creating a scatter plot of Latitude vs Longitude\nsns.scatterplot(x='longitude', y='latitude', data=data, color='#3498db', alpha=0.7, edgecolor='black', s=15)\n\n# Adding labels and title with a touch of style\nplt.xlabel('Longitude', fontsize=14, labelpad=10, color='#34495e')\nplt.ylabel('Latitude', fontsize=14, labelpad=10, color='#34495e')\nplt.title('Scatter Plot: Latitude vs Longitude', fontsize=16, pad=20, fontweight='bold', color='#34495e')\n\n# Set background color\nplt.gca().set_facecolor('#ecf0f1')\n\n# Adding grid lines for better readability\nplt.grid(axis='both', linestyle='--', alpha=0.7)\n\n# Removing spines for aesthetics\nsns.despine()\n\n# Show the plot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:44.174331Z","iopub.execute_input":"2024-01-14T19:22:44.174852Z","iopub.status.idle":"2024-01-14T19:22:44.703634Z","shell.execute_reply.started":"2024-01-14T19:22:44.174796Z","shell.execute_reply":"2024-01-14T19:22:44.702310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The plot above unmistakably indicates that the United States possesses the most favorable ecological balance.","metadata":{}},{"cell_type":"markdown","source":"Let's proceed with feature extraction","metadata":{}},{"cell_type":"code","source":"import librosa\nimport librosa.display\nimport soundfile as sf\nimport matplotlib.pyplot as plt\nimport IPython.display as ipd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:44.704878Z","iopub.execute_input":"2024-01-14T19:22:44.705209Z","iopub.status.idle":"2024-01-14T19:22:44.711283Z","shell.execute_reply.started":"2024-01-14T19:22:44.705180Z","shell.execute_reply":"2024-01-14T19:22:44.710143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amecro1=\"/kaggle/input/birdsong-recognition/train_audio/amecro/XC114551.mp3\"\namecro2=\"/kaggle/input/birdsong-recognition/train_audio/amecro/XC114552.mp3\"\ncoohaw1=\"/kaggle/input/birdsong-recognition/train_audio/coohaw/XC123581.mp3\"\ncoohaw2=\"/kaggle/input/birdsong-recognition/train_audio/coohaw/XC123582.mp3\"\nmerlin1=\"/kaggle/input/birdsong-recognition/train_audio/merlin/XC145140.mp3\"\nmerlin2=\"/kaggle/input/birdsong-recognition/train_audio/merlin/XC137975.mp3\"\nveery1=\"/kaggle/input/birdsong-recognition/train_audio/veery/XC142682.mp3\"\nveery2=\"/kaggle/input/birdsong-recognition/train_audio/veery/XC120869.mp3\"\naldfly1=\"/kaggle/input/birdsong-recognition/train_audio/aldfly/XC137570.mp3\"\naldfly2=\"/kaggle/input/birdsong-recognition/train_audio/aldfly/XC142068.mp3\"","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:44.712408Z","iopub.execute_input":"2024-01-14T19:22:44.712714Z","iopub.status.idle":"2024-01-14T19:22:44.724202Z","shell.execute_reply.started":"2024-01-14T19:22:44.712686Z","shell.execute_reply":"2024-01-14T19:22:44.722972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amecro1_audio, sr1 = librosa.load(amecro1)\namecro2_audio, sr2 = librosa.load(amecro2)\ncoohaw1_audio, sr3 = librosa.load(coohaw1)\ncoohaw2_audio, sr4 = librosa.load(coohaw2)\nmerlin1_audio, sr5 = librosa.load(merlin1)\nmerlin2_audio, sr6 = librosa.load(merlin2)\n\nveery1_audio, sr7 = librosa.load(veery1)\nveery2_audio, sr8 = librosa.load(veery2)\naldfly1_audio, sr9 = librosa.load(aldfly1)\naldfly2_audio, sr10 = librosa.load(aldfly2)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:44.725748Z","iopub.execute_input":"2024-01-14T19:22:44.726071Z","iopub.status.idle":"2024-01-14T19:22:45.279302Z","shell.execute_reply.started":"2024-01-14T19:22:44.726042Z","shell.execute_reply":"2024-01-14T19:22:45.278108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amecro_zcr1= librosa.feature.zero_crossing_rate(amecro1_audio)\namecro_zcr2=librosa.feature.zero_crossing_rate(amecro2_audio)\ncoohaw_zcr1=librosa.feature.zero_crossing_rate(coohaw1_audio)\ncoohaw_zcr2=librosa.feature.zero_crossing_rate(coohaw2_audio)\nmerlin_zcr1=librosa.feature.zero_crossing_rate(merlin1_audio)\nmerlin_zcr2=librosa.feature.zero_crossing_rate(merlin2_audio)\nveery_zcr1=librosa.feature.zero_crossing_rate(veery1_audio)\nveery_zcr2=librosa.feature.zero_crossing_rate(veery2_audio)\naldfly_zcr1=librosa.feature.zero_crossing_rate(aldfly1_audio)\naldfly_zcr2=librosa.feature.zero_crossing_rate(aldfly2_audio)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:45.280601Z","iopub.execute_input":"2024-01-14T19:22:45.281152Z","iopub.status.idle":"2024-01-14T19:22:45.384170Z","shell.execute_reply.started":"2024-01-14T19:22:45.281114Z","shell.execute_reply":"2024-01-14T19:22:45.382719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amecro_energy1=librosa.feature.rms(y=amecro1_audio)\namecro_energy2=librosa.feature.rms(y=amecro2_audio)\ncoohaw_energy1=librosa.feature.rms(y=coohaw1_audio)\ncoohaw_energy2=librosa.feature.rms(y=coohaw2_audio)\nmerlin_energy1=librosa.feature.rms(y=merlin1_audio)\nmerlin_energy2=librosa.feature.rms(y=merlin2_audio)\nveery_energy1=librosa.feature.rms(y=veery1_audio)\nveery_energy2=librosa.feature.rms(y=veery2_audio)\naldfly_energy1=librosa.feature.rms(y=aldfly1_audio)\naldfly_energy2=librosa.feature.rms(y=aldfly2_audio)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:45.386230Z","iopub.execute_input":"2024-01-14T19:22:45.386654Z","iopub.status.idle":"2024-01-14T19:22:45.717354Z","shell.execute_reply.started":"2024-01-14T19:22:45.386611Z","shell.execute_reply":"2024-01-14T19:22:45.716325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#lets find the minimum and maximum frequency of these audios\n\nD = librosa.amplitude_to_db(np.abs(librosa.stft(amecro1_audio)), ref=np.max)\nfrequencies = librosa.fft_frequencies(sr=sr1)\nmin_frequency = np.min(frequencies)\nmax_frequency = np.max(frequencies)\nprint(\"Minimum Frequency:\", min_frequency, \"Hz\")\nprint(\"Maximum Frequency:\", max_frequency, \"Hz\")","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:45.718805Z","iopub.execute_input":"2024-01-14T19:22:45.719132Z","iopub.status.idle":"2024-01-14T19:22:45.754010Z","shell.execute_reply.started":"2024-01-14T19:22:45.719103Z","shell.execute_reply":"2024-01-14T19:22:45.752968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"D = librosa.amplitude_to_db(np.abs(librosa.stft(amecro2_audio)), ref=np.max)\nfrequencies = librosa.fft_frequencies(sr=sr2)\nmin_frequency = np.min(frequencies)\nmax_frequency = np.max(frequencies)\nprint(\"Minimum Frequency:\", min_frequency, \"Hz\")\nprint(\"Maximum Frequency:\", max_frequency, \"Hz\")","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:45.755150Z","iopub.execute_input":"2024-01-14T19:22:45.755479Z","iopub.status.idle":"2024-01-14T19:22:45.798176Z","shell.execute_reply.started":"2024-01-14T19:22:45.755443Z","shell.execute_reply":"2024-01-14T19:22:45.796961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"D = librosa.amplitude_to_db(np.abs(librosa.stft(coohaw1_audio)), ref=np.max)\nfrequencies = librosa.fft_frequencies(sr=sr3)\nmin_frequency = np.min(frequencies)\nmax_frequency = np.max(frequencies)\nprint(\"Minimum Frequency:\", min_frequency, \"Hz\")\nprint(\"Maximum Frequency:\", max_frequency, \"Hz\")","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:45.800066Z","iopub.execute_input":"2024-01-14T19:22:45.801161Z","iopub.status.idle":"2024-01-14T19:22:45.860873Z","shell.execute_reply.started":"2024-01-14T19:22:45.801114Z","shell.execute_reply":"2024-01-14T19:22:45.859875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"D = librosa.amplitude_to_db(np.abs(librosa.stft(coohaw2_audio)), ref=np.max)\nfrequencies = librosa.fft_frequencies(sr=sr4)\nmin_frequency = np.min(frequencies)\nmax_frequency = np.max(frequencies)\nprint(\"Minimum Frequency:\", min_frequency, \"Hz\")\nprint(\"Maximum Frequency:\", max_frequency, \"Hz\")","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:45.861925Z","iopub.execute_input":"2024-01-14T19:22:45.862277Z","iopub.status.idle":"2024-01-14T19:22:45.916971Z","shell.execute_reply.started":"2024-01-14T19:22:45.862246Z","shell.execute_reply":"2024-01-14T19:22:45.915791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"D = librosa.amplitude_to_db(np.abs(librosa.stft(merlin1_audio)), ref=np.max)\nfrequencies = librosa.fft_frequencies(sr=sr5)\nmin_frequency = np.min(frequencies)\nmax_frequency = np.max(frequencies)\nprint(\"Minimum Frequency:\", min_frequency, \"Hz\")\nprint(\"Maximum Frequency:\", max_frequency, \"Hz\")","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:45.918601Z","iopub.execute_input":"2024-01-14T19:22:45.919086Z","iopub.status.idle":"2024-01-14T19:22:45.965923Z","shell.execute_reply.started":"2024-01-14T19:22:45.919044Z","shell.execute_reply":"2024-01-14T19:22:45.964596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"D = librosa.amplitude_to_db(np.abs(librosa.stft(merlin2_audio)), ref=np.max)\nfrequencies = librosa.fft_frequencies(sr=sr6)\nmin_frequency = np.min(frequencies)\nmax_frequency = np.max(frequencies)\nprint(\"Minimum Frequency:\", min_frequency, \"Hz\")\nprint(\"Maximum Frequency:\", max_frequency, \"Hz\")","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:45.967345Z","iopub.execute_input":"2024-01-14T19:22:45.967673Z","iopub.status.idle":"2024-01-14T19:22:46.018497Z","shell.execute_reply.started":"2024-01-14T19:22:45.967643Z","shell.execute_reply":"2024-01-14T19:22:46.017397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"D = librosa.amplitude_to_db(np.abs(librosa.stft(veery1_audio)), ref=np.max)\nfrequencies = librosa.fft_frequencies(sr=sr7)\nmin_frequency = np.min(frequencies)\nmax_frequency = np.max(frequencies)\nprint(\"Minimum Frequency:\", min_frequency, \"Hz\")\nprint(\"Maximum Frequency:\", max_frequency, \"Hz\")","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:46.019700Z","iopub.execute_input":"2024-01-14T19:22:46.020031Z","iopub.status.idle":"2024-01-14T19:22:46.062326Z","shell.execute_reply.started":"2024-01-14T19:22:46.020003Z","shell.execute_reply":"2024-01-14T19:22:46.061324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"D = librosa.amplitude_to_db(np.abs(librosa.stft(veery2_audio)), ref=np.max)\nfrequencies = librosa.fft_frequencies(sr=sr8)\nmin_frequency = np.min(frequencies)\nmax_frequency = np.max(frequencies)\nprint(\"Minimum Frequency:\", min_frequency, \"Hz\")\nprint(\"Maximum Frequency:\", max_frequency, \"Hz\")","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:46.063680Z","iopub.execute_input":"2024-01-14T19:22:46.064060Z","iopub.status.idle":"2024-01-14T19:22:46.142143Z","shell.execute_reply.started":"2024-01-14T19:22:46.064029Z","shell.execute_reply":"2024-01-14T19:22:46.140778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"D = librosa.amplitude_to_db(np.abs(librosa.stft(aldfly1_audio)), ref=np.max)\nfrequencies = librosa.fft_frequencies(sr=sr9)\nmin_frequency = np.min(frequencies)\nmax_frequency = np.max(frequencies)\nprint(\"Minimum Frequency:\", min_frequency, \"Hz\")\nprint(\"Maximum Frequency:\", max_frequency, \"Hz\")","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:46.143821Z","iopub.execute_input":"2024-01-14T19:22:46.144520Z","iopub.status.idle":"2024-01-14T19:22:46.207294Z","shell.execute_reply.started":"2024-01-14T19:22:46.144463Z","shell.execute_reply":"2024-01-14T19:22:46.206167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"D = librosa.amplitude_to_db(np.abs(librosa.stft(aldfly1_audio)), ref=np.max)\nfrequencies = librosa.fft_frequencies(sr=sr9)\nmin_frequency = np.min(frequencies)\nmax_frequency = np.max(frequencies)\nprint(\"Minimum Frequency:\", min_frequency, \"Hz\")\nprint(\"Maximum Frequency:\", max_frequency, \"Hz\")","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:46.209005Z","iopub.execute_input":"2024-01-14T19:22:46.209455Z","iopub.status.idle":"2024-01-14T19:22:46.270267Z","shell.execute_reply.started":"2024-01-14T19:22:46.209414Z","shell.execute_reply":"2024-01-14T19:22:46.269187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"D = librosa.amplitude_to_db(np.abs(librosa.stft(aldfly2_audio)), ref=np.max)\nfrequencies = librosa.fft_frequencies(sr=sr10)\nmin_frequency = np.min(frequencies)\nmax_frequency = np.max(frequencies)\nprint(\"Minimum Frequency:\", min_frequency, \"Hz\")\nprint(\"Maximum Frequency:\", max_frequency, \"Hz\")","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:46.272001Z","iopub.execute_input":"2024-01-14T19:22:46.272320Z","iopub.status.idle":"2024-01-14T19:22:46.309381Z","shell.execute_reply.started":"2024-01-14T19:22:46.272292Z","shell.execute_reply":"2024-01-14T19:22:46.308567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_features(file_path):\n\n    y, sr = librosa.load(file_path)\n\n\n    mfccs = librosa.feature.mfcc(y=y, sr=sr, n_mfcc=13)\n    zero_crossings = librosa.feature.zero_crossing_rate(y)\n    spectral_centroid = librosa.feature.spectral_centroid(y=y, sr=sr)[0]\n    rms_energy = librosa.feature.rms(y=y)[0]\n    spectral_rolloff = librosa.feature.spectral_rolloff(y=y, sr=sr)[0]\n    pitches, magnitudes = librosa.core.pitch.piptrack(y=y, sr=sr)\n\n\n    pitch = np.mean(pitches[pitches > 0])\n\n\n    spectral_flux = librosa.onset.onset_strength(y=y, sr=sr)\n\n    return mfccs, zero_crossings, spectral_centroid, rms_energy, spectral_rolloff, pitch, spectral_flux\n\n\nfile_paths = [\"/kaggle/input/birdsong-recognition/train_audio/amecro/XC114551.mp3\", \"/kaggle/input/birdsong-recognition/train_audio/amecro/XC114552.mp3\",\"/kaggle/input/birdsong-recognition/train_audio/coohaw/XC123581.mp3\",\n              \"/kaggle/input/birdsong-recognition/train_audio/coohaw/XC123582.mp3\", \"/kaggle/input/birdsong-recognition/train_audio/merlin/XC145140.mp3\",\"/kaggle/input/birdsong-recognition/train_audio/merlin/XC137975.mp3\",\n              \"/kaggle/input/birdsong-recognition/train_audio/veery/XC142682.mp3\",\n              \"/kaggle/input/birdsong-recognition/train_audio/veery/XC120869.mp3\",\"/kaggle/input/birdsong-recognition/train_audio/aldfly/XC137570.mp3\",\"/kaggle/input/birdsong-recognition/train_audio/aldfly/XC142068.mp3\"]\n\n\n\nfeatures_list = [extract_features(file_path) for file_path in file_paths]\n\ndef descriptive_stats(feature_index):\n    amecro_feature = [feature_set[feature_index].mean() for feature_set in features_list[:2]]\n    coohaw_feature= [feature_set[feature_index].mean() for feature_set in features_list[2:4]]\n    merlin_feature= [feature_set[feature_index].mean() for feature_set in features_list[4:6]]\n    veery_feature= [feature_set[feature_index].mean() for feature_set in features_list[6:8]]\n    aldfly_feature= [feature_set[feature_index].mean() for feature_set in features_list[8:10]]\n   \n    mean_amecro = np.mean(amecro_feature)\n    std_amecro = np.std(amecro_feature)\n    \n    mean_coohaw = np.mean(coohaw_feature)\n    std_coohaw = np.std(coohaw_feature)\n    \n    mean_merlin = np.mean(merlin_feature)\n    std_merlin = np.std(merlin_feature)\n    \n    mean_veery = np.mean(veery_feature)\n    std_veery = np.std(veery_feature)\n    \n    mean_aldfly = np.mean(aldfly_feature)\n    std_aldfly = np.std(aldfly_feature)\n\n    return amecro_feature, coohaw_feature,merlin_feature,veery_feature, aldfly_feature, mean_amecro, std_amecro, mean_coohaw, std_coohaw, mean_merlin , std_merlin, mean_veery, std_veery,  mean_aldfly, std_aldfly\n\n\nfeature_names = [\"MFCCs\", \"Zero Crossing Rate\", \"Spectral Centroid\", \"RMS Energy\", \"Spectral Rolloff\", \"Pitch\", \"Spectral Flux\"]\n\nfor i, feature_name in enumerate(feature_names):\n    amecro_feature, coohaw_feature,merlin_feature,veery_feature, aldfly_feature, mean_amecro, std_amecro, mean_coohaw, std_coohaw, mean_merlin , std_merlin, mean_veery, std_veery,  mean_aldfly, std_aldfly = descriptive_stats(i)\n\n    print(f\"Comparison for {feature_name}:\")\n    print(f\"Mean of 'amecro' category: {mean_amecro}, Standard Deviation: {std_amecro}\")\n    print(f\"Mean of 'coohaw' category: {mean_coohaw}, Standard Deviation: {std_coohaw}\")\n    print(f\"Mean of 'merlin' category: {mean_merlin}, Standard Deviation: {std_merlin}\")\n    print(f\"Mean of 'veery' category: {mean_veery}, Standard Deviation: {std_veery}\")\n    print(f\"Mean of 'aldfly' category: {mean_aldfly}, Standard Deviation: {std_aldfly}\")\n    \n    plt.hist(amecro_feature, bins=20, alpha=0.9, label='amecro')\n    plt.hist(coohaw_feature, bins=20, alpha=0.9, label='coohaw')\n    plt.hist(merlin_feature, bins=20, alpha=0.9, label='merlin')\n    plt.hist(veery_feature, bins=20, alpha=0.9, label='veery')\n    plt.hist(aldfly_feature, bins=20, alpha=0.9, label='aldfly')\n    \n    plt.title(f'Histogram Comparison for {feature_name}')\n    plt.xlabel('Values')\n    plt.ylabel('Frequency')\n    plt.legend()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:46.310693Z","iopub.execute_input":"2024-01-14T19:22:46.311259Z","iopub.status.idle":"2024-01-14T19:22:55.120277Z","shell.execute_reply.started":"2024-01-14T19:22:46.311227Z","shell.execute_reply":"2024-01-14T19:22:55.119070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport audioread\nimport logging\nimport os\nimport random\nimport time\nimport warnings\n\nimport soundfile as sf\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.utils.data as data\n\nfrom contextlib import contextmanager\nfrom pathlib import Path\nfrom typing import Optional\nfrom fastprogress import progress_bar\nfrom sklearn.metrics import f1_score\nfrom torchvision import models","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:55.122134Z","iopub.execute_input":"2024-01-14T19:22:55.122867Z","iopub.status.idle":"2024-01-14T19:22:57.893388Z","shell.execute_reply.started":"2024-01-14T19:22:55.122821Z","shell.execute_reply":"2024-01-14T19:22:57.892185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed: int=42):\n        random.seed(seed)\n        np.random.seed(seed)\n        os.environ[\"PYTHONHASHSEED\"]=str(seed)\n        torch.manual_seed(seed)\n        torch.cuda.manual_seed(seed)\n        torch.backends.cudnn.deterministic = True\n        torch.backends.cudnn.benchmark= True","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:57.894901Z","iopub.execute_input":"2024-01-14T19:22:57.895561Z","iopub.status.idle":"2024-01-14T19:22:57.902165Z","shell.execute_reply.started":"2024-01-14T19:22:57.895524Z","shell.execute_reply":"2024-01-14T19:22:57.901056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_logger(out_file=None):\n    logger= logging.getLogger()\n    formatter= logging.Formatter(\"%(asctime)s - %(levelname)s - %(message)s\")\n    logger.handlers = []\n    logger.setLevel(logging.INFO)\n    \n    handler = logging.StreamHandler()\n    handler.setFormatter(formatter)\n    handler.setLevel(logging.INFO)\n    logger.addHandler(handler)\n    \n    if out_file is not None:\n        fh= logging.FileHandler(out_file)\n        fh.setFormatter(formatter)\n        fh.setLevel(logging.INFO)\n        logger.addHandler(fh)\n    logger.info(\"logger set up\")\n    return logger\n\n@contextmanager #measure execution time\ndef timer(name: str, logger: Optional[logging.Logger]= None):\n    t0= time.time()\n    msg= f\"[{name}] start\"\n    if logger is None:\n        print(msg)\n    else:\n        logger.info(msg)    \n    yield\n    \n    msg=f\"[{name}] done in {time.time() - t0:.2f} s\"\n    if logger is None:\n        print(msg)\n    else:\n        logger.info(msg)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:57.903476Z","iopub.execute_input":"2024-01-14T19:22:57.903829Z","iopub.status.idle":"2024-01-14T19:22:57.918623Z","shell.execute_reply.started":"2024-01-14T19:22:57.903791Z","shell.execute_reply":"2024-01-14T19:22:57.917498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logger =get_logger(\"main.log\")\nset_seed(1213)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:57.920096Z","iopub.execute_input":"2024-01-14T19:22:57.920600Z","iopub.status.idle":"2024-01-14T19:22:57.938201Z","shell.execute_reply.started":"2024-01-14T19:22:57.920570Z","shell.execute_reply":"2024-01-14T19:22:57.937008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGET_SR=32000","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:57.939674Z","iopub.execute_input":"2024-01-14T19:22:57.940251Z","iopub.status.idle":"2024-01-14T19:22:57.945176Z","shell.execute_reply.started":"2024-01-14T19:22:57.940218Z","shell.execute_reply":"2024-01-14T19:22:57.943952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test=pd.read_csv(\"/kaggle/input/birdcall-check/test.csv\")\ntest_audio = \"/kaggle/input/birdcall-check/test_audio\"\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:57.946859Z","iopub.execute_input":"2024-01-14T19:22:57.947238Z","iopub.status.idle":"2024-01-14T19:22:57.972931Z","shell.execute_reply.started":"2024-01-14T19:22:57.947207Z","shell.execute_reply":"2024-01-14T19:22:57.971604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ResNet(nn.Module): #base class\n    def __init__(self, base_model_name: str, pretrained=False, #constructor #weights \n                 num_classes=264): #constructor method\n        super().__init__()\n        base_model = models.__getattribute__(base_model_name)(\n              pretrained=pretrained)\n        layers= list(base_model.children())[:-2]\n        layers.append(nn.AdaptiveMaxPool2d(1))\n        self.encoder = nn.Sequential(*layers)\n        \n        in_features=base_model.fc.in_features  #number of input features\n        \n        self.classifier = nn.Sequential(\n          nn.Linear(in_features,1024), nn.ReLU(), nn.Dropout(p=0.2),\n          nn.Linear(1024,1024), nn.ReLU(), nn.Dropout(p=0.2),\n          nn.Linear(1024, num_classes))\n        \n    def forward(self,x):\n        batch_size= x.size(0)\n        x= self.encoder(x).view(batch_size, -1)\n        x= self.classifier(x)\n        multiclass_proba = F.softmax(x, dim=1)\n        multilabel_proba = F.sigmoid(x)\n        \n        return {\n            \"logits\":x,\n            \"multiclass_proba\": multiclass_proba,\n            \"multilabel_proba\": multilabel_proba\n        }","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:57.974248Z","iopub.execute_input":"2024-01-14T19:22:57.974752Z","iopub.status.idle":"2024-01-14T19:22:57.984680Z","shell.execute_reply.started":"2024-01-14T19:22:57.974723Z","shell.execute_reply":"2024-01-14T19:22:57.983832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_config ={\n    \"base_model_name\": \"resnet50\",\n    \"pretrained\": False,\n    \"num_classes\": 264   \n}\n\nmelspectrogram_parameters ={\n    \"n_mels\": 128,\n    \"fmin\": 0,\n    \"fmax\":12000\n}\nweights_path = \"../input/birdcall-resnet50-init-weights/best.pth\"","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:57.985825Z","iopub.execute_input":"2024-01-14T19:22:57.986592Z","iopub.status.idle":"2024-01-14T19:22:57.999548Z","shell.execute_reply.started":"2024-01-14T19:22:57.986563Z","shell.execute_reply":"2024-01-14T19:22:57.998319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\ndf=pd.read_csv(\"/kaggle/input/birdsong-recognition/train.csv\")\n\nunique_bird_names = df['ebird_code'].unique()\nlabel_encoder = LabelEncoder()\nencoded_labels = label_encoder.fit_transform(unique_bird_names)\n\nBIRD_CODE = dict(zip(unique_bird_names,encoded_labels))\n\n# for bird_name, label in BIRD_CODE.items():\n#     print(f\"{bird_name}: {label}\")\n\nINV_BIRD_CODE={v: k for k, v in BIRD_CODE.items()}","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:58.000754Z","iopub.execute_input":"2024-01-14T19:22:58.001133Z","iopub.status.idle":"2024-01-14T19:22:58.426924Z","shell.execute_reply.started":"2024-01-14T19:22:58.001097Z","shell.execute_reply":"2024-01-14T19:22:58.425936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mono_to_color(X:np.ndarray,mean=None,std=None,norm_max=None,norm_min= None,eps=1e-6):\n    X=np.stack([X,X,X],axis=-1)\n    \n    mean = mean or X.mean()\n    X=X-mean\n    std=std or X.std()\n    Xstd= X/(std+eps)\n    \n    _min,_max= Xstd.min(),Xstd.max()\n    norm_max= norm_max or _max\n    norm_min= norm_min or _min\n    \n    if(_max - _min)>eps:\n        V=Xstd\n        V[V<norm_min]=norm_min\n        V[V>norm_max]=norm_max\n        \n        V=255*(V-norm_min)/(norm_max - norm_min)\n        V=V.astype(np.uint8)\n    else:\n        V=np.zeroes_like(Xstd, dtype=np.uint8)\n    return V\n\nclass TestDataset(data.Dataset):\n    def __init__(self,df:pd.DataFrame, clip:np.array,img_size=224,melspectrogram_parameters={}):\n        self.df=df\n        self.clip=clip\n        self.img_size= img_size\n        self.melspectrogram_parameters = melspectrogram_parameters\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx: int):\n        SR = 32000\n\n        sample = self.df.loc[idx, :]  # return row\n        site = sample.site\n        row_id = sample.row_id\n\n        if site == \"site_3\":\n            y = self.clip.astype(np.float32)\n            len_y = len(y)\n            start = 0\n            end = SR * 5\n\n            images = []\n\n            while len_y > start:\n                y_batch = y[start:end].astype(np.float32)\n\n                if len(y_batch) != (SR * 5):\n                    break\n                start = end\n                end = end + SR * 5\n\n                melspec = librosa.feature.melspectrogram(y=y_batch, sr=SR, **self.melspectrogram_parameters)\n                melspec = librosa.power_to_db(melspec).astype(np.float32)\n\n                image = mono_to_color(melspec)\n\n                height, width, _ =image.shape\n\n                image = cv2.resize(image, (int(width * self.img_size / height), self.img_size))\n\n                image = np.moveaxis(image, 2, 0)  # color channel axis to the first dimension\n\n                image = (image / 255.0).astype(np.float32)\n                images.append(image)\n\n            images = np.asarray(images)\n\n            return images, row_id, site\n\n        else:\n\n            end_seconds = int(sample.seconds)\n            start_seconds = int(end_seconds - 5)\n            start_index = SR * start_seconds\n            end_index = SR * end_seconds\n\n            y = self.clip[start_index:end_index].astype(np.float32)\n\n            melspec = librosa.feature.melspectrogram(y=y, sr=SR, **self.melspectrogram_parameters)\n\n            melspec = librosa.power_to_db(melspec).astype(np.float32)\n\n            image = mono_to_color(melspec)\n            height, width, _ =image.shape\n            image = cv2.resize(image,(int(width * self.img_size / height),self.img_size))\n            image = np.moveaxis(image, 2, 0)\n            image = (image / 255.0).astype(np.float32)\n            \n            return image, row_id, site\n        ","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:58.428210Z","iopub.execute_input":"2024-01-14T19:22:58.428531Z","iopub.status.idle":"2024-01-14T19:22:58.453301Z","shell.execute_reply.started":"2024-01-14T19:22:58.428502Z","shell.execute_reply":"2024-01-14T19:22:58.452025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model(config: dict, weights_path: str):\n    model = ResNet(**config)\n\n    # Load the checkpoint and map the tensors to the CPU if CUDA is not available\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    checkpoint = torch.load(weights_path, map_location=device)\n\n    # Load the model's learned parameters\n    model.load_state_dict(checkpoint[\"model_state_dict\"])\n\n    # Move the model to the specified device\n    model.to(device)\n\n    # Set the model to evaluation mode\n    model.eval()\n\n    return model","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:58.455028Z","iopub.execute_input":"2024-01-14T19:22:58.455427Z","iopub.status.idle":"2024-01-14T19:22:58.467869Z","shell.execute_reply.started":"2024-01-14T19:22:58.455393Z","shell.execute_reply":"2024-01-14T19:22:58.466822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prediction_for_clip(test_df: pd.DataFrame,\n                        clip: np.ndarray,\n                        model: ResNet,\n                        mel_params: dict,\n                        threshold=0.5):\n\n\n    dataset = TestDataset(df=test_df,\n                          clip=clip,\n                          img_size=224,\n                          melspectrogram_parameters=mel_params)\n\n    loader = data.DataLoader(dataset, batch_size=1, shuffle=False)\n\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n    model.eval()\n\n    prediction_dict = {}\n\n    for image, row_id, site in progress_bar(loader):\n        site = site[0]\n        row_id = row_id[0]\n\n        if site in {\"site_1\", \"site_2\"}:\n            image = image.to(device)\n\n            with torch.no_grad():\n                prediction = model(image)\n                proba = prediction[\"multilabel_proba\"].detach().cpu().numpy().reshape(-1)\n                \n            events = proba >= threshold\n            labels = np.argwhere(events).reshape(-1).tolist()\n            \n        else:\n    # to avoid prediction on large batch\n            image = image.squeeze(0)\n            batch_size = 16\n            whole_size = image.size(0)\n\n            if whole_size % batch_size == 0:\n                n_iter = whole_size // batch_size\n            else:\n                n_iter = whole_size // batch_size + 1\n\n            all_events = set()\n\n            for batch_i in range(n_iter):\n                batch = image[batch_i * batch_size: (batch_i + 1) * batch_size]\n\n                if batch.ndim == 3:\n                    batch = batch.unsqueeze(0)\n                    \n                batch = batch.to(device)\n                with torch.no_grad():\n                    prediction = model(batch)\n\n                    proba = prediction[\"multilabel_proba\"].detach().cpu().numpy()\n                events = proba >= threshold\n\n                for i in range(len(events)):\n                    event = events[i, :]\n                    labels = np.argwhere(event).reshape(-1).tolist()\n\n                    for label in labels:\n                        all_events.add(label)\n\n            labels=list(all_events)\n            \n        if len(labels)==0:\n            prediction_dict[row_id]=\"nocall\"\n        else:\n            labels_str_list=list(map(lambda x: INV_BIRD_CODE[x], labels))\n            label_string=\" \".join(labels_str_list)\n            prediction_dict[row_id]= label_string\n            \n    return prediction_dict\n            # Rest of the code for this section (not provided in the question)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:58.469933Z","iopub.execute_input":"2024-01-14T19:22:58.470470Z","iopub.status.idle":"2024-01-14T19:22:58.489963Z","shell.execute_reply.started":"2024-01-14T19:22:58.470425Z","shell.execute_reply":"2024-01-14T19:22:58.488585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prediction(test_df: pd.DataFrame,\n              test_audio: Path,\n              model_config: dict,\n              mel_params: dict,\n              weights_path: str,\n              threshold=0.5):\n    \n    model = get_model(model_config, weights_path)\n    unique_audio_id = test_df.audio_id.unique()\n\n    warnings.filterwarnings(\"ignore\")\n\n    prediction_dfs = []\n\n    for audio_id in unique_audio_id:\n        with timer(f\"Loading {audio_id}\", logger):\n            clip, _ = librosa.load(test_audio + \"/\" + (audio_id + \".mp3\"),\n                                   sr=TARGET_SR,\n                                   mono=True,\n                                   res_type=\"scipy\")\n\n        test_df_for_audio_id = test_df.query(\n            f\"audio_id == '{audio_id}'\").reset_index(drop=True)\n        with timer(f\"Prediction on {audio_id}\", logger):\n            prediction_dict = prediction_for_clip(test_df_for_audio_id,\n                                                  clip=clip,\n                                                  model=model,\n                                                  mel_params=mel_params,\n                                                  threshold=threshold)\n        row_id = list(prediction_dict.keys())\n        birds = list(prediction_dict.values())\n\n        prediction_df = pd.DataFrame({\n           \"row_id\": row_id,\n           \"birds\": birds\n        })\n\n        prediction_dfs.append(prediction_df)\n\n    prediction_df = pd.concat(prediction_dfs, axis=0, sort=False).reset_index(drop=True)\n\n    return prediction_df","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:58.491983Z","iopub.execute_input":"2024-01-14T19:22:58.492443Z","iopub.status.idle":"2024-01-14T19:22:58.504208Z","shell.execute_reply.started":"2024-01-14T19:22:58.492399Z","shell.execute_reply":"2024-01-14T19:22:58.502935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = prediction(test_df=test,\n                        test_audio=test_audio,\n                        model_config=model_config,\n                        mel_params=melspectrogram_parameters,\n                        weights_path=weights_path,\n                        threshold=0.8)\n\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-14T19:22:58.505976Z","iopub.execute_input":"2024-01-14T19:22:58.506391Z","iopub.status.idle":"2024-01-14T19:23:55.797626Z","shell.execute_reply.started":"2024-01-14T19:22:58.506358Z","shell.execute_reply":"2024-01-14T19:23:55.796580Z"},"trusted":true},"execution_count":null,"outputs":[]}]}