{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-31T16:23:10.792986Z","iopub.execute_input":"2025-03-31T16:23:10.793279Z","iopub.status.idle":"2025-03-31T16:23:57.347468Z","shell.execute_reply.started":"2025-03-31T16:23:10.793251Z","shell.execute_reply":"2025-03-31T16:23:57.346375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import ast\nimport folium\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport os\nimport pandas as pd\nimport seaborn as sns\nimport logging\nimport seaborn as sns\nfrom collections import Counter\nfrom folium.plugins import MarkerCluster\nfrom pathlib import Path # Using pathlib for better path handling\nfrom IPython.display import IFrame, display, HTML #added HTML import\nfrom io import StringIO\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T17:38:12.871027Z","iopub.execute_input":"2025-03-31T17:38:12.871376Z","iopub.status.idle":"2025-03-31T17:38:12.876956Z","shell.execute_reply.started":"2025-03-31T17:38:12.871350Z","shell.execute_reply":"2025-03-31T17:38:12.875767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE_INPUT_PATH = Path(\"/kaggle/input/birdclef-2025\")\nTRAIN_CSV_PATH = BASE_INPUT_PATH / \"train.csv\"\nTAXONOMY_CSV_PATH = BASE_INPUT_PATH / \"taxonomy.csv\"\nSAMPLE_SUBMISSION_PATH = BASE_INPUT_PATH / \"sample_submission.csv\"\nRECORDING_LOCATION_PATH = BASE_INPUT_PATH / \"recording_location.txt\" # Path defined, though not used in original logic beyond existence check\nTRAIN_AUDIO_DIR = BASE_INPUT_PATH / \"train_audio\"\nTRAIN_SOUNDSCAPES_DIR = BASE_INPUT_PATH / \"train_soundscapes\"\nTEST_SOUNDSCAPES_DIR = BASE_INPUT_PATH / \"test_soundscapes\"\nOUTPUT_DIR = Path(\"./\") # Output directory for plots and files\nOUTPUT_DIR.mkdir(exist_ok=True) # Create output directory if it doesn't exist","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T17:16:14.065112Z","iopub.execute_input":"2025-03-31T17:16:14.065442Z","iopub.status.idle":"2025-03-31T17:16:14.071694Z","shell.execute_reply.started":"2025-03-31T17:16:14.065417Z","shell.execute_reply":"2025-03-31T17:16:14.070328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plotting settings\nsns.set_theme(style=\"whitegrid\") # Consistent theme\nplt.rcParams['figure.figsize'] = (14, 7) # Default figure size\nTOP_N_SPECIES_PLOT = 50 # Limit the number of species shown in bar plots","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T17:16:15.039227Z","iopub.execute_input":"2025-03-31T17:16:15.039602Z","iopub.status.idle":"2025-03-31T17:16:15.044467Z","shell.execute_reply.started":"2025-03-31T17:16:15.039575Z","shell.execute_reply":"2025-03-31T17:16:15.043396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Logging configuration\nlogging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T17:16:16.167637Z","iopub.execute_input":"2025-03-31T17:16:16.167989Z","iopub.status.idle":"2025-03-31T17:16:16.172268Z","shell.execute_reply.started":"2025-03-31T17:16:16.167962Z","shell.execute_reply":"2025-03-31T17:16:16.171342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_data(csv_path):\n    \"\"\"Loads a CSV file into a pandas DataFrame with basic error handling.\"\"\"\n    try:\n        df = pd.read_csv(csv_path)\n        logging.info(f\"Successfully loaded data from: {csv_path}. Shape: {df.shape}\")\n        return df\n    except FileNotFoundError:\n        logging.error(f\"Error: File not found at {csv_path}\")\n        return None\n    except Exception as e:\n        logging.error(f\"Error loading {csv_path}: {e}\")\n        return None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T17:16:17.383486Z","iopub.execute_input":"2025-03-31T17:16:17.383809Z","iopub.status.idle":"2025-03-31T17:16:17.389127Z","shell.execute_reply.started":"2025-03-31T17:16:17.383786Z","shell.execute_reply":"2025-03-31T17:16:17.387926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_distribution(data_series, title, xlabel, ylabel, kind='bar', top_n=None, rotation=90, figsize=(14, 7)):\n    \"\"\"Generates and saves a distribution plot for a pandas Series.\"\"\"\n    plt.figure(figsize=figsize)\n    if kind == 'bar':\n        counts = data_series.value_counts()\n        if top_n:\n            counts = counts.head(top_n)\n            title += f\" (Top {top_n})\"\n        counts.plot(kind='bar')\n        plt.xticks(rotation=rotation)\n    elif kind == 'hist':\n        sns.histplot(data_series, kde=True, bins=30)\n        plt.xticks(rotation=0) # Usually no rotation needed for histograms\n    else:\n        logging.warning(f\"Unsupported plot kind: {kind}\")\n        return\n\n    plt.title(title, fontsize=16)\n    plt.xlabel(xlabel, fontsize=12)\n    plt.ylabel(ylabel, fontsize=12)\n    plt.tight_layout() # Adjust layout to prevent labels overlapping\n    # Create a filename-safe version of the title\n    filename_title = \"\".join(c if c.isalnum() else \"_\" for c in title).lower()\n    save_path = OUTPUT_DIR / f\"{filename_title}_distribution.png\"\n    plt.savefig(save_path)\n    logging.info(f\"Saved plot: {save_path}\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T17:16:18.319671Z","iopub.execute_input":"2025-03-31T17:16:18.320130Z","iopub.status.idle":"2025-03-31T17:16:18.327503Z","shell.execute_reply.started":"2025-03-31T17:16:18.319970Z","shell.execute_reply":"2025-03-31T17:16:18.326257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def parse_secondary_labels(label_str):\n    \"\"\"Safely parses the string representation of secondary labels.\"\"\"\n    if pd.isna(label_str) or label_str == \"[]\" or not label_str:\n        return []\n    try:\n        # Using ast.literal_eval is safer than eval()\n        parsed_list = ast.literal_eval(label_str)\n        # Ensure it's actually a list of strings\n        if isinstance(parsed_list, list) and all(isinstance(item, str) for item in parsed_list):\n             return parsed_list\n        else:\n             logging.warning(f\"Could not parse secondary label correctly, unexpected format: {label_str}\")\n             return [] # Return empty list if format is unexpected\n    except (ValueError, SyntaxError, TypeError) as e:\n        logging.warning(f\"Could not parse secondary label string: '{label_str}'. Error: {e}\")\n        return [] # Return empty list on error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T17:16:19.122643Z","iopub.execute_input":"2025-03-31T17:16:19.122959Z","iopub.status.idle":"2025-03-31T17:16:19.129426Z","shell.execute_reply.started":"2025-03-31T17:16:19.122935Z","shell.execute_reply":"2025-03-31T17:16:19.128146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def count_files_in_dir(directory):\n    \"\"\"Counts all files recursively in a given directory.\"\"\"\n    try:\n        return sum(len(files) for _, _, files in os.walk(directory))\n    except FileNotFoundError:\n        logging.error(f\"Directory not found: {directory}\")\n        return 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T17:16:19.899460Z","iopub.execute_input":"2025-03-31T17:16:19.899814Z","iopub.status.idle":"2025-03-31T17:16:19.904804Z","shell.execute_reply.started":"2025-03-31T17:16:19.899786Z","shell.execute_reply":"2025-03-31T17:16:19.903669Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_correlation_matrix(df, columns, title=\"Correlation Matrix\"):\n    \"\"\"Plots a heatmap of the correlation matrix for specified columns.\"\"\"\n    if not columns:\n        logging.warning(\"No columns specified for correlation matrix.\")\n        return\n    corr_matrix = df[columns].corr()\n    plt.figure(figsize=(8, 6))\n    sns.heatmap(corr_matrix, annot=True, cmap='coolwarm', fmt=\".2f\")\n    plt.title(title, fontsize=16)\n    plt.tight_layout()\n    save_path = OUTPUT_DIR / f\"{title.replace(' ', '_').lower()}_heatmap.png\"\n    plt.savefig(save_path)\n    logging.info(f\"Saved plot: {save_path}\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T17:16:20.783412Z","iopub.execute_input":"2025-03-31T17:16:20.783757Z","iopub.status.idle":"2025-03-31T17:16:20.789357Z","shell.execute_reply.started":"2025-03-31T17:16:20.783733Z","shell.execute_reply":"2025-03-31T17:16:20.788497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_geo_locations(df, lat_col='latitude', lon_col='longitude', popup_col='primary_label', map_filename=\"recording_locations_map.html\"):\n    \"\"\"Creates and saves an interactive map of geographical locations.\"\"\"\n    location_data = df.dropna(subset=[lat_col, lon_col]).copy() # Work on a copy\n    if location_data.empty:\n        logging.warning(\"No valid location data found to plot map.\")\n        return\n\n    # Filter out potential invalid coordinates (basic check)\n    location_data = location_data[\n        (location_data[lat_col].between(-90, 90)) &\n        (location_data[lon_col].between(-180, 180))\n    ]\n    if location_data.empty:\n        logging.warning(\"No valid location data after basic lat/lon range filtering.\")\n        return\n\n    logging.info(f\"Plotting {location_data.shape[0]} locations on map.\")\n\n    mean_lat = location_data[lat_col].mean()\n    mean_lon = location_data[lon_col].mean()\n\n    m = folium.Map(location=[mean_lat, mean_lon], zoom_start=5) # Adjusted zoom\n\n    marker_cluster = MarkerCluster().add_to(m)\n\n    for idx, row in location_data.iterrows():\n        popup_text = f\"Species: {row[popup_col]}<br>Rating: {row.get('rating', 'N/A')}\" # Add rating if available\n        folium.Marker(\n            location=[row[lat_col], row[lon_col]],\n            popup=popup_text,\n            tooltip=f\"Click for info on {row[popup_col]}\" # Add tooltip\n        ).add_to(marker_cluster)\n\n    save_path = OUTPUT_DIR / map_filename\n    m.save(save_path)\n    logging.info(f\"Interactive map saved as: {save_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T17:16:21.783151Z","iopub.execute_input":"2025-03-31T17:16:21.783516Z","iopub.status.idle":"2025-03-31T17:16:21.790435Z","shell.execute_reply.started":"2025-03-31T17:16:21.783489Z","shell.execute_reply":"2025-03-31T17:16:21.789410Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if __name__ == \"__main__\":\n\n    logging.info(\"Starting BirdCLEF 2025 EDA Script\")\n\n    # --- Load Data ---\n    train_df = load_data(TRAIN_CSV_PATH)\n    taxonomy_df = load_data(TAXONOMY_CSV_PATH)\n    sample_submission = load_data(SAMPLE_SUBMISSION_PATH)\n\n    # Check if dataframes loaded successfully\n    if train_df is None or taxonomy_df is None or sample_submission is None:\n        logging.error(\"Failed to load one or more essential data files. Exiting.\")\n        exit()\n\n    # --- Basic Data Inspection ---\n    logging.info(\"--- Basic Data Inspection ---\")\n    logging.info(f\"Train data info:\\n{train_df.info()}\")\n    logging.info(f\"Missing values in train_df:\\n{train_df.isnull().sum()}\")\n    logging.info(f\"Duplicates in train_df: {train_df.duplicated().sum()}\")\n\n    # --- Analyze Primary Labels ---\n    logging.info(\"--- Analyzing Primary Labels ---\")\n    plot_distribution(train_df['primary_label'],\n                      title=\"Distribution of Primary Species Labels\",\n                      xlabel=\"Species\",\n                      ylabel=\"Count\",\n                      kind='bar',\n                      top_n=TOP_N_SPECIES_PLOT) # Plot only top N for clarity\n    species_counts = train_df['primary_label'].value_counts()\n    logging.info(f\"Primary label count statistics:\\n{species_counts.describe()}\")\n    logging.info(f\"Most common primary labels:\\n{species_counts.head()}\")\n    logging.info(f\"Least common primary labels:\\n{species_counts.tail()}\")\n\n    # --- Analyze Ratings ---\n    logging.info(\"--- Analyzing Ratings ---\")\n    plot_distribution(train_df['rating'],\n                      title=\"Distribution of Audio Quality Ratings\",\n                      xlabel=\"Rating\",\n                      ylabel=\"Frequency\",\n                      kind='hist')\n\n    # Rating statistics and outlier detection\n    logging.info(f\"Rating Summary Statistics:\\n{train_df['rating'].describe()}\")\n    plt.figure(figsize=(10, 4))\n    sns.boxplot(x=train_df['rating'])\n    plt.title(\"Boxplot of Ratings\")\n    plt.xlabel(\"Rating\")\n    plt.tight_layout()\n    plt.savefig(OUTPUT_DIR / \"rating_boxplot.png\")\n    plt.show()\n\n    Q1 = train_df['rating'].quantile(0.25)\n    Q3 = train_df['rating'].quantile(0.75)\n    IQR = Q3 - Q1\n    lower_bound = Q1 - 1.5 * IQR\n    upper_bound = Q3 + 1.5 * IQR\n    logging.info(f\"Rating outlier bounds (IQR method): Lower = {lower_bound:.2f}, Upper = {upper_bound:.2f}\")\n    rating_outliers = train_df[(train_df['rating'] < lower_bound) | (train_df['rating'] > upper_bound)]\n    logging.info(f\"Number of potential rating outliers: {rating_outliers.shape[0]}\")\n    if not rating_outliers.empty:\n        logging.info(f\"Sample rating outliers:\\n{rating_outliers[['primary_label', 'rating']].head()}\")\n\n    # Average rating by species\n    avg_rating_by_species = train_df.groupby('primary_label')['rating'].mean().sort_values(ascending=False)\n    plt.figure(figsize=(14, 7))\n    avg_rating_by_species.head(TOP_N_SPECIES_PLOT).plot(kind='bar') # Plot top N\n    plt.title(f\"Average Rating by Species (Top {TOP_N_SPECIES_PLOT})\")\n    plt.xlabel(\"Species\")\n    plt.ylabel(\"Average Rating\")\n    plt.xticks(rotation=90)\n    plt.tight_layout()\n    plt.savefig(OUTPUT_DIR / \"average_rating_by_species.png\")\n    plt.show()\n\n\n    # --- Analyze Recording Types ---\n    logging.info(\"--- Analyzing Recording Types ---\")\n    plot_distribution(train_df['type'],\n                      title=\"Distribution of Recording Types\",\n                      xlabel=\"Type\",\n                      ylabel=\"Count\",\n                      kind='bar',\n                      rotation=0) # Horizontal labels likely okay\n\n    # --- Analyze Secondary Labels ---\n    logging.info(\"--- Analyzing Secondary Labels ---\")\n    train_df['secondary_labels_list'] = train_df['secondary_labels'].apply(parse_secondary_labels)\n    train_df['num_secondary_labels'] = train_df['secondary_labels_list'].apply(len)\n\n    plot_distribution(train_df['num_secondary_labels'],\n                      title=\"Distribution of Number of Secondary Labels per Recording\",\n                      xlabel=\"Number of Secondary Labels\",\n                      ylabel=\"Frequency\",\n                      kind='hist',\n                      rotation=0)\n\n    # Flatten the list of all secondary labels\n    all_secondary_labels = [label for sublist in train_df['secondary_labels_list'] for label in sublist]\n    secondary_counter = Counter(all_secondary_labels)\n    logging.info(f\"Total unique secondary labels: {len(secondary_counter)}\")\n    logging.info(f\"Most common secondary labels (Top 15):\\n{secondary_counter.most_common(15)}\")\n\n    # Plot top N secondary labels\n    if secondary_counter:\n        sec_labels_df = pd.DataFrame(secondary_counter.most_common(TOP_N_SPECIES_PLOT), columns=['label', 'count'])\n        plt.figure(figsize=(14, 7))\n        sns.barplot(x='count', y='label', data=sec_labels_df, palette='viridis')\n        plt.title(f'Top {TOP_N_SPECIES_PLOT} Most Common Secondary Labels')\n        plt.xlabel('Count')\n        plt.ylabel('Species Label')\n        plt.tight_layout()\n        plt.savefig(OUTPUT_DIR / \"top_secondary_labels.png\")\n        plt.show()\n    else:\n        logging.info(\"No secondary labels found to plot.\")\n\n    # --- Analyze Taxonomy Data ---\n    logging.info(\"--- Analyzing Taxonomy Data ---\")\n    logging.info(f\"Taxonomy data info:\\n{taxonomy_df.info()}\")\n    logging.info(f\"Actual columns in taxonomy_df: {taxonomy_df.columns.tolist()}\")\n    plot_distribution(taxonomy_df['class_name'],\n                      title=\"Distribution of Taxonomic Classes\",\n                      xlabel=\"Class\",\n                      ylabel=\"Count\",\n                      kind='bar',\n                      rotation=0)\n\n    # Distribution of Orders (more specific than class)\n    plot_distribution(taxonomy_df['primary_label'],\n                      title=\"Distribution of Taxonomic Orders\",\n                      xlabel=\"Order\",\n                      ylabel=\"Count\",\n                      kind='bar',\n                      top_n=30, # Limit for readability\n                      rotation=45)\n\n\n    # --- Merge Train and Taxonomy Data ---\n    # logging.info(\"--- Merging Train and Taxonomy Data ---\")\n    # merged_df = pd.merge(train_df, taxonomy_df, on=\"primary_label\", how=\"left\")\n    # logging.info(f\"Merged data shape: {merged_df.shape}\")\n    # missing_tax_info = merged_df['species_code'].isnull().sum() # Check based on a core taxonomy column\n    # logging.info(f\"Recordings with missing taxonomy info after merge: {missing_tax_info}\")\n    # if missing_tax_info > 0:\n        # missing_labels = train_df[merged_df['species_code'].isnull()]['primary_label'].unique()\n        # logging.warning(f\"Primary labels in train_df missing in taxonomy_df: {list(missing_labels)}\")\n\n    # --- Analyze File Counts ---\n    logging.info(\"--- Analyzing File Counts ---\")\n    logging.info(f\"Train Audio files count: {count_files_in_dir(TRAIN_AUDIO_DIR)}\")\n    logging.info(f\"Train Soundscapes files count: {count_files_in_dir(TRAIN_SOUNDSCAPES_DIR)}\")\n    logging.info(f\"Test Soundscapes files count: {count_files_in_dir(TEST_SOUNDSCAPES_DIR)}\")\n    # Consider adding checks: does each filename in train_df exist?\n\n    # --- Geographical Analysis ---\n    logging.info(\"--- Geographical Analysis ---\")\n    location_data = train_df.dropna(subset=['latitude', 'longitude']).copy() # Use a copy for calculations\n    location_data = location_data[ # Basic sanity check on coordinates\n        (location_data['latitude'].between(-90, 90)) &\n        (location_data['longitude'].between(-180, 180))\n    ]\n\n    if not location_data.empty:\n        # Scatter plot\n        plt.figure(figsize=(10, 8))\n        sns.scatterplot(x='longitude', y='latitude', data=location_data, alpha=0.3, hue='rating', size='rating', palette='viridis', legend='brief')\n        plt.title(\"Scatter Plot of Recording Locations (Colored by Rating)\")\n        plt.xlabel(\"Longitude\")\n        plt.ylabel(\"Latitude\")\n        plt.tight_layout()\n        plt.savefig(OUTPUT_DIR / \"location_scatter_plot.png\")\n        plt.show()\n\n        # Geographical outlier detection (using IQR)\n        lat_Q1 = location_data['latitude'].quantile(0.25)\n        lat_Q3 = location_data['latitude'].quantile(0.75)\n        lat_IQR = lat_Q3 - lat_Q1\n        lat_lower_bound = lat_Q1 - 1.5 * lat_IQR\n        lat_upper_bound = lat_Q3 + 1.5 * lat_IQR\n\n        lon_Q1 = location_data['longitude'].quantile(0.25)\n        lon_Q3 = location_data['longitude'].quantile(0.75)\n        lon_IQR = lon_Q3 - lon_Q1\n        lon_lower_bound = lon_Q1 - 1.5 * lon_IQR\n        lon_upper_bound = lon_Q3 + 1.5 * lon_IQR\n\n        logging.info(f\"Geographical outlier bounds (IQR): Latitude [{lat_lower_bound:.2f}, {lat_upper_bound:.2f}], Longitude [{lon_lower_bound:.2f}, {lon_upper_bound:.2f}]\")\n\n        geo_outliers = location_data[\n            (location_data['latitude'] < lat_lower_bound) | (location_data['latitude'] > lat_upper_bound) |\n            (location_data['longitude'] < lon_lower_bound) | (location_data['longitude'] > lon_upper_bound)\n        ]\n        logging.info(f\"Number of potential geographical outliers: {geo_outliers.shape[0]}\")\n        if not geo_outliers.empty:\n            logging.info(f\"Sample geographical outliers:\\n{geo_outliers[['primary_label', 'latitude', 'longitude', 'rating']].head()}\")\n\n        # Interactive Map\n        plot_geo_locations(train_df) # Pass the original df with rating info\n\n    else:\n        logging.warning(\"Skipping geographical analysis due to missing or invalid location data.\")\n\n\n    # --- Analyze Collection Source ---\n    logging.info(\"--- Analyzing Collection Source ---\")\n    if 'collection' in train_df.columns:\n        plot_distribution(train_df['collection'],\n                          title=\"Distribution of Recordings by Collection Source\",\n                          xlabel=\"Collection\",\n                          ylabel=\"Count\",\n                          kind='bar')\n        logging.info(f\"Collection counts:\\n{train_df['collection'].value_counts()}\")\n    else:\n        logging.info(\"Column 'collection' not found in train_df.\")\n\n\n    # --- Correlation Analysis ---\n    logging.info(\"--- Correlation Analysis ---\")\n    # Only 'rating' is numerical in the original set, add others if feature engineered (e.g., duration)\n    numerical_cols = ['rating', 'num_secondary_labels'] # Add latitude/longitude if desired\n    plot_correlation_matrix(train_df, numerical_cols, title=\"Correlation Matrix of Numerical Features\")\n\n\n    # --- Prepare Submission File (Example) ---\n    logging.info(\"--- Preparing Sample Submission ---\")\n    # This part just saves the sample submission, actual prediction logic is needed for a real submission\n    try:\n        submission_path = OUTPUT_DIR / 'submission.csv'\n        sample_submission.to_csv(submission_path, index=False)\n        logging.info(f\"Sample submission file saved to: {submission_path}\")\n    except Exception as e:\n        logging.error(f\"Failed to save submission file: {e}\")\n\n    logging.info(\"EDA Script Finished.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T17:25:32.413513Z","iopub.execute_input":"2025-03-31T17:25:32.413857Z","iopub.status.idle":"2025-03-31T17:27:22.309786Z","shell.execute_reply.started":"2025-03-31T17:25:32.413833Z","shell.execute_reply":"2025-03-31T17:27:22.308841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import geopandas as gpd\n\n# --- Geographical Analysis ---\nlogging.info(\"--- Geographical Analysis ---\")\n\n# Filter valid location data\nlocation_data = train_df.dropna(subset=['latitude', 'longitude']).copy()\nlocation_data = location_data[\n    (location_data['latitude'].between(-90, 90)) &\n    (location_data['longitude'].between(-180, 180))\n]\n\nif not location_data.empty:\n    # Load world map\n    world = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres'))\n\n    # Scatter plot with world map\n    fig, ax = plt.subplots(figsize=(12, 8))\n    world.plot(ax=ax, color='lightgrey')  # Background world map\n    sns.scatterplot(x='longitude', y='latitude', data=location_data, alpha=0.3,\n                    hue='rating', size='rating', palette='viridis', legend='brief', ax=ax)\n    plt.title(\"Scatter Plot of Recording Locations (Colored by Rating)\")\n    plt.xlabel(\"Longitude\")\n    plt.ylabel(\"Latitude\")\n    plt.tight_layout()\n    plt.savefig(OUTPUT_DIR / \"location_scatter_plot.png\")\n    plt.show()\n\n    # Geographical outlier detection (using IQR)\n    lat_Q1, lat_Q3 = location_data['latitude'].quantile([0.25, 0.75])\n    lon_Q1, lon_Q3 = location_data['longitude'].quantile([0.25, 0.75])\n    lat_IQR, lon_IQR = lat_Q3 - lat_Q1, lon_Q3 - lon_Q1\n\n    lat_lower, lat_upper = lat_Q1 - 1.5 * lat_IQR, lat_Q3 + 1.5 * lat_IQR\n    lon_lower, lon_upper = lon_Q1 - 1.5 * lon_IQR, lon_Q3 + 1.5 * lon_IQR\n\n    logging.info(f\"Geographical outlier bounds (IQR): Latitude [{lat_lower:.2f}, {lat_upper:.2f}], \"\n                 f\"Longitude [{lon_lower:.2f}, {lon_upper:.2f}]\")\n\n    geo_outliers = location_data[\n        (location_data['latitude'] < lat_lower) | (location_data['latitude'] > lat_upper) |\n        (location_data['longitude'] < lon_lower) | (location_data['longitude'] > lon_upper)\n    ]\n\n    logging.info(f\"Number of potential geographical outliers: {geo_outliers.shape[0]}\")\n    if not geo_outliers.empty:\n        logging.info(f\"Sample geographical outliers:\\n{geo_outliers[['primary_label', 'latitude', 'longitude', 'rating']].head()}\")\n\n    # Interactive Map\n    world_map = folium.Map(location=[location_data['latitude'].mean(), location_data['longitude'].mean()], zoom_start=2)\n    marker_cluster = MarkerCluster().add_to(world_map)\n\n    for _, row in location_data.iterrows():\n        folium.Marker(\n            location=[row['latitude'], row['longitude']],\n            popup=f\"{row['primary_label']} (Rating: {row['rating']})\",\n            icon=folium.Icon(color='blue', icon='info-sign')\n        ).add_to(marker_cluster)\n\n    map_path = OUTPUT_DIR / \"geographical_map.html\"\n    world_map.save(map_path)\n    logging.info(f\"Interactive map saved at {map_path}\")\n\nelse:\n    logging.warning(\"Skipping geographical analysis due to missing or invalid location data.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T17:44:15.763936Z","iopub.execute_input":"2025-03-31T17:44:15.764329Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T17:43:22.853926Z","iopub.execute_input":"2025-03-31T17:43:22.854338Z","iopub.status.idle":"2025-03-31T17:43:22.875726Z","shell.execute_reply.started":"2025-03-31T17:43:22.854309Z","shell.execute_reply":"2025-03-31T17:43:22.874594Z"}},"outputs":[],"execution_count":null}]}