{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"89e175e4","cell_type":"markdown","source":"# iWildCam 2020 — Exploratory Data Analysis\n**Dataset:** [iWildCam 2020 FGVC7](https://www.kaggle.com/competitions/iwildcam-2020-fgvc7/data)  \n**File analysed:** `iwildcam2020_train_annotations.json` (COCO-style format)  \n\nThis notebook covers:\n1. Loading & structure inspection\n2. Missing values & data quality\n3. Dependent variable (species/category) analysis\n4. Predictors / covariates overview\n5. Visualisations — distributions, relationships, temporal & spatial patterns","metadata":{}},{"id":"f04a93c8","cell_type":"markdown","source":"## 0. Setup","metadata":{}},{"id":"4efc7bb0","cell_type":"code","source":"import json\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.ticker as mticker\nimport seaborn as sns\nfrom collections import Counter\nfrom pathlib import Path\n\n# ── reproducibility & aesthetics ──────────────────────────────────────────────\nnp.random.seed(42)\nsns.set_theme(style='whitegrid', palette='muted', font_scale=1.1)\nplt.rcParams['figure.dpi'] = 120\n\n# ── path — update if your Kaggle input path differs ───────────────────────────\nDATA_PATH = Path('/kaggle/input/competitions/iwildcam-2020-fgvc7/iwildcam2020_train_annotations.json')\n# DATA_PATH = Path('iwildcam2020_train_annotations.json')  # local fallback\n\nprint('Libraries loaded ✓')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:06:36.057146Z","iopub.execute_input":"2026-06-18T22:06:36.057451Z","iopub.status.idle":"2026-06-18T22:06:36.066043Z","shell.execute_reply.started":"2026-06-18T22:06:36.057427Z","shell.execute_reply":"2026-06-18T22:06:36.064913Z"}},"outputs":[],"execution_count":null},{"id":"67fbf47c","cell_type":"markdown","source":"## 1. Load & inspect raw structure","metadata":{}},{"id":"b09e4676-081a-4138-807c-1c2232eb6b1b","cell_type":"code","source":"import os\n\nfor root, dirs, files in os.walk('/kaggle/input'):\n    for file in files:\n        if file.endswith('.json'):\n            print(os.path.join(root, file))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:09:44.978880Z","iopub.execute_input":"2026-06-18T22:09:44.979996Z","iopub.status.idle":"2026-06-18T22:12:27.766382Z","shell.execute_reply.started":"2026-06-18T22:09:44.979950Z","shell.execute_reply":"2026-06-18T22:12:27.765358Z"}},"outputs":[],"execution_count":null},{"id":"b2420aca","cell_type":"code","source":"with open(DATA_PATH) as f:\n    raw = json.load(f)\n\nprint('Top-level keys:', list(raw.keys()))\nprint(f\"  images      : {len(raw['images']):,} records\")\nprint(f\"  annotations : {len(raw['annotations']):,} records\")\nprint(f\"  categories  : {len(raw['categories']):,} species categories\")\n\n# sample one record from each section to understand fields\nprint('\\n── sample image record ──')\nprint(json.dumps(raw['images'][0], indent=2))\n\nprint('\\n── sample annotation record ──')\nprint(json.dumps(raw['annotations'][0], indent=2))\n\nprint('\\n── sample category record ──')\nprint(json.dumps(raw['categories'][0], indent=2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:06:51.083546Z","iopub.execute_input":"2026-06-18T22:06:51.083923Z","iopub.status.idle":"2026-06-18T22:06:52.677399Z","shell.execute_reply.started":"2026-06-18T22:06:51.083895Z","shell.execute_reply":"2026-06-18T22:06:52.676358Z"}},"outputs":[],"execution_count":null},{"id":"66cb3a9f","cell_type":"markdown","source":"## 2. Build DataFrames","metadata":{}},{"id":"0929416b","cell_type":"code","source":"# ── images ────────────────────────────────────────────────────────────────────\ndf_img = pd.DataFrame(raw['images'])\n\n# parse datetime\ndf_img['datetime'] = pd.to_datetime(df_img['datetime'], errors='coerce')\ndf_img['hour']     = df_img['datetime'].dt.hour\ndf_img['month']    = df_img['datetime'].dt.month\ndf_img['year']     = df_img['datetime'].dt.year\ndf_img['dayofweek']= df_img['datetime'].dt.dayofweek  # 0=Mon … 6=Sun\n\n# ── annotations ───────────────────────────────────────────────────────────────\ndf_ann = pd.DataFrame(raw['annotations'])\n\n# ── categories ────────────────────────────────────────────────────────────────\ndf_cat = pd.DataFrame(raw['categories'])\ncat_map = dict(zip(df_cat['id'], df_cat['name']))  # id → species name\n\n# ── merge: one row per annotation, enriched with image metadata ────────────────\ndf = df_ann.merge(df_img, left_on='image_id', right_on='id', suffixes=('_ann','_img'))\ndf['species'] = df['category_id'].map(cat_map)\n\nprint('Merged DataFrame shape:', df.shape)\nprint(df.dtypes)\ndf.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:06:57.741452Z","iopub.execute_input":"2026-06-18T22:06:57.742404Z","iopub.status.idle":"2026-06-18T22:06:58.946703Z","shell.execute_reply.started":"2026-06-18T22:06:57.742356Z","shell.execute_reply":"2026-06-18T22:06:58.945700Z"}},"outputs":[],"execution_count":null},{"id":"59d10496","cell_type":"markdown","source":"## 3. Missing values & data quality","metadata":{}},{"id":"f1c4ffb1","cell_type":"code","source":"# ── 3a. missing values per column ─────────────────────────────────────────────\nmissing = df.isnull().sum()\nmissing_pct = (missing / len(df) * 100).round(2)\nmissing_df = pd.DataFrame({'null_count': missing, 'null_pct': missing_pct})\nmissing_df = missing_df[missing_df['null_count'] > 0].sort_values('null_count', ascending=False)\n\nprint(f'Total cells  : {df.size:,}')\nprint(f'Total nulls  : {df.isnull().sum().sum():,}')\nprint(f'\\nColumns with missing values ({len(missing_df)} of {df.shape[1]}):')\nprint(missing_df.to_string())\n\n# plot\nif not missing_df.empty:\n    fig, ax = plt.subplots(figsize=(8, max(3, len(missing_df)*0.5)))\n    missing_df['null_pct'].sort_values().plot.barh(ax=ax, color='#e07b54')\n    ax.set_xlabel('% missing')\n    ax.set_title('Missing values by column')\n    plt.tight_layout()\n    plt.show()\nelse:\n    print('No missing values found in merged DataFrame.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:07:13.839277Z","iopub.execute_input":"2026-06-18T22:07:13.840305Z","iopub.status.idle":"2026-06-18T22:07:14.025066Z","shell.execute_reply.started":"2026-06-18T22:07:13.840272Z","shell.execute_reply":"2026-06-18T22:07:14.024165Z"}},"outputs":[],"execution_count":null},{"id":"b27c9b24","cell_type":"code","source":"# ── 3b. datetime parse failures (count as a separate quality check) ────────────\nbad_dt = df_img['datetime'].isnull().sum()\nprint(f'Images with unparseable datetime: {bad_dt:,} ({bad_dt/len(df_img)*100:.1f}%)')\n\n# ── 3c. duplicate image IDs ───────────────────────────────────────────────────\ndup_imgs = df_img['id'].duplicated().sum()\nprint(f'Duplicate image IDs: {dup_imgs}')\n\n# ── 3d. annotations with unknown category ─────────────────────────────────────\nunknown_cat = df['category_id'].isin([0]).sum()   # category 0 = 'empty' in iWildCam\nprint(f'Annotations with category 0 (empty/unknown): {unknown_cat:,}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:07:21.220001Z","iopub.execute_input":"2026-06-18T22:07:21.220467Z","iopub.status.idle":"2026-06-18T22:07:21.261519Z","shell.execute_reply.started":"2026-06-18T22:07:21.220439Z","shell.execute_reply":"2026-06-18T22:07:21.260334Z"}},"outputs":[],"execution_count":null},{"id":"982af2ec","cell_type":"markdown","source":"## 4. Dependent variable — species (category_id)","metadata":{}},{"id":"f23be333","cell_type":"code","source":"# ── 4a. overall species distribution ──────────────────────────────────────────\nspecies_counts = df['species'].value_counts()\nprint(f'Unique species (including empty): {df[\"species\"].nunique()}')\nprint(f'Most common  : {species_counts.index[0]}  ({species_counts.iloc[0]:,} annotations)')\nprint(f'Least common : {species_counts.index[-1]}  ({species_counts.iloc[-1]:,} annotations)')\nprint(f'\\nTop 15:\\n{species_counts.head(15).to_string()}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:07:24.546315Z","iopub.execute_input":"2026-06-18T22:07:24.547319Z","iopub.status.idle":"2026-06-18T22:07:24.583471Z","shell.execute_reply.started":"2026-06-18T22:07:24.547274Z","shell.execute_reply":"2026-06-18T22:07:24.582643Z"}},"outputs":[],"execution_count":null},{"id":"3b3b993c","cell_type":"code","source":"# ── 4b. top-20 species bar chart ──────────────────────────────────────────────\ntop20 = species_counts.head(20)\n\nfig, ax = plt.subplots(figsize=(10, 6))\ntop20.sort_values().plot.barh(ax=ax, color='#5b8db8')\nax.set_xlabel('Number of annotations')\nax.set_title('Top 20 most common species')\nax.xaxis.set_major_formatter(mticker.FuncFormatter(lambda x, _: f'{int(x):,}'))\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:07:28.156904Z","iopub.execute_input":"2026-06-18T22:07:28.158041Z","iopub.status.idle":"2026-06-18T22:07:28.636916Z","shell.execute_reply.started":"2026-06-18T22:07:28.157986Z","shell.execute_reply":"2026-06-18T22:07:28.635805Z"}},"outputs":[],"execution_count":null},{"id":"e1807728","cell_type":"code","source":"# ── 4c. class imbalance — log-scale distribution ──────────────────────────────\nfig, ax = plt.subplots(figsize=(10, 4))\nax.bar(range(len(species_counts)), species_counts.values, color='#5b8db8', width=1.0)\nax.set_yscale('log')\nax.set_xlabel('Species rank (most → least common)')\nax.set_ylabel('Annotation count (log scale)')\nax.set_title('Class imbalance across all species (log scale)')\nplt.tight_layout()\nplt.show()\n\n# imbalance ratio\nratio = species_counts.iloc[0] / species_counts.iloc[-1]\nprint(f'Imbalance ratio (most / least common): {ratio:,.0f}x')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:07:34.287568Z","iopub.execute_input":"2026-06-18T22:07:34.288019Z","iopub.status.idle":"2026-06-18T22:07:35.279020Z","shell.execute_reply.started":"2026-06-18T22:07:34.287988Z","shell.execute_reply":"2026-06-18T22:07:35.278047Z"}},"outputs":[],"execution_count":null},{"id":"3b02ac85","cell_type":"code","source":"# ── 4d. pie chart — top 10 vs rest ────────────────────────────────────────────\ntop10 = species_counts.head(10)\nrest  = pd.Series({'All other species': species_counts.iloc[10:].sum()})\npie_data = pd.concat([top10, rest])\n\nfig, ax = plt.subplots(figsize=(8, 8))\nax.pie(pie_data, labels=pie_data.index, autopct='%1.1f%%',\n       startangle=140, pctdistance=0.82,\n       colors=sns.color_palette('muted', len(pie_data)))\nax.set_title('Share of annotations: top 10 species vs rest')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:07:41.945909Z","iopub.execute_input":"2026-06-18T22:07:41.946335Z","iopub.status.idle":"2026-06-18T22:07:42.164436Z","shell.execute_reply.started":"2026-06-18T22:07:41.946304Z","shell.execute_reply":"2026-06-18T22:07:42.163518Z"}},"outputs":[],"execution_count":null},{"id":"f0474802","cell_type":"markdown","source":"## 5. Predictors / covariates","metadata":{}},{"id":"72684355","cell_type":"code","source":"# ── 5a. covariate summary ─────────────────────────────────────────────────────\n# Columns available as potential predictors (non-target, non-ID fields)\n# From the image metadata:\n#   location, datetime (→ hour, month, year), width, height (if present),\n#   seq_id, seq_num_frames, frame_num\n# From annotations:\n#   bbox (bounding box of labelled animal)\n\ncov_cols = ['location', 'hour', 'month', 'year',\n            'seq_num_frames', 'frame_num']\ncov_cols = [c for c in cov_cols if c in df.columns]\n\nprint(f'Potential covariate columns identified: {len(cov_cols)}')\nfor c in cov_cols:\n    nuniq = df[c].nunique()\n    nulls = df[c].isnull().sum()\n    print(f'  {c:<20} unique={nuniq:>6,}   nulls={nulls:>6,}')\n\n# bounding box — stored as list [x, y, w, h]\nif 'bbox' in df.columns:\n    has_bbox = df['bbox'].notna() & (df['bbox'].apply(lambda x: len(x) == 4 if isinstance(x, list) else False))\n    print(f'\\n  {\"bbox\":<20} valid={has_bbox.sum():>6,}   nulls={(~has_bbox).sum():>6,}')\n    df.loc[has_bbox, 'bbox_w'] = df.loc[has_bbox, 'bbox'].apply(lambda b: b[2])\n    df.loc[has_bbox, 'bbox_h'] = df.loc[has_bbox, 'bbox'].apply(lambda b: b[3])\n    df.loc[has_bbox, 'bbox_area'] = df['bbox_w'] * df['bbox_h']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:07:47.550700Z","iopub.execute_input":"2026-06-18T22:07:47.551087Z","iopub.status.idle":"2026-06-18T22:07:47.578537Z","shell.execute_reply.started":"2026-06-18T22:07:47.551060Z","shell.execute_reply":"2026-06-18T22:07:47.577452Z"}},"outputs":[],"execution_count":null},{"id":"3c0f1894","cell_type":"code","source":"# ── 5b. location distribution ─────────────────────────────────────────────────\nloc_counts = df['location'].value_counts()\nprint(f'Unique camera locations: {loc_counts.nunique():,}')\nprint(f'Annotations per location — mean: {loc_counts.mean():.0f}, '\n      f'median: {loc_counts.median():.0f}, max: {loc_counts.max():,}')\n\nfig, axes = plt.subplots(1, 2, figsize=(13, 4))\n\n# histogram of annotations per location\naxes[0].hist(loc_counts.values, bins=50, color='#5b8db8', edgecolor='white')\naxes[0].set_xlabel('Annotations per location')\naxes[0].set_ylabel('Number of locations')\naxes[0].set_title('Distribution of annotations per camera location')\n\n# top 20 locations\nloc_counts.head(20).sort_values().plot.barh(ax=axes[1], color='#5b8db8')\naxes[1].set_xlabel('Annotation count')\naxes[1].set_title('Top 20 most-photographed locations')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:07:51.850136Z","iopub.execute_input":"2026-06-18T22:07:51.850446Z","iopub.status.idle":"2026-06-18T22:07:52.374284Z","shell.execute_reply.started":"2026-06-18T22:07:51.850421Z","shell.execute_reply":"2026-06-18T22:07:52.373175Z"}},"outputs":[],"execution_count":null},{"id":"86e80f79","cell_type":"markdown","source":"## 6. Temporal patterns","metadata":{}},{"id":"2e87a55b","cell_type":"code","source":"# ── 6a. activity by hour of day ───────────────────────────────────────────────\nhourly = df.groupby('hour').size().reindex(range(24), fill_value=0)\n\nfig, ax = plt.subplots(figsize=(10, 4))\nax.bar(hourly.index, hourly.values, color='#5b8db8')\nax.axvspan(0, 6,   alpha=0.08, color='navy', label='Night (0–6h)')\nax.axvspan(18, 23, alpha=0.08, color='navy')\nax.set_xlabel('Hour of day')\nax.set_ylabel('Annotation count')\nax.set_title('Annotation volume by hour of day  (shading = night hours)')\nax.set_xticks(range(24))\nax.legend()\nplt.tight_layout()\nplt.show()\n\n# daytime vs nighttime split\ndf['is_night'] = df['hour'].apply(lambda h: h < 6 or h >= 18 if pd.notna(h) else None)\nprint(df['is_night'].value_counts().rename({True:'Night', False:'Day'}).to_string())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:07:56.509380Z","iopub.execute_input":"2026-06-18T22:07:56.509801Z","iopub.status.idle":"2026-06-18T22:07:56.981099Z","shell.execute_reply.started":"2026-06-18T22:07:56.509746Z","shell.execute_reply":"2026-06-18T22:07:56.980358Z"}},"outputs":[],"execution_count":null},{"id":"88b681cd","cell_type":"code","source":"# ── 6b. monthly seasonality ───────────────────────────────────────────────────\nmonth_names = ['Jan','Feb','Mar','Apr','May','Jun',\n               'Jul','Aug','Sep','Oct','Nov','Dec']\nmonthly = df.groupby('month').size().reindex(range(1,13), fill_value=0)\n\nfig, ax = plt.subplots(figsize=(10, 4))\nax.plot(monthly.index, monthly.values, marker='o', color='#5b8db8', linewidth=2)\nax.fill_between(monthly.index, monthly.values, alpha=0.15, color='#5b8db8')\nax.set_xticks(range(1, 13))\nax.set_xticklabels(month_names)\nax.set_ylabel('Annotation count')\nax.set_title('Annotation volume by month')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:08:01.281956Z","iopub.execute_input":"2026-06-18T22:08:01.283695Z","iopub.status.idle":"2026-06-18T22:08:01.508838Z","shell.execute_reply.started":"2026-06-18T22:08:01.283647Z","shell.execute_reply":"2026-06-18T22:08:01.507731Z"}},"outputs":[],"execution_count":null},{"id":"74857acb","cell_type":"code","source":"# ── 6c. top-10 species activity by hour (heatmap) ─────────────────────────────\ntop10_names = species_counts.head(10).index.tolist()\ndf_top = df[df['species'].isin(top10_names)].copy()\n\npivot = df_top.pivot_table(index='species', columns='hour',\n                            values='id_ann', aggfunc='count', fill_value=0)\npivot = pivot.reindex(columns=range(24), fill_value=0)\n\n# normalise each row to show relative activity (0–1)\npivot_norm = pivot.div(pivot.max(axis=1), axis=0)\n\nfig, ax = plt.subplots(figsize=(14, 5))\nsns.heatmap(pivot_norm, ax=ax, cmap='Blues', linewidths=0.3,\n            cbar_kws={'label': 'Relative activity (row-normalised)'})\nax.set_xlabel('Hour of day')\nax.set_ylabel('')\nax.set_title('Activity patterns by hour — top 10 species')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:08:17.505885Z","iopub.execute_input":"2026-06-18T22:08:17.506739Z","iopub.status.idle":"2026-06-18T22:08:18.113929Z","shell.execute_reply.started":"2026-06-18T22:08:17.506707Z","shell.execute_reply":"2026-06-18T22:08:18.112930Z"}},"outputs":[],"execution_count":null},{"id":"56a6fe21","cell_type":"markdown","source":"## 7. Spatial patterns — species per location","metadata":{}},{"id":"54eb4c4d","cell_type":"code","source":"# ── 7a. species richness per location ─────────────────────────────────────────\nrichness = df[df['category_id'] != 0].groupby('location')['species'].nunique()\n\nfig, ax = plt.subplots(figsize=(9, 4))\nrichness.hist(bins=30, ax=ax, color='#7abf8e', edgecolor='white')\nax.set_xlabel('Number of unique species')\nax.set_ylabel('Number of locations')\nax.set_title('Species richness per camera location')\nplt.tight_layout()\nplt.show()\n\nprint(f'Median species per location : {richness.median():.0f}')\nprint(f'Max species in one location : {richness.max()}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:08:24.282590Z","iopub.execute_input":"2026-06-18T22:08:24.282945Z","iopub.status.idle":"2026-06-18T22:08:24.556322Z","shell.execute_reply.started":"2026-06-18T22:08:24.282917Z","shell.execute_reply":"2026-06-18T22:08:24.555417Z"}},"outputs":[],"execution_count":null},{"id":"ac762636","cell_type":"code","source":"# ── 7b. top-5 species: which locations do they appear in? ─────────────────────\ntop5 = species_counts.head(5).index.tolist()\nloc_presence = (\n    df[df['species'].isin(top5)]\n    .groupby(['species', 'location'])\n    .size()\n    .reset_index(name='count')\n    .groupby('species')['location']\n    .nunique()\n    .sort_values(ascending=False)\n)\n\nfig, ax = plt.subplots(figsize=(8, 4))\nloc_presence.plot.bar(ax=ax, color='#7abf8e', edgecolor='white', rot=30)\nax.set_ylabel('Number of camera locations')\nax.set_title('How many locations does each top-5 species appear in?')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:08:28.297643Z","iopub.execute_input":"2026-06-18T22:08:28.298034Z","iopub.status.idle":"2026-06-18T22:08:28.677373Z","shell.execute_reply.started":"2026-06-18T22:08:28.298006Z","shell.execute_reply":"2026-06-18T22:08:28.676364Z"}},"outputs":[],"execution_count":null},{"id":"3b6e0560","cell_type":"markdown","source":"## 8. Bounding box analysis (predictor ↔ target relationship)","metadata":{}},{"id":"c41ca652","cell_type":"code","source":"# Only runs if bbox columns were parsed in section 5a\nif 'bbox_area' in df.columns:\n    df_bbox = df[df['bbox_area'].notna() & df['species'].isin(top10_names)].copy()\n\n    # ── 8a. bbox area distribution per species (boxplot) ──────────────────────\n    fig, ax = plt.subplots(figsize=(12, 5))\n    order = df_bbox.groupby('species')['bbox_area'].median().sort_values(ascending=False).index\n    sns.boxplot(data=df_bbox, x='species', y='bbox_area', order=order,\n                ax=ax, palette='muted', showfliers=False)\n    ax.set_yscale('log')\n    ax.set_xlabel('')\n    ax.set_ylabel('Bounding box area (px², log scale)')\n    ax.set_title('Bounding box size by species — top 10  (outliers hidden)')\n    plt.xticks(rotation=30, ha='right')\n    plt.tight_layout()\n    plt.show()\n\n    # ── 8b. bbox aspect ratio ─────────────────────────────────────────────────\n    df_bbox['aspect_ratio'] = df_bbox['bbox_w'] / df_bbox['bbox_h'].replace(0, np.nan)\n    fig, ax = plt.subplots(figsize=(8, 4))\n    df_bbox['aspect_ratio'].clip(0, 5).hist(bins=60, ax=ax, color='#9b7bc4', edgecolor='white')\n    ax.axvline(1, color='red', linestyle='--', label='Square (ratio=1)')\n    ax.set_xlabel('Width / height ratio')\n    ax.set_ylabel('Count')\n    ax.set_title('Distribution of bounding box aspect ratios')\n    ax.legend()\n    plt.tight_layout()\n    plt.show()\nelse:\n    print('No valid bbox data found — skipping section 8.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:08:38.225195Z","iopub.execute_input":"2026-06-18T22:08:38.225930Z","iopub.status.idle":"2026-06-18T22:08:38.235408Z","shell.execute_reply.started":"2026-06-18T22:08:38.225896Z","shell.execute_reply":"2026-06-18T22:08:38.234382Z"}},"outputs":[],"execution_count":null},{"id":"5b406b0d","cell_type":"markdown","source":"## 9. Correlation matrix — numeric covariates","metadata":{}},{"id":"77161ee8","cell_type":"code","source":"# Build a numeric-only frame for correlation\nnum_cols = ['hour', 'month', 'year', 'frame_num', 'seq_num_frames']\nif 'bbox_area' in df.columns:\n    num_cols += ['bbox_area', 'bbox_w', 'bbox_h']\nnum_cols = [c for c in num_cols if c in df.columns]\n\ndf_num = df[num_cols].dropna()\n\ncorr = df_num.corr()\n\nfig, ax = plt.subplots(figsize=(max(6, len(num_cols)), max(5, len(num_cols)-1)))\nmask = np.triu(np.ones_like(corr, dtype=bool))  # upper triangle mask\nsns.heatmap(corr, mask=mask, annot=True, fmt='.2f', cmap='coolwarm',\n            center=0, linewidths=0.5, ax=ax, vmin=-1, vmax=1)\nax.set_title('Correlation matrix — numeric covariates')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:08:43.901622Z","iopub.execute_input":"2026-06-18T22:08:43.902009Z","iopub.status.idle":"2026-06-18T22:08:44.178418Z","shell.execute_reply.started":"2026-06-18T22:08:43.901981Z","shell.execute_reply":"2026-06-18T22:08:44.177274Z"}},"outputs":[],"execution_count":null},{"id":"8baa8811","cell_type":"markdown","source":"## 10. Day vs night: does species distribution shift?","metadata":{}},{"id":"8cf3f32b","cell_type":"code","source":"# Compare top-10 species proportion in daytime vs nighttime images\ndf_daynight = df[df['species'].isin(top10_names) & df['is_night'].notna()].copy()\ndf_daynight['period'] = df_daynight['is_night'].map({True: 'Night', False: 'Day'})\n\n# normalise within each period\nperiod_species = (\n    df_daynight.groupby(['period', 'species'])\n    .size()\n    .reset_index(name='count')\n)\ntotals = period_species.groupby('period')['count'].transform('sum')\nperiod_species['pct'] = period_species['count'] / totals * 100\n\npivot_dn = period_species.pivot(index='species', columns='period', values='pct').fillna(0)\n\nfig, ax = plt.subplots(figsize=(10, 5))\npivot_dn.sort_values('Day', ascending=False).plot.bar(ax=ax, color=['#f0c040','#3a5f8a'], edgecolor='white')\nax.set_ylabel('% of annotations in that period')\nax.set_xlabel('')\nax.set_title('Species share: day vs night  (top 10 species)')\nplt.xticks(rotation=30, ha='right')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:08:48.173552Z","iopub.execute_input":"2026-06-18T22:08:48.174681Z","iopub.status.idle":"2026-06-18T22:08:48.584901Z","shell.execute_reply.started":"2026-06-18T22:08:48.174643Z","shell.execute_reply":"2026-06-18T22:08:48.583803Z"}},"outputs":[],"execution_count":null},{"id":"5fff958e","cell_type":"markdown","source":"## 11. Summary of findings","metadata":{}},{"id":"3cb57d75","cell_type":"code","source":"n_imgs       = len(df_img)\nn_ann        = len(df_ann)\nn_species    = df_cat['name'].nunique()\nn_locations  = df_img['location'].nunique() if 'location' in df_img.columns else 'N/A'\ntotal_nulls  = df.isnull().sum().sum()\nimbalance_r  = int(species_counts.iloc[0] / species_counts.iloc[-1])\n\nprint('=' * 55)\nprint('  iWildCam 2020 — EDA SUMMARY')\nprint('=' * 55)\nprint(f'  Images                  : {n_imgs:>10,}')\nprint(f'  Annotations             : {n_ann:>10,}')\nprint(f'  Species categories      : {n_species:>10,}')\nprint(f'  Camera locations        : {n_locations if isinstance(n_locations, str) else n_locations:>10,}')\nprint(f'  Total null cells        : {total_nulls:>10,}')\nprint(f'  Class imbalance ratio   : {imbalance_r:>9,}x')\nprint()\nprint('  DEPENDENT VARIABLE')\nprint('    category_id / species — the animal species (or empty)')\nprint('    present in each image.')\nprint()\nprint('  KEY PREDICTORS / COVARIATES')\nprint('    - location     : camera trap site ID (spatial context)')\nprint('    - hour         : time of day (day vs. night imaging)')\nprint('    - month        : seasonal variation')\nprint('    - frame_num    : position within a burst sequence')\nprint('    - seq_num_frames: length of the trigger sequence')\nprint('    - bbox         : animal size/position in frame')\nprint('    - image pixels : raw visual features (for the CNN)')\nprint()\nprint('  KEY CONCERNS FOR MODELLING')\nprint('    ⚠  Severe class imbalance (use weighted loss / SMOTE)')\nprint('    ⚠  Location leakage — split by location, not randomly')\nprint('    ⚠  Day/night domain shift — IR vs colour images')\nprint('    ⚠  Category 0 (empty frames) — decide include or filter')\nprint('=' * 55)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T22:08:52.436382Z","iopub.execute_input":"2026-06-18T22:08:52.437392Z","iopub.status.idle":"2026-06-18T22:08:52.548219Z","shell.execute_reply.started":"2026-06-18T22:08:52.437348Z","shell.execute_reply":"2026-06-18T22:08:52.547223Z"}},"outputs":[],"execution_count":null}]}