{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":126777,"databundleVersionId":15314950,"sourceType":"competition"}],"dockerImageVersionId":31259,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1>📊 1. EDA & Data Integrity</h1>\n<p>In this section, we will:</p>\n<ul>\n   <li>Load the data and verify file integrity.</li>\n   <li>Analyze distribution of jaguars (long-tail problem).</li>\n   <li>Inspect image properties (resolution, RGBA channels).</li>\n   <li><strong>Crucial Step</strong>: Visualize how to use Alpha channel (Mask) to isolate jaguar from background.</li>\n </ul>\n <p>Using mask correctly is key to preventing the model from learning features like \"green bushes\" or \"specific rocks\" instead of jaguar's spots.</p>","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom pathlib import Path\nimport warnings\nfrom PIL import Image\nimport random\nimport imagehash\nfrom tqdm import tqdm\nfrom collections import Counter\nfrom sklearn.model_selection import GroupShuffleSplit\nimport os\nimport math\nfrom multiprocessing import Pool, cpu_count\nfrom scipy.spatial.distance import cdist\nfrom collections import defaultdict\nfrom scipy.sparse import csr_matrix\nfrom scipy.sparse.csgraph import connected_components\n\n# Settings\nsns.set(style=\"whitegrid\", palette=\"muted\")\nwarnings.filterwarnings(\"ignore\")\n\n# Define paths (Kaggle specific)\nBASE_DIR = Path(\"/kaggle/input/jaguar-re-id\")\nTRAIN_DIR = BASE_DIR / \"train/train\"\nTEST_DIR = BASE_DIR / \"test/test\"\n\nprint(f\"Base directory exists: {BASE_DIR.exists()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T09:35:31.268972Z","iopub.execute_input":"2026-02-14T09:35:31.269574Z","iopub.status.idle":"2026-02-14T09:35:31.286097Z","shell.execute_reply.started":"2026-02-14T09:35:31.269514Z","shell.execute_reply":"2026-02-14T09:35:31.284946Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 📂 1.2 Loading Data & Integrity Check\n\nLet's load train.csv and test.csv files and check for any missing files on the disk.","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(BASE_DIR / \"train.csv\")\ntest_df = pd.read_csv(BASE_DIR / \"test.csv\")\n\nprint(\"-\" * 30)\nprint(\"TRAIN SET INFO\")\nprint(\"-\" * 30)\nprint(f\"Shape: {train_df.shape}\")\nprint(f\"Columns: {train_df.columns.tolist()}\")\nprint(f\"Missing values:\\n{train_df.isnull().sum()}\")\nprint(\"\\nFirst 3 rows:\")\ndisplay(train_df.head(3))\n\nprint(\"-\" * 30)\nprint(\"TEST SET INFO\")\nprint(\"-\" * 30)\nprint(f\"Shape: {test_df.shape}\")\nprint(f\"Unique query images: {test_df['query_image'].nunique()}\")\nprint(f\"Unique gallery images: {test_df['gallery_image'].nunique()}\")\nprint(\"\\nFirst 3 rows:\")\ndisplay(test_df.head(3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T09:24:47.296506Z","iopub.execute_input":"2026-02-14T09:24:47.296997Z","iopub.status.idle":"2026-02-14T09:24:47.468408Z","shell.execute_reply.started":"2026-02-14T09:24:47.296968Z","shell.execute_reply":"2026-02-14T09:24:47.467450Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 📏 1.3 Class Distribution Analysis\n\nWe observe a significant Long Tail distribution:\n\n* Some jaguars have 170+ images.\n* Others have fewer than 20.\n\nThis imbalance suggests we might need weighted loss functions or specific data augmentation strategies later on.","metadata":{}},{"cell_type":"code","source":"# Analyze counts per jaguar\nidentity_counts = train_df['ground_truth'].value_counts().sort_values(ascending=False)\ndesc_counts = identity_counts.describe()\n\nprint(f\"Total unique jaguars: {len(identity_counts)}\")\nprint(f\"Max images per jaguar: {desc_counts['max']:.2f}\")\nprint(f\"Min images per jaguar: {desc_counts['min']:.2f}\")\nprint(f\"Mean images per jaguar: {desc_counts['mean']:.2f}\")\n\nplt.figure(figsize=(12, 6))\n\n# Bar chart\nplt.subplot(1, 2, 1)\ncolor_palette = ['red' if x < 20 else 'skyblue' for x in identity_counts.values]\nsns.barplot(x=identity_counts.values, y=identity_counts.index, palette=color_palette)\nplt.title(\"Image Count per Jaguar\", fontsize=14)\nplt.xlabel(\"Number of Images\")\n\n# Histogram\nplt.subplot(1, 2, 2)\nsns.histplot(identity_counts.values, bins=15, kde=True, color='purple')\nplt.title(\"Distribution of Class Sizes\", fontsize=14)\nplt.xlabel(\"Images per Jaguar\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T09:24:47.469750Z","iopub.execute_input":"2026-02-14T09:24:47.470099Z","iopub.status.idle":"2026-02-14T09:24:48.420114Z","shell.execute_reply.started":"2026-02-14T09:24:47.470067Z","shell.execute_reply":"2026-02-14T09:24:48.419085Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 🖼️ 1.4 Image Properties & Resolution\n\nLet's check the resolutions and modes of the images.Note that we expect RGBA mode because of the segmentation mask.","metadata":{}},{"cell_type":"code","source":"def analyze_images(df, img_dir, img_col=\"filename\", limit=None):\n    \"\"\"Analyzes image resolution and mode\"\"\"\n    widths = []\n    heights = []\n    modes = []\n\n    filenames = df[img_col].tolist()\n    if limit:\n        filenames = filenames[:limit]\n    \n    for fn in tqdm(filenames, desc=\"Analyzing images\"):\n        path = img_dir / fn\n        try:\n            with Image.open(path) as img:\n                widths.append(img.width)\n                heights.append(img.height)\n                modes.append(img.mode)\n        except Exception as e:\n            print(f\"Error loading {fn}: {e}\")\n    \n    return widths, heights, modes\n\n# Analyze images\nwidths, heights, modes = analyze_images(train_df, TRAIN_DIR)\n\n# Stats DataFrame\nstats_df = pd.DataFrame({\n    'width': widths,\n    'height': heights,\n    'mode': modes\n})\n\nprint(\"Image Statistics (Sample):\")\nprint(stats_df.describe())\nprint(\"\\nMode distribution:\")\nprint(stats_df['mode'].value_counts())\n\n# Plot Histograms\nplt.figure(figsize=(8, 4))\nplt.hist(widths, bins=50, alpha=0.6, label=\"width\")\nplt.hist(heights, bins=50, alpha=0.6, label=\"height\")\nplt.legend()\nplt.title(\"Distribution of image widths and heights (train)\")\nplt.xlabel(\"Pixels\")\nplt.ylabel(\"Count\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T09:24:48.422398Z","iopub.execute_input":"2026-02-14T09:24:48.422777Z","iopub.status.idle":"2026-02-14T09:25:09.109720Z","shell.execute_reply.started":"2026-02-14T09:24:48.422742Z","shell.execute_reply":"2026-02-14T09:25:09.108552Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 🎨 1.5 Visualizing Samples & RGBA Masking\n\nThe images are in **RGBA** mode:\n\n- RGB: The jaguar itself.\n- A (Alpha): The segmentation mask.\n\nWe will visualize:\n\n1. The RGB image (as is).\n2. The Alpha mask (Black/White).\n3. The RGB image pasted onto a black background  using the mask.\n\nThis proves we can successfully isolate the jaguar from the cluttered background.","metadata":{}},{"cell_type":"code","source":"def show_jaguar_images(jaguar_id, n_max=8):\n    \"\"\"Shows grid images for specific jaguar\"\"\"\n    fnames = train_df[train_df[\"ground_truth\"] == jaguar_id][\"filename\"].head(n_max).tolist()\n    n = len(fnames)\n    cols = 4\n    rows = (n + cols - 1) // cols\n\n    plt.figure(figsize=(cols * 3, rows * 3))\n    for i, fn in enumerate(fnames):\n        path = TRAIN_DIR / fn\n        with Image.open(path) as img:\n            # Convert to RGB for simple display\n            if img.mode == \"RGBA\":\n                img = img.convert(\"RGB\")\n            plt.subplot(rows, cols, i + 1)\n            plt.imshow(img)\n            plt.title(fn)\n            plt.axis(\"off\")\n    plt.suptitle(f\"Jaguar: {jaguar_id}\", fontsize=16)\n    plt.tight_layout()\n    plt.show()\n\nshow_jaguar_images(\"Marcela\", n_max=8)\nshow_jaguar_images(\"Ousado\", n_max=8)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T09:25:09.110926Z","iopub.execute_input":"2026-02-14T09:25:09.111236Z","iopub.status.idle":"2026-02-14T09:25:21.102951Z","shell.execute_reply.started":"2026-02-14T09:25:09.111209Z","shell.execute_reply":"2026-02-14T09:25:21.101957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_rgba_samples(df, img_dir, n=3):\n    sample_fns = df['filename'].sample(n, random_state=21).tolist()\n\n    for fn in sample_fns:\n        path = img_dir / fn\n        with Image.open(path) as img:\n            if img.mode == \"RGBA\":\n                rgb = img.convert(\"RGB\")\n                r, g, b, a = img.split()\n                \n                fig, axs = plt.subplots(1, 3, figsize=(16, 4))\n                \n                # 1. RGB Image\n                axs[0].imshow(rgb)\n                axs[0].set_title(\"Original RGB\")\n                axs[0].axis(\"off\")\n                \n                # 2. Alpha Mask\n                axs[1].imshow(a, cmap='gray')\n                axs[1].set_title(\"Alpha Mask (Segmentation)\")\n                axs[1].axis(\"off\")\n\n                # 3. Isolated Jaguar (Black Background)\n                bg = Image.new(\"RGB\", img.size, (0, 0, 0))\n                bg.paste(rgb, mask=a)\n                axs[2].imshow(bg)\n                axs[2].set_title(\"Isolated Jaguar\")\n                axs[2].axis(\"off\")\n                \n                plt.tight_layout()\n                plt.show()\n\nvisualize_rgba_samples(train_df, TRAIN_DIR, n=4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T09:25:21.104165Z","iopub.execute_input":"2026-02-14T09:25:21.104544Z","iopub.status.idle":"2026-02-14T09:25:31.485241Z","shell.execute_reply.started":"2026-02-14T09:25:21.104517Z","shell.execute_reply":"2026-02-14T09:25:31.484260Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🕵️‍♂️ 2. Near-Duplicate Problem & Clean Split Strategy\n\n## 📂 2.1 Introduction to Data Leakage\n\nIn Camera Trap / Re-ID datasets, photos are often taken in bursts (rapid fire).This results in many Near-Duplicates — photos that are visually identical (same jaguar, same pose, same bush).\n\nIf we use a standard train_test_split:\n- Image A (duplicate) goes to **Train**.\n- Image B (duplicate) goes to **Validation**.\n\nThe model sees \"Image A\" during training, recognizes it easily as \"Image B\" during validation. **Your validation score becomes inflated (Fake High Score)**, but real performance drops.\n\n**Solution**:We must group duplicates into \"sessions\". All photos from one session go entirely to Train OR entirely to Val.\n\n## 🔍 2.2 Finding Duplicates via Image Hashing\n\nWe use **Perceptual Hashing (pHash)**. It converts an image to a hash string. Similar images have hashes with a small Hamming distance.\n\n* We group images with distance <= **optimal threshold**.\n\nWe find near-duplicates in the training set to prevent data leakage during validation splits.","metadata":{}},{"cell_type":"code","source":"def _calculate_hash(args):\n    \"\"\"Worker function to calculate pHash for a single file.\"\"\"\n    fn, img_dir = args\n    try:\n        path = Path(img_dir) / fn\n        with Image.open(path) as img:\n            # Convert to RGB if RGBA to ensure consistency\n            if img.mode == \"RGBA\": \n                img = img.convert(\"RGB\")\n            h = imagehash.phash(img)\n            return fn, h.hash.flatten().astype(np.int8)\n    except Exception:\n        return fn, None\n\n# 1. Calculate Hashes (Parallel)\nprint(f\"Calculating perceptual hashes using {cpu_count()} CPU cores...\")\nargs_list = [(fn, str(TRAIN_DIR)) for fn in train_df['filename']]\n\nwith Pool(processes=cpu_count()) as pool:\n    results = list(tqdm(pool.imap(_calculate_hash, args_list), total=len(train_df)))\n\n# Filter valid results\nvalid_results = [r for r in results if r[1] is not None]\nfilenames_clean = [r[0] for r in valid_results]\nhash_matrix = np.array([r[1] for r in valid_results], dtype=np.int8)\n\n# Get labels aligned with hash_matrix\nlabel_map = train_df.set_index('filename')['ground_truth'].to_dict()\nlabel_vector = np.array([label_map[fn] for fn in filenames_clean])\n\n# 2. Compute Distance Matrix (Vectorized)\nprint(\"Computing distance matrix...\")\ndist_matrix = cdist(hash_matrix, hash_matrix, metric='hamming') * hash_matrix.shape[1]\n\n# 3. Optimal Threshold Search (Vectorized)\nprint(\"Searching for optimal threshold...\")\nN = len(filenames_clean)\n\n# Take upper triangle only (unique pairs)\ntriu_indices = np.triu_indices(N, k=1)\ndists = dist_matrix[triu_indices]\nlabels_match = (label_vector[triu_indices[0]] == label_vector[triu_indices[1]])\n\n# Vectorized statistics calculation for thresholds 0-20\nthresholds = np.arange(0, 21)\nstats = []\n\nfor t in thresholds:\n    mask = (dists <= t)\n    \n    correct_links = np.sum(mask & labels_match)\n    collisions = np.sum(mask & ~labels_match)\n    \n    stats.append({\n        'threshold': t,\n        'correct_links': correct_links,\n        'collisions': collisions\n    })\n\nres_df = pd.DataFrame(stats)\n\n# Recommendation: Max threshold with 0 collisions\nsafe_threshold = res_df[res_df['collisions'] == 0]['threshold'].max()\nif pd.isna(safe_threshold): safe_threshold = 0\n\nprint(f\"✅ Safe Threshold selected: {safe_threshold}\")\n\n# 4. Build Groups using Connected Components\n# Create binary adjacency matrix\nadjacency = (dist_matrix <= safe_threshold)\n# Fill diagonal with False to prevent self-loops\nnp.fill_diagonal(adjacency, False)\n\n# Build graph and find connected components\n# This automatically merges transitive links (A-B and B-C -> A-B-C)\ngraph = csr_matrix(adjacency)\nn_components, labels = connected_components(csgraph=graph, directed=False, return_labels=True)\n\n# Collect groups\ndup_groups = []\nfor i in range(n_components):\n    indices = np.where(labels == i)[0]\n    if len(indices) > 1:\n        group = [filenames_clean[idx] for idx in indices]\n        dup_groups.append(group)\n\nprint(f\"\\n✅ Found {len(dup_groups)} groups of near-duplicates.\")\nprint(f\"📸 Total images involved in duplicate groups: {sum(len(g) for g in dup_groups)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T09:35:41.704555Z","iopub.execute_input":"2026-02-14T09:35:41.704921Z","iopub.status.idle":"2026-02-14T09:40:35.353764Z","shell.execute_reply.started":"2026-02-14T09:35:41.704889Z","shell.execute_reply":"2026-02-14T09:40:35.352598Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 🖼️ 2.3 Visualizing Duplicate Groups\n\nLet's look at some duplicate groups to ensure our logic is correct.","metadata":{}},{"cell_type":"code","source":"def visualize_duplicate_groups(groups, img_dir, n_groups_to_show=3):\n    for group in groups[:n_groups_to_show]:\n        print(f\"Group size: {len(group)} images\")\n        plt.figure(figsize=(15, 5))\n        for i, fn in enumerate(group):\n            path = img_dir / fn\n            with Image.open(path) as img:\n                plt.subplot(1, len(group), i + 1)\n                plt.imshow(img.convert('RGB'))\n                plt.title(fn, fontsize=8)\n                plt.axis('off')\n        plt.show()\n        print(\"-\" * 50)\n\n# Show first 3 groups\nvisualize_duplicate_groups(dup_groups, TRAIN_DIR, n_groups_to_show=3)\n\n# Find the largest group to visualize\nlargest_group = max(dup_groups, key=len)\nprint(f\"Largest group has {len(largest_group)} images.\")\n\ndef visualize_large_group(group, img_dir):\n    n = len(group)\n    cols = 5\n    rows = (n + cols - 1) // cols\n    \n    plt.figure(figsize=(cols * 3, rows * 3))\n    \n    for i, fn in enumerate(group):\n        path = img_dir / fn\n        with Image.open(path) as img:\n            plt.subplot(rows, cols, i + 1)\n            plt.imshow(img.convert('RGB'))\n            plt.title(fn.split('/')[-1], fontsize=8)\n            plt.axis('off')\n            \n    plt.suptitle(f\"Duplicate Group Analysis: {n} images\", fontsize=16)\n    plt.tight_layout()\n    plt.show()\n\n# Check who owns this large group\nfirst_file = largest_group[0]\njaguar_name = train_df[train_df['filename'] == first_file]['ground_truth'].values[0]\nprint(f\"These {len(largest_group)} images belong to: {jaguar_name}\")\nvisualize_large_group(largest_group, TRAIN_DIR)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 📉 2.4 Real Data Deficit\n\nFor rare jaguars, model has very few unique viewpoints to learn from.","metadata":{}},{"cell_type":"code","source":"# 1. Identify all files that are part of a duplicate group\nduplicate_files = set()\nfor group in dup_groups:\n    duplicate_files.update(group)\n\nprint(f\"Total files marked as duplicates: {len(duplicate_files)}\")\n\n# 2. Add flag to DataFrame\ntrain_df['is_duplicate'] = train_df['filename'].apply(lambda x: x in duplicate_files)\n\n# 3. Aggregate statistics per jaguar\njaguar_stats = train_df.groupby('ground_truth').agg(\n    total_images=('filename', 'count'),\n    duplicate_images=('is_duplicate', 'sum')\n).reset_index()\n\n# Calculate unique count\njaguar_stats['unique_images'] = jaguar_stats['total_images'] - jaguar_stats['duplicate_images']\njaguar_stats = jaguar_stats.sort_values('total_images', ascending=False)\n\nprint(jaguar_stats)\n\n# 4. Visualization: Unique vs Duplicates\nplt.figure(figsize=(14, 8))\n\nplt.bar(jaguar_stats['ground_truth'], jaguar_stats['unique_images'], label='Unique Images', color='skyblue')\nplt.bar(jaguar_stats['ground_truth'], jaguar_stats['duplicate_images'], bottom=jaguar_stats['unique_images'], \n        label='Near-Duplicates', color='orange')\n\nplt.title(\"Unique Images vs Near-Duplicates per Jaguar\", fontsize=16)\nplt.xlabel(\"Jaguar Identity\", fontsize=12)\nplt.ylabel(\"Image Count\", fontsize=12)\nplt.xticks(rotation=90)\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T09:40:35.400790Z","iopub.execute_input":"2026-02-14T09:40:35.401140Z","iopub.status.idle":"2026-02-14T09:40:35.913307Z","shell.execute_reply.started":"2026-02-14T09:40:35.401101Z","shell.execute_reply":"2026-02-14T09:40:35.912256Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## ✅ 2.5 Clean Split Strategy: GroupShuffleSplit\n\nTo prevent leakage, we use GroupShuffleSplit.\n\nLogic:\n\n1. Assign a group_id to every image.\n    - If image is in a duplicate group -> group_id = group_0, group_1...\n    - If image is unique -> group_id = filename (it is its own group).\n2. Split based on group_id.\n3. Verify: Ensure no group appears in both Train and Val.","metadata":{}},{"cell_type":"code","source":"# 1. Assign group IDs using the MERGED dup_groups list\nimage_to_group_id = {}\n\n# Assign IDs to groups\nfor group_idx, group in enumerate(dup_groups):\n    group_id = f\"group_{group_idx}\"\n    for filename in group:\n        image_to_group_id[filename] = group_id\n\n# 2. Handle unique files (files not in any group)\nall_files = set(train_df['filename'].tolist())\nunique_files = all_files - set(image_to_group_id.keys())\nfor filename in unique_files:\n    image_to_group_id[filename] = filename # unique group for unique file\n\n# 3. Add to DataFrame\ntrain_df['group_id'] = train_df['filename'].map(image_to_group_id)\n\nprint(\"Sample of train_df with group_id:\")\nprint(train_df[['filename', 'ground_truth', 'group_id']].head(15))\n\n# 4. Perform Split\ngss = GroupShuffleSplit(n_splits=1, test_size=0.3, random_state=21)\ntrain_idx, val_idx = next(gss.split(train_df, groups=train_df['group_id']))\n\ntrain_split = train_df.iloc[train_idx].reset_index(drop=True)\nval_split = train_df.iloc[val_idx].reset_index(drop=True)\n\nprint(f\"\\nTrain size: {len(train_split)} images\")\nprint(f\"Val size:   {len(val_split)} images\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T09:41:14.926383Z","iopub.execute_input":"2026-02-14T09:41:14.926897Z","iopub.status.idle":"2026-02-14T09:41:14.951230Z","shell.execute_reply.started":"2026-02-14T09:41:14.926853Z","shell.execute_reply":"2026-02-14T09:41:14.950406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Get all group_ids in Val\nval_groups = set(val_split['group_id'].tolist())\n\n# 2. Check if any of these groups exist in Train\nleakage_check = train_split[train_split['group_id'].isin(val_groups)]\n\nprint(f\"Train size: {len(train_split)}\")\nprint(f\"Leaked images found (should be 0): {len(leakage_check)}\")\n\nif len(leakage_check) == 0:\n    print(\"✅ SUCCESS: No leakage. Train and Val are strictly separated by duplicate groups.\")\nelse:\n    print(\"❌ ERROR: Data Leakage detected! Some duplicate groups are in both sets.\")\n    display(leakage_check)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T09:41:14.952277Z","iopub.execute_input":"2026-02-14T09:41:14.952531Z","iopub.status.idle":"2026-02-14T09:41:14.977462Z","shell.execute_reply.started":"2026-02-14T09:41:14.952508Z","shell.execute_reply":"2026-02-14T09:41:14.976408Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 📊 2.6 Final Split Statistics\n\nLet's visualize distribution in our new clean Train and Validation sets.","metadata":{}},{"cell_type":"code","source":"def get_split_stats(df_split):\n    stats = df_split.groupby('ground_truth').agg(\n        total=('filename', 'count'),\n        dups=('is_duplicate', 'sum')\n    ).reset_index()\n    stats['unique'] = stats['total'] - stats['dups']\n    stats = stats.sort_values('total', ascending=False)\n    return stats\n\ntrain_stats = get_split_stats(train_split)\nval_stats = get_split_stats(val_split)\n\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(24, 8), sharey=True)\n\n# Train Plot\nax1.bar(train_stats['ground_truth'], train_stats['unique'], label='Unique', color='skyblue')\nax1.bar(train_stats['ground_truth'], train_stats['dups'], bottom=train_stats['unique'], label='Duplicates', color='orange')\nax1.set_title('TRAIN SET Distribution', fontsize=16)\nax1.set_ylabel('Image Count', fontsize=12)\nax1.tick_params(axis='x', rotation=90)\n\n# Val Plot\nax2.bar(val_stats['ground_truth'], val_stats['unique'], label='Unique', color='skyblue')\nax2.bar(val_stats['ground_truth'], val_stats['dups'], bottom=val_stats['unique'], label='Duplicates', color='orange')\nax2.set_title('VALIDATION SET Distribution', fontsize=16)\nax2.tick_params(axis='x', rotation=90)\n\n# Common Legend\nhandles, labels = ax1.get_legend_handles_labels()\nfig.legend(handles, labels, loc='upper center', bbox_to_anchor=(0.5, 1.02), ncol=2, fontsize=14)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-14T09:41:14.978444Z","iopub.execute_input":"2026-02-14T09:41:14.979299Z","iopub.status.idle":"2026-02-14T09:41:15.877078Z","shell.execute_reply.started":"2026-02-14T09:41:14.979259Z","shell.execute_reply":"2026-02-14T09:41:15.875685Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🚀 Conclusion\nIf you found this notebook helpful or clean split strategy useful for your modeling, please consider giving it an Upvote 👍.\n\nIt helps others find this resource.\n\nGood luck with competition! 🐆","metadata":{}}]}