{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# A Practical Field Guide to Petals to the Metal\n\nThis notebook is an operations-first EDA for the TPU Getting Started competition.\nInstead of generic charts, the goal is to answer the questions that actually matter\nbefore you train your first serious baseline.\n\nWhat we will verify:\n- What is physically present in the mounted data path?\n- How many examples are implied by TFRecord shard naming?\n- Are there signs of split imbalance or resolution-driven tradeoffs?\n- Can we decode records reliably and inspect real samples quickly?\n- What are the exact submission constraints we must never violate?\n\nIf you are trying to climb the leaderboard efficiently, this is the preflight checklist.","metadata":{}},{"cell_type":"code","source":"from __future__ import annotations\n\nimport re\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nsns.set_theme(style='whitegrid', context='notebook')\nplt.rcParams['figure.dpi'] = 120\nplt.rcParams['axes.titlesize'] = 12\nplt.rcParams['axes.labelsize'] = 10\npd.set_option('display.max_columns', 100)\npd.set_option('display.width', 160)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T04:36:31.521895Z","iopub.execute_input":"2026-08-19T04:36:31.522716Z","iopub.status.idle":"2026-08-19T04:36:35.180894Z","shell.execute_reply.started":"2026-08-19T04:36:31.522683Z","shell.execute_reply":"2026-08-19T04:36:35.179976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CANDIDATES = [\n    Path('/kaggle/input/competitions/tpu-getting-started'),\n    Path('/kaggle/input/tpu-getting-started'),\n    Path(r'D:/GitHub/kaggle-agent/competitions/Petal Metal/data/tpu-getting-started'),\n]\n\nDATA = next((p for p in CANDIDATES if p.exists()), CANDIDATES[0])\nprint(f'Using data path: {DATA}')\nprint('Exists:', DATA.exists())\n\ncomp_facts = pd.DataFrame(\n    [\n        ('Competition', 'Petals to the Metal - Flower Classification on TPU'),\n        ('Metric', 'Macro F1 (multiclass)'),\n        ('Code requirement', 'Submission from Kaggle Notebook/Script output'),\n        ('Runtime cap', '3 hours per notebook session'),\n        ('Primary failure mode', 'Bad submission format or mismatched rows/columns'),\n    ],\n    columns=['Field', 'Value'],\n)\ncomp_facts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T04:36:35.182678Z","iopub.execute_input":"2026-08-19T04:36:35.183209Z","iopub.status.idle":"2026-08-19T04:36:35.228709Z","shell.execute_reply.started":"2026-08-19T04:36:35.183178Z","shell.execute_reply":"2026-08-19T04:36:35.227790Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1) Data Mount Health Check\n\nBefore any modeling, verify that your notebook can actually see the expected assets.\nThis avoids wasting a full run on path assumptions that break only on Kaggle.","metadata":{}},{"cell_type":"code","source":"def safe_list(path: Path):\n    if not path.exists():\n        return []\n    return sorted(path.iterdir(), key=lambda p: p.name)\n\nroot_items = safe_list(DATA)\ninv = []\nfor item in root_items:\n    inv.append(\n        {\n            'name': item.name,\n            'kind': 'DIR' if item.is_dir() else 'FILE',\n            'size_mb': round(item.stat().st_size / (1024**2), 3) if item.is_file() else np.nan,\n        }\n    )\n\ninventory = pd.DataFrame(inv)\nprint(f'Items at root: {len(inventory)}')\nif inventory.empty:\n    print('No visible data items under DATA path.')\nelse:\n    display(inventory)\n\nmissing_core = [x for x in ['sample_submission.csv'] if not (DATA / x).exists()]\nprint('Missing core files:', missing_core if missing_core else 'None')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T04:36:35.229747Z","iopub.execute_input":"2026-08-19T04:36:35.230190Z","iopub.status.idle":"2026-08-19T04:36:35.249629Z","shell.execute_reply.started":"2026-08-19T04:36:35.230155Z","shell.execute_reply":"2026-08-19T04:36:35.248509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_path = DATA / 'sample_submission.csv'\nif sample_path.exists():\n    sample = pd.read_csv(sample_path)\n    print('sample_submission shape:', sample.shape)\n    display(sample.head())\nelse:\n    print('sample_submission.csv not found at', sample_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T04:36:35.250694Z","iopub.execute_input":"2026-08-19T04:36:35.251098Z","iopub.status.idle":"2026-08-19T04:36:35.279003Z","shell.execute_reply.started":"2026-08-19T04:36:35.251070Z","shell.execute_reply":"2026-08-19T04:36:35.278138Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2) TFRecord Topology by Resolution and Split\n\nTFRecord filenames carry useful metadata: shard index, resolution, and records per shard.\nWe aggregate this into a compact operational view of data volume and fragmentation.","metadata":{}},{"cell_type":"code","source":"pattern = re.compile(r'(?P<shard>\\d+)-(?P<size>\\d+x\\d+)-(?P<count>\\d+)\\.tfrec$')\nsize_roots = sorted(DATA.glob('tfrecords-jpeg-*x*')) if DATA.exists() else []\nrecords = []\n\nfor size_dir in size_roots:\n    for split in ['train', 'val', 'test']:\n        folder = size_dir / split\n        if not folder.exists():\n            continue\n        for f in sorted(folder.glob('*.tfrec')):\n            m = pattern.match(f.name)\n            if not m:\n                continue\n            records.append(\n                {\n                    'resolution': m.group('size'),\n                    'split': split,\n                    'shard': int(m.group('shard')),\n                    'examples_in_shard': int(m.group('count')),\n                    'bytes': f.stat().st_size,\n                    'size_root': size_dir.name,\n                    'path': str(f),\n                }\n            )\n\nshards = pd.DataFrame(records)\nif shards.empty:\n    print('No TFRecord shards found under expected folders.')\nelse:\n    shards['mb'] = shards['bytes'] / (1024**2)\n    shards['examples_per_mb'] = shards['examples_in_shard'] / shards['mb'].replace(0, np.nan)\n\n    display(shards.head())\n\n    summary = (\n        shards.groupby(['resolution', 'split'], as_index=False)\n        .agg(\n            n_shards=('shard', 'count'),\n            est_examples=('examples_in_shard', 'sum'),\n            total_gb=('bytes', lambda x: x.sum() / (1024**3)),\n            avg_examples_per_shard=('examples_in_shard', 'mean'),\n            cv_examples_per_shard=('examples_in_shard', lambda x: np.std(x) / (np.mean(x) + 1e-9)),\n        )\n        .sort_values(['resolution', 'split'])\n    )\n    display(summary)\n\n    with pd.option_context('display.float_format', lambda x: f'{x:,.3f}'):\n        print('Total estimated examples by split:')\n        print(shards.groupby('split')['examples_in_shard'].sum().sort_values(ascending=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T04:36:35.280014Z","iopub.execute_input":"2026-08-19T04:36:35.280339Z","iopub.status.idle":"2026-08-19T04:36:35.580474Z","shell.execute_reply.started":"2026-08-19T04:36:35.280310Z","shell.execute_reply":"2026-08-19T04:36:35.579563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if not shards.empty:\n    fig, axes = plt.subplots(1, 2, figsize=(13.5, 4.5))\n\n    plot_df = (\n        shards.groupby(['resolution', 'split'], as_index=False)['examples_in_shard']\n        .sum()\n        .rename(columns={'examples_in_shard': 'estimated_examples'})\n    )\n    sns.barplot(data=plot_df, x='resolution', y='estimated_examples', hue='split', ax=axes[0])\n    axes[0].set_title('Estimated examples by resolution and split')\n    axes[0].set_xlabel('Resolution')\n    axes[0].set_ylabel('Estimated examples')\n    for container in axes[0].containers:\n        axes[0].bar_label(container, fmt='%.0f', fontsize=8)\n\n    sns.boxplot(data=shards, x='split', y='mb', hue='resolution', ax=axes[1], fliersize=1.5)\n    axes[1].set_title('Shard file size distribution (MB)')\n    axes[1].set_xlabel('Split')\n    axes[1].set_ylabel('Shard size (MB)')\n\n    handles, labels = axes[1].get_legend_handles_labels()\n    if handles:\n        axes[1].legend(handles, labels, title='Resolution', loc='upper right')\n\n    plt.tight_layout()\n\n    split_imbalance = (\n        shards.groupby('split')['examples_in_shard'].sum()\n        .pipe(lambda s: s / s.sum())\n        .sort_values(ascending=False)\n    )\n    print('\\nSplit proportion (by estimated examples):')\n    print(split_imbalance.apply(lambda x: f'{x:.2%}'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T04:36:35.582439Z","iopub.execute_input":"2026-08-19T04:36:35.582713Z","iopub.status.idle":"2026-08-19T04:36:36.423643Z","shell.execute_reply.started":"2026-08-19T04:36:35.582685Z","shell.execute_reply":"2026-08-19T04:36:36.422612Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3) Submission Contract Audit\n\nLeaderboard progress is often blocked by file-format mistakes, not model quality.\nThis section enforces the submission contract explicitly.","metadata":{}},{"cell_type":"code","source":"sample_path = DATA / 'sample_submission.csv'\nif sample_path.exists():\n    sample = pd.read_csv(sample_path)\n    expected_columns = ['id', 'label']\n\n    print('Expected columns:', expected_columns)\n    print('Actual columns:  ', sample.columns.tolist())\n    print('Rows:', len(sample))\n    print('Unique ids:', sample['id'].nunique())\n    print('Duplicate ids:', int(sample['id'].duplicated().sum()))\n    print('Missing labels:', int(sample['label'].isna().sum()))\n\n    id_len = sample['id'].astype(str).str.len()\n    print('ID length range:', (int(id_len.min()), int(id_len.max())))\n\n    display(sample.head(8))\n\n    assert sample.columns.tolist() == expected_columns, 'Column mismatch in sample_submission format.'\n    assert sample['id'].nunique() == len(sample), 'Duplicate ids found.'\nelse:\n    print('sample_submission.csv not found at', sample_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T04:36:36.424873Z","iopub.execute_input":"2026-08-19T04:36:36.425215Z","iopub.status.idle":"2026-08-19T04:36:36.452976Z","shell.execute_reply.started":"2026-08-19T04:36:36.425186Z","shell.execute_reply":"2026-08-19T04:36:36.452088Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4) Record-Level Probe (Labels + Image Samples)\n\nNow we inspect actual TFRecord payloads.\n- First: parse labels from a bounded sample of records to estimate class spread.\n- Then: decode a small image grid for human sanity-checks.\n\nThe logic is resilient to minor schema differences (`class` vs `label`, `image` vs `image_raw`).","metadata":{}},{"cell_type":"code","source":"try:\n    import tensorflow as tf\nexcept Exception as e:\n    tf = None\n    print('TensorFlow not available in this environment:', e)\n\nif tf is None or shards.empty:\n    print('Skipping TFRecord record-level probe.')\nelse:\n    preferred = shards.loc[shards['split'].eq('train')]\n    if preferred.empty:\n        preferred = shards.loc[shards['split'].eq('val')]\n    if preferred.empty:\n        preferred = shards.copy()\n\n    probe_files = preferred.sort_values(['resolution', 'shard']).head(3)['path'].tolist()\n    print('Probing files:')\n    for p in probe_files:\n        print('-', p)\n\n    label_schemas = [\n        {'class': tf.io.FixedLenFeature([], tf.int64)},\n        {'label': tf.io.FixedLenFeature([], tf.int64)},\n    ]\n\n    image_schemas = [\n        {'class': tf.io.FixedLenFeature([], tf.int64), 'image': tf.io.FixedLenFeature([], tf.string)},\n        {'label': tf.io.FixedLenFeature([], tf.int64), 'image': tf.io.FixedLenFeature([], tf.string)},\n        {'class': tf.io.FixedLenFeature([], tf.int64), 'image_raw': tf.io.FixedLenFeature([], tf.string)},\n        {'label': tf.io.FixedLenFeature([], tf.int64), 'image_raw': tf.io.FixedLenFeature([], tf.string)},\n    ]\n\n    # Label probe (bounded for speed)\n    labels = []\n    max_records = 1200\n    for path in probe_files:\n        ds = tf.data.TFRecordDataset([path]).take(max_records)\n        for raw in ds:\n            parsed = None\n            for schema in label_schemas:\n                try:\n                    parsed = tf.io.parse_single_example(raw, schema)\n                    break\n                except Exception:\n                    continue\n            if parsed is None:\n                continue\n            key = 'class' if 'class' in parsed else 'label'\n            labels.append(int(parsed[key].numpy()))\n\n    if labels:\n        label_series = pd.Series(labels, name='label')\n        print('Parsed labels:', len(label_series))\n        print('Distinct labels in probe:', label_series.nunique())\n\n        top_labels = label_series.value_counts().head(12)\n        fig, ax = plt.subplots(figsize=(8.5, 3.8))\n        sns.barplot(x=top_labels.index.astype(str), y=top_labels.values, color='#2a9d8f', ax=ax)\n        ax.set_title('Most frequent labels in the probe sample')\n        ax.set_xlabel('Label')\n        ax.set_ylabel('Count')\n        plt.xticks(rotation=45)\n        plt.tight_layout()\n        plt.show()\n    else:\n        print('No labels parsed from probe files with fallback schemas.')\n\n    # Image grid probe\n    decoded_images = []\n    decoded_labels = []\n    target_images = 9\n    for path in probe_files:\n        ds = tf.data.TFRecordDataset([path]).take(300)\n        for raw in ds:\n            parsed = None\n            used_schema = None\n            for schema in image_schemas:\n                try:\n                    parsed = tf.io.parse_single_example(raw, schema)\n                    used_schema = schema\n                    break\n                except Exception:\n                    continue\n            if parsed is None:\n                continue\n\n            label_key = 'class' if 'class' in parsed else 'label'\n            image_key = 'image' if 'image' in parsed else 'image_raw'\n            try:\n                image = tf.io.decode_jpeg(parsed[image_key].numpy()).numpy()\n                decoded_images.append(image)\n                decoded_labels.append(int(parsed[label_key].numpy()))\n            except Exception:\n                continue\n\n            if len(decoded_images) >= target_images:\n                break\n        if len(decoded_images) >= target_images:\n            break\n\n    if decoded_images:\n        fig, axes = plt.subplots(3, 3, figsize=(8, 8))\n        for i, ax in enumerate(axes.ravel()):\n            if i < len(decoded_images):\n                ax.imshow(decoded_images[i])\n                ax.set_title(f'label={decoded_labels[i]}', fontsize=9)\n                ax.axis('off')\n            else:\n                ax.axis('off')\n        fig.suptitle('Sample decoded flowers from TFRecord probe', y=1.01)\n        plt.tight_layout()\n        plt.show()\n    else:\n        print('Could not decode sample images from probed records.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-19T04:36:36.454221Z","iopub.execute_input":"2026-08-19T04:36:36.454587Z","iopub.status.idle":"2026-08-19T04:37:09.164461Z","shell.execute_reply.started":"2026-08-19T04:36:36.454546Z","shell.execute_reply":"2026-08-19T04:37:09.163371Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5) Actionable Takeaways\n\n1. **Data path reliability first**: verify mounted files before spending TPU time.\n2. **Resolution is a primary experiment axis**: data is sharded across multiple sizes, so runtime/accuracy tradeoffs are expected.\n3. **Shard diagnostics are useful early warnings**: unusually skewed shard sizes can signal ingestion issues.\n4. **Submission integrity is non-negotiable**: keep strict checks for column names, row count, and unique IDs every run.\n5. **Record probing reduces blind spots**: quick label/image extraction catches schema assumptions early.\n\n### Suggested next notebook\nA compact baseline-training notebook that logs:\n- train/val macro F1 per epoch\n- inference latency per 1k examples\n- submission smoke checks before final write\n\nIf this field guide saved you setup/debug time, an upvote is appreciated.","metadata":{}},{"cell_type":"code","source":"# ==========================================\n# Create submission.csv for Petals to Metal\n# ==========================================\n\nimport os\nimport glob\nimport pandas as pd\nimport numpy as np\n\n# 1) البحث عن sample_submission.csv تلقائياً\nfiles = glob.glob('/kaggle/input/**/sample_submission.csv', recursive=True)\n\nif len(files) == 0:\n    raise FileNotFoundError(\"لم يتم العثور على sample_submission.csv\")\n\nsample_path = files[0]\n\nprint(\"Found sample submission:\")\nprint(sample_path)\n\n# 2) قراءة نموذج ملف التسليم\nsubmission = pd.read_csv(sample_path)\n\nprint(\"\\nColumns:\")\nprint(submission.columns)\n\nprint(\"\\nNumber of rows:\")\nprint(len(submission))\n\nprint(\"\\nFirst rows:\")\nprint(submission.head())\n\n# 3) إنشاء توقعات مبدئية\n# مسابقة Petals to the Metal تحتوي على 104 فئات\nnp.random.seed(42)\nsubmission['label'] = np.random.randint(\n    0,\n    104,\n    size=len(submission)\n)\n\n# 4) حفظ الملف بالاسم الذي تطلبه Kaggle\noutput_path = '/kaggle/working/submission.csv'\n\nsubmission.to_csv(\n    output_path,\n    index=False\n)\n\n# 5) التأكد من إنشاء الملف\nprint(\"\\n=================================\")\nprint(\"SUBMISSION CREATED SUCCESSFULLY\")\nprint(\"=================================\")\n\nprint(\"Path:\", output_path)\nprint(\"Exists:\", os.path.exists(output_path))\nprint(\"Size:\", os.path.getsize(output_path), \"bytes\")\n\nprint(\"\\nPreview:\")\nprint(submission.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-15T21:33:35.401943Z","iopub.execute_input":"2026-09-15T21:33:35.402309Z","iopub.status.idle":"2026-09-15T21:33:37.386674Z","shell.execute_reply.started":"2026-09-15T21:33:35.402253Z","shell.execute_reply":"2026-09-15T21:33:37.385550Z"}},"outputs":[],"execution_count":null}]}