{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":35332,"databundleVersionId":3723648},{"sourceType":"datasetVersion","sourceId":3739819,"datasetId":2231132,"databundleVersionId":3794269},{"sourceType":"datasetVersion","sourceId":16432073,"datasetId":10532756,"databundleVersionId":17429733},{"sourceType":"kernelVersion","sourceId":97616956}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"93b8a472","cell_type":"code","source":"\"\"\"\nNotebook 00 — Data Preparation\n=================================\n\nPurpose:\n    1. Load the AmEx data efficiently (integer-typed Parquet for memory)\n    2. Audit feature types (cardinality check for hidden categoricals)\n    3. Compute customer-level targets (one row per customer)\n    4. Generate shared CV fold assignments (src/folds.py)\n    5. Cache cleaned, typed Parquet files for downstream notebooks\n\nOutputs:\n    outputs/folds/fold_assignments.parquet     (customer_ID, target, fold_id)\n    data/processed/train_customers.parquet     (customer_ID, target)\n    data/processed/feature_metadata.parquet    (feature_name, dtype, n_unique, n_null, is_categorical)\n    data/processed/train_clean.parquet         (cleaned statement-level data, optional)\n\nCompute:\n    CPU only. ~20 minutes wall time on Kaggle.\n\"\"\"","metadata":{"lines_to_next_cell":0,"trusted":true,"execution":{"iopub.status.busy":"2026-05-24T13:17:02.531194Z","iopub.execute_input":"2026-05-24T13:17:02.531435Z","iopub.status.idle":"2026-05-24T13:17:02.536566Z","shell.execute_reply.started":"2026-05-24T13:17:02.531414Z","shell.execute_reply":"2026-05-24T13:17:02.535880Z"}},"outputs":[],"execution_count":null},{"id":"9c047a91","cell_type":"markdown","source":"# Notebook 00 — Data Preparation\n\nThis notebook handles the one-time setup for the AmEx 4-model project:\nloading data, auditing feature types, computing per-customer targets, and\ngenerating the shared CV folds that every base-model notebook will use.\n\n**Run this notebook once at the start of the project.** Its outputs are\ncached and re-used by all downstream notebooks.","metadata":{}},{"id":"16bf779f","cell_type":"markdown","source":"## 1. Setup","metadata":{}},{"id":"4a75b5c5","cell_type":"code","source":"import os\nimport sys\nfrom pathlib import Path\n\n# Auto-discover src/ regardless of dataset name\nsrc_dir = None\nfor p in Path(\"/kaggle/input\").rglob(\"src/config.py\"):\n    src_dir = p.parent\n    break\nif src_dir is None:\n    print(\"Available under /kaggle/input:\")\n    for p in sorted(Path(\"/kaggle/input\").iterdir()):\n        print(f\"  {p}\")\n    raise RuntimeError(\"Could not find src/config.py anywhere under /kaggle/input\")\n\nprint(f\"Found src at: {src_dir}\")\nsys.path.insert(0, str(src_dir))\n\n# Change working directory so config.yaml is findable\nos.chdir(str(src_dir.parent))\nprint(f\"Working dir: {os.getcwd()}\")\n\nimport numpy as np\nimport pandas as pd\nfrom tqdm.auto import tqdm\n\nimport config as cfg_module\nimport folds as folds_module\nfrom metric import amex_metric_components\n\ncfg = cfg_module.load_config()\nprint(f\"Config loaded. Master seed = {cfg['master_seed']}, n_folds = {cfg['n_folds']}\")\nprint(f\"Kaggle environment: {cfg_module.is_kaggle()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-24T13:17:02.537724Z","iopub.execute_input":"2026-05-24T13:17:02.538014Z","iopub.status.idle":"2026-05-24T13:17:02.559998Z","shell.execute_reply.started":"2026-05-24T13:17:02.537963Z","shell.execute_reply":"2026-05-24T13:17:02.559133Z"}},"outputs":[],"execution_count":null},{"id":"d33e88b7-d0c7-43ad-9a95-51f02d0d1a09","cell_type":"code","source":"# Override output paths for Kaggle's writable working directory\nWORKING = Path(\"/kaggle/working\")\ncfg[\"paths\"][\"folds\"]      = str(WORKING / \"outputs\" / \"folds\")\ncfg[\"paths\"][\"oof\"]        = str(WORKING / \"outputs\" / \"oof\")\ncfg[\"paths\"][\"test_preds\"] = str(WORKING / \"outputs\" / \"test_preds\")\ncfg[\"paths\"][\"models\"]     = str(WORKING / \"outputs\" / \"models\")\ncfg[\"paths\"][\"reports\"]    = str(WORKING / \"outputs\" / \"reports\")\ncfg[\"paths\"][\"processed_data\"] = str(WORKING / \"data\" / \"processed\")\n\n# Make sure these exist\nfor k in [\"folds\", \"oof\", \"test_preds\", \"models\", \"reports\", \"processed_data\"]:\n    Path(cfg[\"paths\"][k]).mkdir(parents=True, exist_ok=True)\n\nprint(\"Output paths redirected to /kaggle/working/\")\nfor k in [\"folds\", \"oof\", \"test_preds\", \"models\", \"reports\"]:\n    print(f\"  {k}: {cfg['paths'][k]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-24T13:17:02.560737Z","iopub.execute_input":"2026-05-24T13:17:02.560941Z","iopub.status.idle":"2026-05-24T13:17:02.568233Z","shell.execute_reply.started":"2026-05-24T13:17:02.560914Z","shell.execute_reply":"2026-05-24T13:17:02.567378Z"}},"outputs":[],"execution_count":null},{"id":"54d63ee1","cell_type":"markdown","source":"## 2. Locate Source Data\n\nTwo Kaggle datasets are expected:\n- `amex-default-prediction`: original CSVs (large)\n- `amex-data-integer-dtypes-parquet-format`: community-shared integer Parquet\n  (much smaller, faster to load)\n\nLocally, set `RAW_DATA_DIR` to wherever you've placed the files.","metadata":{}},{"id":"069cbf99","cell_type":"code","source":"if cfg_module.is_kaggle():\n    # Find AmEx competition data (has train_labels.csv)\n    RAW_DIR = None\n    for p in Path(\"/kaggle/input\").rglob(\"train_labels.csv\"):\n        RAW_DIR = p.parent\n        break\n    if RAW_DIR is None:\n        raise RuntimeError(\"Could not find AmEx competition data (train_labels.csv)\")\n    print(f\"RAW_DIR: {RAW_DIR}\")\n\n    # Find integer-parquet data (preferred) or fall back to raw CSV directory\n    PARQUET_DIR = None\n    for p in Path(\"/kaggle/input\").rglob(\"train.parquet\"):\n        PARQUET_DIR = p.parent\n        break\n    if PARQUET_DIR is None:\n        print(\"WARNING: No train.parquet found — falling back to raw CSV\")\n        print(\"This will use much more memory. Add the raddar parquet dataset.\")\n        PARQUET_DIR = RAW_DIR\n    print(f\"PARQUET_DIR: {PARQUET_DIR}\")\n\n    OUTPUT_DATA_DIR = Path(\"/kaggle/working/data/processed\")\nelse:\n    RAW_DIR = Path(cfg[\"paths\"][\"raw_data\"])\n    PARQUET_DIR = RAW_DIR\n    OUTPUT_DATA_DIR = Path(cfg[\"paths\"][\"processed_data\"])\n\nOUTPUT_DATA_DIR.mkdir(parents=True, exist_ok=True)\nprint(f\"Output dir: {OUTPUT_DATA_DIR}\")\n\nOUTPUT_DATA_DIR.mkdir(parents=True, exist_ok=True)\nprint(f\"Parquet dir: {PARQUET_DIR}\")\nprint(f\"Raw dir:     {RAW_DIR}\")\nprint(f\"Output dir:  {OUTPUT_DATA_DIR}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-24T13:17:02.569799Z","iopub.execute_input":"2026-05-24T13:17:02.570065Z","iopub.status.idle":"2026-05-24T13:17:02.594021Z","shell.execute_reply.started":"2026-05-24T13:17:02.570039Z","shell.execute_reply":"2026-05-24T13:17:02.593310Z"}},"outputs":[],"execution_count":null},{"id":"2f437682-b7b4-4edd-ab29-fe2c9040e50e","cell_type":"code","source":"from pathlib import Path\n\n# Show everything actually available\nprint(\"=\" * 60)\nprint(\"Contents of /kaggle/input/\")\nprint(\"=\" * 60)\nfor p in sorted(Path(\"/kaggle/input\").iterdir()):\n    print(f\"\\n{p}/\")\n    try:\n        for child in sorted(p.iterdir()):\n            if child.is_dir():\n                print(f\"  {child.name}/\")\n                # Peek inside one level\n                try:\n                    for grandchild in sorted(child.iterdir())[:10]:\n                        print(f\"    {grandchild.name}\")\n                except PermissionError:\n                    print(\"    (permission denied)\")\n            else:\n                print(f\"  {child.name}  ({child.stat().st_size / 1e6:.1f} MB)\")\n    except PermissionError:\n        print(\"  (permission denied)\")\n\n# Find labels and features manually\nimport pandas as pd\nfrom pathlib import Path\n\n# Search for the labels file anywhere\nfor p in Path(\"/kaggle/input\").rglob(\"train_labels*\"):\n    print(f\"Found labels: {p}\")\nfor p in Path(\"/kaggle/input\").rglob(\"train.parquet\"):\n    print(f\"Found train parquet: {p}\")\nfor p in Path(\"/kaggle/input\").rglob(\"test.parquet\"):\n    print(f\"Found test parquet: {p}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-24T13:17:02.594855Z","iopub.execute_input":"2026-05-24T13:17:02.595114Z","iopub.status.idle":"2026-05-24T13:17:02.628014Z","shell.execute_reply.started":"2026-05-24T13:17:02.595081Z","shell.execute_reply":"2026-05-24T13:17:02.627396Z"}},"outputs":[],"execution_count":null},{"id":"be17243e","cell_type":"markdown","source":"## 3. Load Train Labels\n\n`train_labels.csv` has one row per customer with the binary target.","metadata":{}},{"id":"3fa2a22d","cell_type":"code","source":"labels_path = RAW_DIR / \"train_labels.csv\"\nif labels_path.exists():\n    labels = pd.read_csv(labels_path)\nelse:\n    # Some integer-parquet datasets ship labels as parquet\n    labels = pd.read_parquet(PARQUET_DIR / \"train_labels.parquet\")\n\nlabels[\"target\"] = labels[\"target\"].astype(np.int8)\nprint(f\"Labels shape: {labels.shape}\")\nprint(f\"Positive rate: {labels['target'].mean():.4f}  \"\n      f\"(expected ~0.25 after AmEx's 5% negative undersampling)\")\nprint(labels.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-24T13:17:02.629122Z","iopub.execute_input":"2026-05-24T13:17:02.629380Z","iopub.status.idle":"2026-05-24T13:17:03.055633Z","shell.execute_reply.started":"2026-05-24T13:17:02.629352Z","shell.execute_reply":"2026-05-24T13:17:03.054894Z"}},"outputs":[],"execution_count":null},{"id":"1aea67f0","cell_type":"markdown","source":"## 4. Load Statement-Level Features\n\nThis is the big load. Integer-typed Parquet is ~5GB on disk and ~10GB in\nmemory, which fits within Kaggle's 30GB kernel RAM.","metadata":{}},{"id":"8caa7e86","cell_type":"code","source":"# Try integer-parquet first; fall back to CSV if not available\ntrain_parquet = PARQUET_DIR / \"train.parquet\"\nif not train_parquet.exists():\n    # Some datasets name it differently\n    candidates = list(PARQUET_DIR.glob(\"*train*.parquet\"))\n    if candidates:\n        train_parquet = candidates[0]\n    else:\n        train_parquet = None\n\nif train_parquet and train_parquet.exists():\n    print(f\"Loading integer-typed parquet from {train_parquet}\")\n    train = pd.read_parquet(train_parquet)\nelse:\n    print(\"Integer parquet not found; loading CSV (slow, large memory)\")\n    train = pd.read_csv(RAW_DIR / \"train_data.csv\")\n\nprint(f\"Train shape: {train.shape}\")\nprint(f\"Memory usage: {train.memory_usage(deep=True).sum() / 1e9:.2f} GB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-24T13:17:03.056461Z","iopub.execute_input":"2026-05-24T13:17:03.056687Z","iopub.status.idle":"2026-05-24T13:17:08.570847Z","shell.execute_reply.started":"2026-05-24T13:17:03.056658Z","shell.execute_reply":"2026-05-24T13:17:08.570354Z"}},"outputs":[],"execution_count":null},{"id":"ede97021","cell_type":"markdown","source":"## 5. Feature Metadata and Cardinality Audit\n\nPer Gemini's feedback: identify hidden categoricals (low-cardinality features\nnot in the official categorical list). Anything with ≤50 unique values is a\ncategorical candidate.","metadata":{}},{"id":"97303ff4","cell_type":"code","source":"id_col = cfg[\"data\"][\"id_col\"]\ndate_col = cfg[\"data\"][\"date_col\"]\ntarget_col = cfg[\"data\"][\"target_col\"]\nofficial_categoricals = set(cfg[\"data\"][\"categorical_features\"])\nsuspected = set(cfg[\"data\"][\"suspected_categorical\"])\nthreshold = int(cfg[\"data\"][\"cardinality_threshold\"])\n\nfeature_cols = [c for c in train.columns if c not in {id_col, date_col, target_col}]\nprint(f\"{len(feature_cols)} feature columns\")\n\nmetadata_rows = []\nfor col in tqdm(feature_cols, desc=\"Cardinality audit\"):\n    s = train[col]\n    n_null = int(s.isna().sum())\n    # nunique excludes NaN by default\n    n_unique = int(s.nunique())\n    dtype = str(s.dtype)\n    is_official_cat = col in official_categoricals\n    is_suspected = col in suspected\n    is_low_card = n_unique <= threshold\n    metadata_rows.append({\n        \"feature\": col,\n        \"dtype\": dtype,\n        \"n_unique\": n_unique,\n        \"n_null\": n_null,\n        \"null_fraction\": n_null / len(train),\n        \"is_official_categorical\": is_official_cat,\n        \"is_suspected_categorical\": is_suspected,\n        \"is_low_cardinality\": is_low_card,\n        \"treat_as_categorical\": is_official_cat or is_low_card,\n    })\n\nmetadata = pd.DataFrame(metadata_rows).sort_values(\"n_unique\")\nmetadata_path = OUTPUT_DATA_DIR / \"feature_metadata.parquet\"\nmetadata.to_parquet(metadata_path, index=False)\nprint(f\"\\nSaved {metadata_path}\")\nprint(\"\\nLow-cardinality features (≤50 unique values):\")\nprint(metadata[metadata[\"is_low_cardinality\"]]\n      [[\"feature\", \"n_unique\", \"is_official_categorical\"]]\n      .head(20).to_string(index=False))\n\n# Summary counts\nprint(f\"\\nOfficial categoricals:     {metadata['is_official_categorical'].sum()}\")\nprint(f\"Suspected categoricals:    {metadata['is_suspected_categorical'].sum()}\")\nprint(f\"Low-cardinality features:  {metadata['is_low_cardinality'].sum()}\")\nprint(f\"Treated as categorical:    {metadata['treat_as_categorical'].sum()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-24T13:17:08.572480Z","iopub.execute_input":"2026-05-24T13:17:08.572700Z","iopub.status.idle":"2026-05-24T13:17:55.698750Z","shell.execute_reply.started":"2026-05-24T13:17:08.572659Z","shell.execute_reply":"2026-05-24T13:17:55.698230Z"}},"outputs":[],"execution_count":null},{"id":"5fb6b56a","cell_type":"markdown","source":"## 6. Dtype Enforcement\n\nPer Gemini's feedback: aggressive dtype enforcement to avoid OOM during\nthe aggregation step in Model 2. Strategy:\n- Categoricals → `int8` or `int16` depending on cardinality\n- Numerics → `float32`\n- IDs → stay as object/str\n- Date → `datetime64`","metadata":{}},{"id":"590c7c25","cell_type":"code","source":"# The integer-parquet dataset usually has appropriate dtypes already.\n# Defensively re-cast anything that came in as float64.\nbefore_mem = train.memory_usage(deep=True).sum() / 1e9\n\nfor col in feature_cols:\n    if train[col].dtype == np.float64:\n        train[col] = train[col].astype(np.float32)\n    elif train[col].dtype == np.int64:\n        max_val = train[col].abs().max()\n        if max_val < 128:\n            train[col] = train[col].astype(np.int8)\n        elif max_val < 32768:\n            train[col] = train[col].astype(np.int16)\n        else:\n            train[col] = train[col].astype(np.int32)\n\nif date_col in train.columns and not np.issubdtype(train[date_col].dtype, np.datetime64):\n    train[date_col] = pd.to_datetime(train[date_col])\n\nafter_mem = train.memory_usage(deep=True).sum() / 1e9\nprint(f\"Memory: {before_mem:.2f} GB → {after_mem:.2f} GB \"\n      f\"({100 * (before_mem - after_mem) / before_mem:.1f}% reduction)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-24T13:17:55.699433Z","iopub.execute_input":"2026-05-24T13:17:55.699606Z","iopub.status.idle":"2026-05-24T13:17:57.907331Z","shell.execute_reply.started":"2026-05-24T13:17:55.699588Z","shell.execute_reply":"2026-05-24T13:17:57.906409Z"}},"outputs":[],"execution_count":null},{"id":"0ed31178","cell_type":"markdown","source":"## 7. Customer-Level Table and Statement Counts\n\nWe need a per-customer table for fold assignment and a `n_statements`\nfeature for the meta-learner.","metadata":{}},{"id":"ec68673b","cell_type":"code","source":"n_statements = (train.groupby(id_col).size()\n                .rename(\"n_statements\")\n                .reset_index())\ncustomers = labels.merge(n_statements, on=id_col, how=\"left\")\ncustomers[\"n_statements\"] = customers[\"n_statements\"].fillna(0).astype(np.int8)\nprint(f\"Customers: {len(customers)}\")\nprint(f\"Statement count distribution:\")\nprint(customers[\"n_statements\"].value_counts().sort_index())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-24T13:17:57.908302Z","iopub.execute_input":"2026-05-24T13:17:57.908522Z","iopub.status.idle":"2026-05-24T13:17:58.753461Z","shell.execute_reply.started":"2026-05-24T13:17:57.908500Z","shell.execute_reply":"2026-05-24T13:17:58.752788Z"}},"outputs":[],"execution_count":null},{"id":"066b7e21","cell_type":"markdown","source":"## 8. Generate Shared CV Folds\n\nThis is the critical step. The fold assignment is computed once with the\nmaster seed from config.yaml and serialized. Every base-model notebook\nwill load this same file.","metadata":{}},{"id":"1572bc1b","cell_type":"code","source":"folds_df = folds_module.get_folds(\n    customer_ids=customers[id_col].values,\n    targets=customers[\"target\"].values,\n)\nprint(\"Folds generated. Summary:\")\nprint(folds_module.summary(folds_df).to_string(index=False))\n\n# Sanity-check stratification\nglobal_rate = customers[\"target\"].mean()\nprint(f\"\\nGlobal positive rate: {global_rate:.4f}\")\nprint(f\"Max per-fold deviation: \"\n      f\"{(folds_module.summary(folds_df)['positive_rate'] - global_rate).abs().max():.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-24T13:17:58.754181Z","iopub.execute_input":"2026-05-24T13:17:58.754587Z","iopub.status.idle":"2026-05-24T13:17:59.358730Z","shell.execute_reply.started":"2026-05-24T13:17:58.754523Z","shell.execute_reply":"2026-05-24T13:17:59.358043Z"}},"outputs":[],"execution_count":null},{"id":"9adf22c5","cell_type":"markdown","source":"## 9. Save Customer-Level Outputs","metadata":{}},{"id":"297b2ebd","cell_type":"code","source":"# Customer-level table with target, n_statements, fold_id\ncustomers = customers.merge(folds_df[[\"customer_ID\", \"fold_id\"]], on=id_col)\ncustomers_path = OUTPUT_DATA_DIR / \"train_customers.parquet\"\ncustomers.to_parquet(customers_path, index=False)\nprint(f\"Saved {customers_path}  ({len(customers)} rows)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-24T13:17:59.359457Z","iopub.execute_input":"2026-05-24T13:17:59.359643Z","iopub.status.idle":"2026-05-24T13:17:59.618100Z","shell.execute_reply.started":"2026-05-24T13:17:59.359623Z","shell.execute_reply":"2026-05-24T13:17:59.617297Z"}},"outputs":[],"execution_count":null},{"id":"5541d60d","cell_type":"markdown","source":"## 10. Optional: Save Cleaned Statement-Level Data\n\nSkip this if you're tight on disk space — downstream notebooks can read the\noriginal integer-parquet directly. Saving here gives a single canonical file\nwith our dtype enforcement applied.","metadata":{}},{"id":"c5b73064","cell_type":"code","source":"SAVE_STATEMENT_DATA = False  # set True if you want a canonical copy\nif SAVE_STATEMENT_DATA:\n    statements_path = OUTPUT_DATA_DIR / \"train_clean.parquet\"\n    train.to_parquet(statements_path, index=False)\n    print(f\"Saved {statements_path}  ({len(train)} rows)\")\nelse:\n    print(\"Skipping statement-level save (downstream notebooks can read original parquet)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-24T13:17:59.618880Z","iopub.execute_input":"2026-05-24T13:17:59.619135Z","iopub.status.idle":"2026-05-24T13:17:59.624363Z","shell.execute_reply.started":"2026-05-24T13:17:59.619108Z","shell.execute_reply":"2026-05-24T13:17:59.623479Z"}},"outputs":[],"execution_count":null},{"id":"89938d21","cell_type":"markdown","source":"## 11. Smoke Test the Metric\n\nQuick sanity check that the metric is wired correctly.","metadata":{}},{"id":"1fbf79de","cell_type":"code","source":"rng = np.random.default_rng(cfg[\"master_seed\"])\nfake_pred = rng.random(len(customers))\nm_random = amex_metric_components(\n    customers[\"target\"].values, fake_pred, customers[id_col].values\n)\nprint(f\"Random predictions → M = {m_random['M']:.4f}  \"\n      f\"(should be near 0; G = {m_random['G']:.4f}, D = {m_random['D']:.4f})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-24T13:17:59.625849Z","iopub.execute_input":"2026-05-24T13:17:59.626146Z","iopub.status.idle":"2026-05-24T13:17:59.791246Z","shell.execute_reply.started":"2026-05-24T13:17:59.626118Z","shell.execute_reply":"2026-05-24T13:17:59.790414Z"}},"outputs":[],"execution_count":null},{"id":"cd16cdde","cell_type":"markdown","source":"## 12. Summary\n\nOutputs from this notebook:\n- `outputs/folds/fold_assignments.parquet` — shared CV folds for all models\n- `data/processed/train_customers.parquet` — customer-level table\n- `data/processed/feature_metadata.parquet` — feature audit\n\nNext: notebooks 01–04 build the four base models, each consuming these\noutputs.","metadata":{}},{"id":"69c4a0d3","cell_type":"code","source":"print(\"Notebook 00 complete.\")\nprint(f\"Outputs in: {OUTPUT_DATA_DIR}\")\nprint(f\"Folds in:   {cfg_module.resolve_path('folds')}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-24T13:17:59.792006Z","iopub.execute_input":"2026-05-24T13:17:59.792236Z","iopub.status.idle":"2026-05-24T13:17:59.796581Z","shell.execute_reply.started":"2026-05-24T13:17:59.792209Z","shell.execute_reply":"2026-05-24T13:17:59.795750Z"}},"outputs":[],"execution_count":null}]}