{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":130932,"databundleVersionId":15769099}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-02-21T07:54:59.698470Z","iopub.execute_input":"2026-02-21T07:54:59.698719Z","iopub.status.idle":"2026-02-21T07:55:33.418398Z","shell.execute_reply.started":"2026-02-21T07:54:59.698697Z","shell.execute_reply":"2026-02-21T07:55:33.417740Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport zipfile\nimport shutil\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\n\nprint('Libraries loaded.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T08:03:07.336835Z","iopub.execute_input":"2026-02-21T08:03:07.337474Z","iopub.status.idle":"2026-02-21T08:03:07.341906Z","shell.execute_reply.started":"2026-02-21T08:03:07.337446Z","shell.execute_reply":"2026-02-21T08:03:07.341049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE = Path('/kaggle/input/automatic-lens-correction')\n\n# Find test directory\ntest_candidates = list(BASE.rglob('test')) + list(BASE.glob('*/test'))\nTEST_DIR = None\nfor candidate in test_candidates:\n    if candidate.is_dir() and len(list(candidate.glob('*.jpg'))) > 0:\n        TEST_DIR = candidate\n        break\n\n# Fallback: search for any jpg not in train folder\nif TEST_DIR is None:\n    for d in BASE.iterdir():\n        if d.is_dir() and 'test' in d.name.lower():\n            TEST_DIR = d\n            break\n\nif TEST_DIR is None:\n    # Last resort: check common locations\n    for p in [BASE / 'test', BASE / 'test_images']:\n        if p.exists():\n            TEST_DIR = p\n            break\n\nprint(f'Test dir: {TEST_DIR}')\n\nTRAIN_DIR = BASE / 'lens-correction-train-cleaned'\nprint(f'Train dir exists: {TRAIN_DIR.exists()}')\n\nOUTPUT_DIR = Path('/kaggle/working/corrected')\nZIP_PATH   = Path('/kaggle/working/corrected_images.zip')\nCSV_PATH   = Path('/kaggle/working/submission.csv')\nOUTPUT_DIR.mkdir(parents=True, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T08:03:17.066284Z","iopub.execute_input":"2026-02-21T08:03:17.066625Z","iopub.status.idle":"2026-02-21T08:03:38.276988Z","shell.execute_reply.started":"2026-02-21T08:03:17.066599Z","shell.execute_reply":"2026-02-21T08:03:38.276357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if TEST_DIR and TEST_DIR.exists():\n    test_images = sorted(TEST_DIR.glob('*.jpg'))\nelse:\n    # If no separate test dir, look for test images in base\n    test_images = []\n    for jpg in BASE.rglob('*.jpg'):\n        name = jpg.name.lower()\n        if 'original' not in name and 'generated' not in name and 'train' not in str(jpg):\n            test_images.append(jpg)\n    test_images = sorted(test_images)\n\nprint(f'Found {len(test_images)} test images')\nif test_images:\n    print(f'Example: {test_images[0]}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T08:03:44.272768Z","iopub.execute_input":"2026-02-21T08:03:44.273359Z","iopub.status.idle":"2026-02-21T08:03:44.287584Z","shell.execute_reply.started":"2026-02-21T08:03:44.273332Z","shell.execute_reply":"2026-02-21T08:03:44.286877Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def correct_lens_distortion(img, k1=-0.25, k2=0.05):\n    h, w = img.shape[:2]\n    cx, cy = w / 2.0, h / 2.0\n    f = w\n    K = np.array([[f, 0, cx], [0, f, cy], [0, 0, 1]], dtype=np.float64)\n    dist_coeffs = np.array([k1, k2, 0.0, 0.0, 0.0], dtype=np.float64)\n    new_K, roi = cv2.getOptimalNewCameraMatrix(K, dist_coeffs, (w, h), alpha=0)\n    corrected = cv2.undistort(img, K, dist_coeffs, None, new_K)\n    x, y, rw, rh = roi\n    if rw > 0 and rh > 0 and (rw < w or rh < h):\n        corrected = corrected[y:y+rh, x:x+rw]\n        corrected = cv2.resize(corrected, (w, h), interpolation=cv2.INTER_LANCZOS4)\n    return corrected\n\ndef ssim_score(img1, img2):\n    \"\"\"Quick SSIM between two images (resized to same shape).\"\"\"\n    if img1.shape != img2.shape:\n        img2 = cv2.resize(img2, (img1.shape[1], img1.shape[0]))\n    gray1 = cv2.cvtColor(img1, cv2.COLOR_BGR2GRAY).astype(np.float32)\n    gray2 = cv2.cvtColor(img2, cv2.COLOR_BGR2GRAY).astype(np.float32)\n    mu1, mu2 = gray1.mean(), gray2.mean()\n    s1 = gray1.std(); s2 = gray2.std()\n    cov = ((gray1 - mu1) * (gray2 - mu2)).mean()\n    C1, C2 = 6.5025, 58.5225\n    return ((2*mu1*mu2+C1)*(2*cov+C2)) / ((mu1**2+mu2**2+C1)*(s1**2+s2**2+C2))\n\n# Sample a few training pairs to find best k1\nprint('Estimating best k1 from training pairs...')\noriginals = list(TRAIN_DIR.glob('*_original.jpg'))[:30]  # sample 30 pairs\n\nbest_k1 = -0.25\nbest_score = -1\n\nif len(originals) > 0:\n    for k1_test in [-0.35, -0.30, -0.25, -0.20, -0.15, -0.10]:\n        scores = []\n        for orig_path in originals[:15]:\n            gen_path = Path(str(orig_path).replace('_original.jpg', '_generated.jpg'))\n            if not gen_path.exists():\n                continue\n            orig = cv2.imread(str(orig_path))\n            gen  = cv2.imread(str(gen_path))\n            if orig is None or gen is None:\n                continue\n            corrected = correct_lens_distortion(orig, k1=k1_test)\n            scores.append(ssim_score(corrected, gen))\n        \n        if scores:\n            avg = np.mean(scores)\n            print(f'  k1={k1_test:.2f}  avg SSIM={avg:.4f}')\n            if avg > best_score:\n                best_score = avg\n                best_k1 = k1_test\n\nprint(f'\\nBest k1: {best_k1}  (SSIM={best_score:.4f})')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T08:03:48.086122Z","iopub.execute_input":"2026-02-21T08:03:48.086741Z","iopub.status.idle":"2026-02-21T08:04:04.344120Z","shell.execute_reply.started":"2026-02-21T08:03:48.086711Z","shell.execute_reply":"2026-02-21T08:04:04.343380Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_ids = []\nerrors    = []\n\nfor img_path in tqdm(test_images, desc='Correcting'):\n    image_id = img_path.stem\n    image_ids.append(image_id)\n    try:\n        img = cv2.imread(str(img_path))\n        if img is None:\n            raise ValueError(f'Cannot read: {img_path}')\n        corrected = correct_lens_distortion(img, k1=best_k1)\n        out_path = OUTPUT_DIR / img_path.name\n        cv2.imwrite(str(out_path), corrected, [cv2.IMWRITE_JPEG_QUALITY, 95])\n    except Exception as e:\n        errors.append((image_id, str(e)))\n        shutil.copy(str(img_path), str(OUTPUT_DIR / img_path.name))\n\nprint(f'Done. {len(image_ids)} processed, {len(errors)} errors.')\nif errors:\n    for eid, emsg in errors[:5]:\n        print(f'  ERROR [{eid}]: {emsg}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T08:04:36.076208Z","iopub.execute_input":"2026-02-21T08:04:36.076811Z","iopub.status.idle":"2026-02-21T08:06:49.047357Z","shell.execute_reply.started":"2026-02-21T08:04:36.076782Z","shell.execute_reply":"2026-02-21T08:06:49.046513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corrected_files = list(OUTPUT_DIR.glob('*.jpg'))\nprint(f'Zipping {len(corrected_files)} images...')\n\nwith zipfile.ZipFile(ZIP_PATH, 'w', zipfile.ZIP_DEFLATED) as zf:\n    for fpath in tqdm(corrected_files, desc='Zipping'):\n        zf.write(fpath, arcname=fpath.name)\n\nprint(f'ZIP: {ZIP_PATH}  ({ZIP_PATH.stat().st_size / 1e6:.1f} MB)')\nprint()\nprint('>>> Upload corrected_images.zip to bounty.autohdr.com')\nprint('>>> Download the real submission.csv and submit that to Kaggle.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T08:06:53.771972Z","iopub.execute_input":"2026-02-21T08:06:53.772288Z","iopub.status.idle":"2026-02-21T08:07:15.517501Z","shell.execute_reply.started":"2026-02-21T08:06:53.772260Z","shell.execute_reply":"2026-02-21T08:07:15.516869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Format image_id: strip _g0 suffix if already present, then add _g0\ndef make_submission_id(stem):\n    # The stem is the filename without .jpg\n    # Competition format expects {uuid}_g0\n    # If stem already ends in _g0, _g1 etc — use as-is\n    import re\n    if re.search(r'_g\\d+$', stem):\n        return stem\n    return f'{stem}_g0'\n\ndf = pd.DataFrame({\n    'image_id': [make_submission_id(iid) for iid in image_ids],\n    'score':    [0.5] * len(image_ids)\n})\n\ndf.to_csv(CSV_PATH, index=False)\nprint(f'submission.csv saved: {len(df)} rows')\nprint(df.head(5).to_string(index=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T08:07:19.200832Z","iopub.execute_input":"2026-02-21T08:07:19.201467Z","iopub.status.idle":"2026-02-21T08:07:19.232155Z","shell.execute_reply.started":"2026-02-21T08:07:19.201428Z","shell.execute_reply":"2026-02-21T08:07:19.231511Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"assert CSV_PATH.exists(), 'submission.csv MISSING — Kaggle will reject!'\nassert ZIP_PATH.exists(), 'ZIP MISSING!'\n\nprint('All output files confirmed:')\nprint(f'  submission.csv  — {CSV_PATH.stat().st_size / 1e3:.1f} KB  ({len(df)} rows)')\nprint(f'  corrected_images.zip  — {ZIP_PATH.stat().st_size / 1e6:.1f} MB')\nprint()\nprint('NEXT STEPS:')\nprint('  1. Download corrected_images.zip from Output tab')\nprint('  2. Upload to bounty.autohdr.com → get real submission.csv')\nprint('  3. Submit the REAL submission.csv to Kaggle leaderboard')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-21T08:07:22.782940Z","iopub.execute_input":"2026-02-21T08:07:22.783455Z","iopub.status.idle":"2026-02-21T08:07:22.788776Z","shell.execute_reply.started":"2026-02-21T08:07:22.783427Z","shell.execute_reply":"2026-02-21T08:07:22.787921Z"}},"outputs":[],"execution_count":null}]}