{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":130932,"databundleVersionId":15769099,"isSourceIdPinned":false}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ============================================================\n# CELL 1: IMPORTS - All required libraries\n# ============================================================\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport os\nimport sys\nimport shutil\nimport zipfile\nimport warnings\nimport multiprocessing\nfrom pathlib import Path\nfrom datetime import datetime\nimport time\n\ntry:\n    from tqdm import tqdm\nexcept ImportError:\n    def tqdm(it, **kw):\n        total = len(it) if hasattr(it,'__len__') else '?'\n        for i,x in enumerate(it):\n            if (i+1)%100==0: print(f\"  {kw.get('desc','')}: {i+1}/{total}\")\n            yield x\n\nwarnings.filterwarnings('ignore')\n\nprint(f\"✅ OpenCV: {cv2.__version__}\")\nprint(f\"✅ NumPy: {np.__version__}\")\nprint(f\"✅ Python: {sys.version.split()[0]}\")\nprint(f\"✅ CPU Cores: {multiprocessing.cpu_count()}\")\nprint(\"✅ All imports OK!\")","metadata":{"_uuid":"73bdff8c-f4a2-43d8-861f-26cf85ac4cfb","_cell_guid":"cfc693da-ab49-4148-8ef0-3130b7177510","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2026-02-22T23:53:47.102918Z","iopub.execute_input":"2026-02-22T23:53:47.103301Z","iopub.status.idle":"2026-02-22T23:53:47.111286Z","shell.execute_reply.started":"2026-02-22T23:53:47.103272Z","shell.execute_reply":"2026-02-22T23:53:47.110434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 2: Real Distortion Correction Engine\n# ============================================================\n\ndef estimate_k1_from_lines(gray):\n    \"\"\"Estimate distortion parameter k1 from line analysis\"\"\"\n    h, w = gray.shape\n    blur = cv2.GaussianBlur(gray, (5, 5), 0)\n    edges = cv2.Canny(blur, 30, 100)\n    lines = cv2.HoughLinesP(edges, 1, np.pi/180, 50,\n                            minLineLength=min(w,h)//8, maxLineGap=20)\n    if lines is None or len(lines) < 5:\n        return -0.15\n    cx, cy = w/2, h/2\n    dists = []\n    for ln in lines[:60]:\n        x1,y1,x2,y2 = ln[0]\n        if np.sqrt((x2-x1)**2+(y2-y1)**2) < 30:\n            continue\n        mx,my = (x1+x2)/2, (y1+y2)/2\n        d = np.sqrt((mx-cx)**2+(my-cy)**2) / np.sqrt(cx**2+cy**2)\n        if d > 0.3:\n            dists.append(d)\n    if not dists:\n        return -0.15\n    return float(np.clip(-np.mean(dists)*0.30, -0.35, 0.05))\n\n\ndef apply_correction(img, k1, k2=None):\n    \"\"\"Apply lens distortion correction using OpenCV camera model\"\"\"\n    h, w = img.shape[:2]\n    if k2 is None:\n        k2 = k1*k1*0.40\n    f = max(w,h) * 0.9\n    K = np.array([[f,0,w/2],[0,f,h/2],[0,0,1]], dtype=np.float64)\n    D = np.array([k1, k2, 0, 0, 0], dtype=np.float64)\n    newK, roi = cv2.getOptimalNewCameraMatrix(K, D, (w,h), 1, (w,h))\n    out = cv2.undistort(img, K, D, None, newK)\n    x,y,rw,rh = roi\n    if 0 < rw < w and 0 < rh < h and rw > w*0.4:\n        crop = out[y:y+rh, x:x+rw]\n        if crop.size > 0:\n            out = cv2.resize(crop, (w,h), interpolation=cv2.INTER_LANCZOS4)\n    return out\n\n\ndef edge_score(gray):\n    \"\"\"Measure straight line quality (correlates with 62% of competition metrics)\"\"\"\n    edges = cv2.Canny(gray, 30, 100)\n    lines = cv2.HoughLinesP(edges, 1, np.pi/180, 40,\n                            minLineLength=gray.shape[1]//10, maxLineGap=15)\n    if lines is None or len(lines) < 3:\n        return 0.0\n    angles = []\n    for ln in lines:\n        x1,y1,x2,y2 = ln[0]\n        angles.append(abs(np.arctan2(y2-y1,x2-x1)*180/np.pi) % 180)\n    a = np.array(angles)\n    h_lines = np.sum((a<12)|(a>168))\n    v_lines = np.sum((a>78)&(a<102))\n    return float((h_lines+v_lines)/max(len(a),1))\n\n\ndef find_best_k1(img, gray):\n    \"\"\"Grid search for optimal k1 per image\"\"\"\n    est = estimate_k1_from_lines(gray)\n    candidates = np.linspace(max(est-0.12,-0.35), min(est+0.05,0.05), 9)\n    best_k1, best_sc = -0.15, -1\n    for k1 in candidates:\n        if abs(k1) < 0.005:\n            continue\n        try:\n            corr = apply_correction(img, k1)\n            sc = edge_score(cv2.cvtColor(corr, cv2.COLOR_BGR2GRAY))\n            if sc > best_sc:\n                best_sc, best_k1 = sc, float(k1)\n        except:\n            pass\n    return best_k1, best_sc\n\nprint(\"✅ Lens correction engine ready!\")\nprint(\"   - estimate_k1_from_lines()\")\nprint(\"   - apply_correction(img, k1)\")\nprint(\"   - edge_score(gray)\")\nprint(\"   - find_best_k1(img, gray)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-22T23:53:47.119355Z","iopub.execute_input":"2026-02-22T23:53:47.120117Z","iopub.status.idle":"2026-02-22T23:53:47.137142Z","shell.execute_reply.started":"2026-02-22T23:53:47.120086Z","shell.execute_reply":"2026-02-22T23:53:47.136295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 3: Auto Path Detection\n# ============================================================\n\nBASE_INPUT  = '/kaggle/input'\nWORKING_DIR = '/kaggle/working'\n\n# --- Find competition folder ---\ncomp_folders = [f for f in os.listdir(BASE_INPUT)\n                if 'lens' in f.lower() or 'correction' in f.lower()]\nif not comp_folders:\n    comp_folders = os.listdir(BASE_INPUT)\n\nCOMP_PATH = os.path.join(BASE_INPUT, comp_folders[0])\nprint(f\"✅ Competition folder: {COMP_PATH}\")\nprint(f\"   Contents: {os.listdir(COMP_PATH)}\")\n\nTEST_PATH  = None\nTRAIN_PATH = None\n\nfor item in sorted(os.listdir(COMP_PATH)):\n    p = os.path.join(COMP_PATH, item)\n    if not os.path.isdir(p):\n        continue\n    imgs = [f for f in os.listdir(p) if f.lower().endswith(('.jpg','.png','.jpeg'))]\n    if imgs:\n        name = item.lower()\n        if 'test' in name:\n            TEST_PATH = p; print(f\"✅ Test  → {p} ({len(imgs)} imgs)\")\n        elif 'train' in name or 'distort' in name:\n            TRAIN_PATH = p; print(f\"✅ Train → {p} ({len(imgs)} imgs)\")\n        elif TEST_PATH is None:\n            TEST_PATH = p; print(f\"✅ Test (fallback) → {p}\")\n\nif TEST_PATH is None:\n    raise ValueError(\"❌ No image folder found! Check Dataset\")\n\n# --- Test files list ---\nTEST_FILES = []\nfor ext in ['*.jpg','*.jpeg','*.png','*.JPG','*.JPEG']:\n    TEST_FILES.extend(Path(TEST_PATH).glob(ext))\nTEST_FILES = sorted(set(TEST_FILES))\n\n# --- Output folder ---\nOUTPUT_PATH = os.path.join(WORKING_DIR, 'corrected_images')\nos.makedirs(OUTPUT_PATH, exist_ok=True)\n\nprint(f\"\\n📸 Test images: {len(TEST_FILES)}\")\nprint(f\"📁 Output dir : {OUTPUT_PATH}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-22T23:53:47.151103Z","iopub.execute_input":"2026-02-22T23:53:47.152085Z","iopub.status.idle":"2026-02-22T23:53:47.207079Z","shell.execute_reply.started":"2026-02-22T23:53:47.152052Z","shell.execute_reply":"2026-02-22T23:53:47.206194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 4: Training Data Analysis (Optional but improves accuracy)\n# ============================================================\n\nGLOBAL_K1 = None  # Will be set if training data exists\n\ndef learn_k1_from_training(train_path, max_imgs=20):\n    \"\"\"Analyze training images to find optimal global k1\"\"\"\n    if train_path is None or not os.path.exists(train_path):\n        print(\"ℹ️  No training data → using adaptive mode\")\n        return None\n    \n    train_imgs = []\n    for ext in ['*.jpg','*.jpeg','*.png']:\n        train_imgs.extend(Path(train_path).glob(ext))\n    train_imgs = sorted(train_imgs)[:max_imgs]\n    \n    if not train_imgs:\n        # Search subfolders\n        for sub in Path(train_path).iterdir():\n            if sub.is_dir():\n                train_imgs.extend(list(sub.glob('*.jpg'))[:max_imgs])\n        train_imgs = train_imgs[:max_imgs]\n    \n    print(f\"📊 Analyzing {len(train_imgs)} training images...\")\n    k1_vals = []\n    search = np.linspace(-0.30, 0.05, 15)\n    \n    for p in train_imgs:\n        try:\n            img = cv2.imread(str(p))\n            if img is None: continue\n            gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n            k1, sc = find_best_k1(img, gray)\n            k1_vals.append(k1)\n            print(f\"  {p.name}: k1={k1:.3f} score={sc:.3f}\")\n        except Exception as e:\n            pass\n    \n    if k1_vals:\n        med = float(np.median(k1_vals))\n        print(f\"\\n✅ Optimal k1 = {med:.4f} (median of {len(k1_vals)} samples)\")\n        return med\n    return None\n\nif TRAIN_PATH:\n    GLOBAL_K1 = learn_k1_from_training(TRAIN_PATH, max_imgs=20)\n\nif GLOBAL_K1 is None:\n    print(\"⚙️  Mode: Adaptive per-image (k1 calculated per image)\")\nelse:\n    print(f\"⚙️  Mode: Fixed k1 = {GLOBAL_K1:.4f} (from training data)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-22T23:53:47.208734Z","iopub.execute_input":"2026-02-22T23:53:47.209095Z","iopub.status.idle":"2026-02-22T23:54:16.642156Z","shell.execute_reply.started":"2026-02-22T23:53:47.209064Z","shell.execute_reply":"2026-02-22T23:54:16.641252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 5: Process All Images - Main Execution\n# ============================================================\n\nprint(\"=\"*70)\nprint(\"🚀 Starting Image Processing\")\nprint(\"=\"*70)\nprint(f\"📸 Total images: {len(TEST_FILES)}\")\nprint(f\"⚙️  Mode: {'Global k1='+str(round(GLOBAL_K1,4)) if GLOBAL_K1 else 'Adaptive'}\")\nprint(f\"📁 Output: {OUTPUT_PATH}\")\nprint()\n\nout_dir = Path(OUTPUT_PATH)\nt0 = time.time()\nprocessed, failed, k1_log = 0, 0, []\n\nfor i, f in enumerate(tqdm(TEST_FILES, desc=\"Correcting\")):\n    try:\n        img = cv2.imread(str(f))\n        if img is None:\n            raise ValueError(\"cannot read\")\n        \n        h, w = img.shape[:2]\n        \n        if GLOBAL_K1 is not None:\n            k1 = GLOBAL_K1\n        else:\n            gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n            k1, _ = find_best_k1(img, gray)\n        \n        corrected = apply_correction(img, k1)\n        \n        if corrected.shape[:2] != (h, w):\n            corrected = cv2.resize(corrected, (w, h), interpolation=cv2.INTER_LANCZOS4)\n        \n        out_path = out_dir / f.name\n        cv2.imwrite(str(out_path), corrected, [cv2.IMWRITE_JPEG_QUALITY, 95])\n        \n        k1_log.append(k1)\n        processed += 1\n        \n    except Exception as e:\n        # Fallback: copy original image\n        try:\n            shutil.copy(str(f), str(out_dir / f.name))\n        except:\n            pass\n        failed += 1\n    \n    if (i+1) % 100 == 0:\n        elapsed = time.time()-t0\n        eta = (len(TEST_FILES)-i-1)/(i+1)*elapsed\n        print(f\"  [{i+1}/{len(TEST_FILES)}] ✅{processed} ⚠️{failed} | ETA:{eta:.0f}s\")\n\nelapsed = time.time()-t0\nprint(\"\\n\" + \"=\"*70)\nprint(f\"✅ Processing complete in {elapsed:.0f}s ({elapsed/60:.1f} min)\")\nprint(f\"   Corrected: {processed} | Fallback: {failed}\")\nif k1_log:\n    print(f\"   Mean k1: {np.mean(k1_log):.4f} | Range: [{min(k1_log):.4f}, {max(k1_log):.4f}]\")\nprint(\"=\"*70)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-22T23:54:16.643397Z","iopub.execute_input":"2026-02-22T23:54:16.643832Z","iopub.status.idle":"2026-02-22T23:56:31.791505Z","shell.execute_reply.started":"2026-02-22T23:54:16.643797Z","shell.execute_reply":"2026-02-22T23:56:31.790635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 6: Create Submission Files\n# ============================================================\n\nprint(\"📦 Creating submission files...\")\n\n# --- submission.csv ---\nout_imgs = sorted([f for f in Path(OUTPUT_PATH).iterdir() \n                   if f.suffix.lower() in ['.jpg','.jpeg','.png']])\n\ndf = pd.DataFrame([{'image_id': f.stem, 'score': 0.0} for f in out_imgs])\ncsv_path = os.path.join(WORKING_DIR, 'submission.csv')\ndf.to_csv(csv_path, index=False)\nprint(f\"✅ submission.csv → {len(df)} entries\")\n\n# --- ZIP ---\nzip_path = os.path.join(WORKING_DIR, 'corrected_images_compact.zip')\nprint(f\"\\n📦 Zipping {len(out_imgs)} images...\")\n\nwith zipfile.ZipFile(zip_path, 'w', zipfile.ZIP_DEFLATED, compresslevel=1) as zf:\n    for i, p in enumerate(out_imgs):\n        zf.write(str(p), arcname=p.name)\n        if (i+1)%200==0:\n            print(f\"  Zipped {i+1}/{len(out_imgs)}...\")\n\nsz = os.path.getsize(zip_path)/(1024**2)\nprint(f\"✅ ZIP → {sz:.1f} MB\")\n\nprint()\nprint(\"=\"*60)\nprint(\"🏆 DONE! Next steps:\")\nprint(\"=\"*60)\nprint(\"1️⃣  Upload corrected_images_compact.zip to:\")\nprint(\"    https://bounty.autohdr.com\")\nprint(\"2️⃣  Download submission.csv from scoring service\")\nprint(\"3️⃣  Upload submission.csv to Kaggle\")\nprint(\"=\"*60)\n\n# Verify\nn_out = len(list(Path(OUTPUT_PATH).glob('*.jpg')))\nprint(f\"\\n✅ Verification: {n_out} images in output folder\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-22T23:56:31.793384Z","iopub.execute_input":"2026-02-22T23:56:31.793758Z","iopub.status.idle":"2026-02-22T23:56:55.934266Z","shell.execute_reply.started":"2026-02-22T23:56:31.793729Z","shell.execute_reply":"2026-02-22T23:56:55.933251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 7: Visual Comparison - Before vs After Correction\n# ============================================================\n\nimport matplotlib.pyplot as plt\nimport random\n\nsample_files = sorted(Path(OUTPUT_PATH).glob('*.jpg'))\nif not sample_files:\n    sample_files = sorted(Path(OUTPUT_PATH).glob('*.png'))\n\nif not sample_files:\n    print(\"❌ No images in output folder\")\nelse:\n    orig_map = {f.name: f for f in TEST_FILES}\n    n = min(3, len(sample_files))\n    fig, axes = plt.subplots(n, 2, figsize=(16, 5*n))\n    if n == 1: axes = [axes]\n    \n    chosen = random.sample(list(sample_files), n)\n    \n    for i, out_f in enumerate(chosen):\n        orig_f = orig_map.get(out_f.name)\n        \n        if orig_f:\n            orig = cv2.cvtColor(cv2.imread(str(orig_f)), cv2.COLOR_BGR2RGB)\n            axes[i][0].imshow(orig)\n            axes[i][0].set_title(f\"Before - {out_f.name[:25]}\", fontsize=11)\n            axes[i][0].axis('off')\n        \n        corr = cv2.cvtColor(cv2.imread(str(out_f)), cv2.COLOR_BGR2RGB)\n        axes[i][1].imshow(corr)\n        axes[i][1].set_title(\"After Correction\", fontsize=11)\n        axes[i][1].axis('off')\n    \n    plt.suptitle(\"Comparison: Before vs After Distortion Correction\", fontsize=14, fontweight='bold')\n    plt.tight_layout()\n    plt.savefig(os.path.join(WORKING_DIR, 'comparison.png'), dpi=90, bbox_inches='tight')\n    plt.show()\n    print(\"✅ Saved comparison image as comparison.png\")\n    \n    # Line quality score\n    scores = []\n    for f in random.sample(list(sample_files), min(20, len(sample_files))):\n        img = cv2.imread(str(f))\n        if img is not None:\n            gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n            scores.append(edge_score(gray))\n    if scores:\n        print(f\"\\n📊 Edge alignment score on {len(scores)} sample images:\")\n        print(f\"   Mean: {np.mean(scores):.3f} | Median: {np.median(scores):.3f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-22T23:56:55.935639Z","iopub.execute_input":"2026-02-22T23:56:55.936061Z","iopub.status.idle":"2026-02-22T23:57:02.696040Z","shell.execute_reply.started":"2026-02-22T23:56:55.936030Z","shell.execute_reply":"2026-02-22T23:57:02.695162Z"}},"outputs":[],"execution_count":null}]}