{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":19231,"databundleVersionId":1413778,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport shutil\nfrom pathlib import Path\nfrom sklearn.model_selection import train_test_split\nimport yaml\nimport numpy as np\n\n# Set random seeds for reproducibility\nnp.random.seed(42)\nimport random\nrandom.seed(42)\nimport torch\ntorch.manual_seed(42)\n\nmapping_df = pd.read_csv(\"/kaggle/input/landmark-recognition-2020/train.csv\")\ns = mapping_df.landmark_id\ncounts = s.value_counts()\ntop_10_landmark_ids = counts.head(10).index.tolist() \ntop_10_images_list = []\n\nfor landmark_id in top_10_landmark_ids:\n    class_images = mapping_df[mapping_df['landmark_id'] == landmark_id]\n    n_samples = min(900, len(class_images))\n    sampled_images = class_images.sample(n=n_samples, random_state=42)\n    top_10_images_list.append(sampled_images)\n\ntop_10_images_df = pd.concat(top_10_images_list, ignore_index=True)\n\ndef find_image_files_optimized(image_names, train_folder_path):\n    found_images = []\n    missing_images = []\n    \n    for i, image_name in enumerate(image_names):\n        image_name = str(image_name)\n        \n        if len(image_name) < 3:\n            missing_images.append(image_name)\n            continue\n            \n        first_char = image_name[0]\n        second_char = image_name[1]\n        third_char = image_name[2]\n        \n        folder_path = os.path.join(train_folder_path, first_char, second_char, third_char)\n        \n        if not os.path.exists(folder_path):\n            missing_images.append(image_name)\n            continue\n        \n        found = False\n        for ext in ['.jpg']:\n            file_path = os.path.join(folder_path, f\"{image_name}{ext}\")\n            if os.path.exists(file_path):\n                found_images.append(file_path)\n                found = True\n                break\n        \n        if not found:\n            missing_images.append(image_name)\n    \n    return found_images, missing_images\n\nif 'id' in top_10_images_df.columns:\n    image_names_list = top_10_images_df['id'].tolist()\nelif 'image_id' in top_10_images_df.columns:\n    image_names_list = top_10_images_df['image_id'].tolist()\nelse:\n    available_cols = [col for col in top_10_images_df.columns if col != 'landmark_id']\n    if available_cols:\n        image_names_list = top_10_images_df[available_cols[0]].tolist()\n    else:\n        image_names_list = []\n\ntrain_folder = \"/kaggle/input/landmark-recognition-2020/train\"\nfound_paths, missing_images = find_image_files_optimized(image_names_list, train_folder)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T16:53:44.458567Z","iopub.execute_input":"2025-09-24T16:53:44.459163Z","iopub.status.idle":"2025-09-24T16:53:54.831049Z","shell.execute_reply.started":"2025-09-24T16:53:44.459135Z","shell.execute_reply":"2025-09-24T16:53:54.830407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndataset_root = \"/kaggle/working/landmark_dataset_fixed\"\nif os.path.exists(dataset_root):\n    shutil.rmtree(dataset_root)\n\ntrain_dir = os.path.join(dataset_root, \"train\")\nval_dir = os.path.join(dataset_root, \"val\")\nos.makedirs(train_dir, exist_ok=True)\nos.makedirs(val_dir, exist_ok=True)\n\nunique_classes = sorted(top_10_images_df['landmark_id'].unique())\n\nimage_data = []\nimage_names_set = set()\n\nfor image_path in found_paths:\n    image_name = Path(image_path).stem\n    if image_name in image_names_set:\n        continue\n    image_names_set.add(image_name)\n    \n    corresponding_row = top_10_images_df[top_10_images_df['id'] == image_name]\n    if not corresponding_row.empty:\n        landmark_id = corresponding_row['landmark_id'].iloc[0]\n        image_data.append({\n            'image_path': image_path,\n            'image_name': image_name,\n            'landmark_id': landmark_id\n        })\n\ndf = pd.DataFrame(image_data)\n\nif len(df) < 100:\n    raise ValueError(f\"Too few images: {len(df)}\")\n\ndf = df.sample(frac=1, random_state=np.random.randint(1, 1000)).reset_index(drop=True)\n\nmin_samples_per_class = df['landmark_id'].value_counts().min()\nif min_samples_per_class < 10:\n    raise ValueError(\"Some classes have too few samples\")\n\ntrain_df, val_df = train_test_split(\n    df, \n    test_size=0.2,\n    random_state=42,\n    stratify=df['landmark_id'],\n    shuffle=True\n)\n\nfor class_id in unique_classes:\n    class_name = str(class_id)\n    os.makedirs(os.path.join(train_dir, class_name), exist_ok=True)\n    os.makedirs(os.path.join(val_dir, class_name), exist_ok=True)\n\ndef safe_copy_images(df, dest_dir, split_name):\n    copied = []\n    for _, row in df.iterrows():\n        src = row['image_path']\n        class_id = str(row['landmark_id'])\n        filename = os.path.basename(src)\n        dest = os.path.join(dest_dir, class_id, filename)\n        \n        if not os.path.exists(src):\n            continue\n            \n        try:\n            shutil.copy2(src, dest)\n            copied.append(filename)\n        except Exception as e:\n            continue\n    \n    return set(copied)\n\ntrain_copied = safe_copy_images(train_df, train_dir, \"Train\")\nval_copied = safe_copy_images(val_df, val_dir, \"Validation\")\n\noverlap = train_copied.intersection(val_copied)\nif overlap:\n    shutil.rmtree(dataset_root)\n    os.makedirs(train_dir, exist_ok=True)\n    os.makedirs(val_dir, exist_ok=True)\n    for class_id in unique_classes:\n        class_name = str(class_id)\n        os.makedirs(os.path.join(train_dir, class_name), exist_ok=True)\n        os.makedirs(os.path.join(val_dir, class_name), exist_ok=True)\n    \n    train_df_clean = train_df[~train_df['image_name'].isin(overlap)]\n    val_df_clean = val_df[~val_df['image_name'].isin(overlap)]\n    safe_copy_images(train_df_clean, train_dir, \"Train Clean\")\n    safe_copy_images(val_df_clean, val_dir, \"Validation Clean\")\n\nyaml_dir = \"/kaggle/working/config_fixed\"\nos.makedirs(yaml_dir, exist_ok=True)\nyaml_path = os.path.join(yaml_dir, \"data.yaml\")\n\nyaml_content = {\n    'path': dataset_root,\n    'train': 'train', \n    'val': 'val',\n    'nc': len(unique_classes),\n    'names': {i: str(cid) for i, cid in enumerate(unique_classes)}\n}\n\nwith open(yaml_path, 'w') as f:\n    yaml.dump(yaml_content, f, default_flow_style=False, sort_keys=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T16:54:00.765577Z","iopub.execute_input":"2025-09-24T16:54:00.766284Z","iopub.status.idle":"2025-09-24T16:54:25.797834Z","shell.execute_reply.started":"2025-09-24T16:54:00.766259Z","shell.execute_reply":"2025-09-24T16:54:25.79723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install ultralytics\n\n# If you're in a Kaggle notebook, you might also need:\n!pip install roboflow\n\n# For additional functionality (optional)\n!pip install albumentations\n!pip install Pillow","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T16:51:44.381478Z","iopub.execute_input":"2025-09-24T16:51:44.381839Z","iopub.status.idle":"2025-09-24T16:52:22.994519Z","shell.execute_reply.started":"2025-09-24T16:51:44.381789Z","shell.execute_reply":"2025-09-24T16:52:22.993809Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfrom ultralytics import YOLO\n\nmodel = YOLO('yolov8n-cls.pt')\n\nresults = model.train(\n    data=dataset_root,\n    epochs=24,\n    imgsz=224,\n    batch=32,\n    lr0=0.001,\n    lrf=0.01,\n    optimizer='AdamW',\n    weight_decay=0.01,\n    augment=True,\n    hsv_h=0.01,\n    hsv_s=0.5,\n    hsv_v=0.3,\n    translate=0.05,\n    scale=0.1,\n    fliplr=0.3,\n    dropout=0.1,\n    patience=20,\n    cos_lr=True,\n    warmup_epochs=2,\n    warmup_momentum=0.8,\n    warmup_bias_lr=0.1,\n    device=0,\n    workers=4,\n    seed=42,\n    pretrained=True,\n    verbose=True,\n    val=True,\n    plots=True,\n    save_dir='/kaggle/working/runs/classify',\n    exist_ok=True,\n    save_period=2,\n)\n\nif hasattr(results, 'results'):\n    metrics = results.results\n    \n    for key in ['train/loss', 'val/loss', 'metrics/accuracy_top1', 'metrics/accuracy_top5']:\n        if key in metrics:\n            pass\n    \n    if 'train/loss' in metrics and 'val/loss' in metrics:\n        train_loss = metrics['train/loss']\n        val_loss = metrics['val/loss']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T16:54:37.905116Z","iopub.execute_input":"2025-09-24T16:54:37.905394Z","iopub.status.idle":"2025-09-24T17:09:43.548466Z","shell.execute_reply.started":"2025-09-24T16:54:37.905371Z","shell.execute_reply":"2025-09-24T17:09:43.547671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom pathlib import Path\n\ntest_root = \"/kaggle/input/landmark-recognition-2020/test\"\n\n# Recursively find all JPG images\ntest_image_paths = [str(p) for p in Path(test_root).rglob(\"*.jpg\")]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T18:47:35.991122Z","iopub.execute_input":"2025-09-24T18:47:35.991826Z","iopub.status.idle":"2025-09-24T18:48:08.404765Z","shell.execute_reply.started":"2025-09-24T18:47:35.991802Z","shell.execute_reply":"2025-09-24T18:48:08.403947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random.seed(42)\n# Take 1000 random samples\nsampled_test_images = random.sample(test_image_paths, k=1000)\ntrained_model_path = \"/kaggle/working/my_dataset/runs/classify/train/weights/best.pt\"\nmodel = YOLO(trained_model_path)\nresults = model.predict(\n    source=sampled_test_images,\n    imgsz=224,\n    batch=32,\n    device=0,\n    augment=False,\n    verbose=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T18:49:03.198148Z","iopub.execute_input":"2025-09-24T18:49:03.198443Z","iopub.status.idle":"2025-09-24T18:49:25.061151Z","shell.execute_reply.started":"2025-09-24T18:49:03.198421Z","shell.execute_reply":"2025-09-24T18:49:25.060407Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = []\nfor r in results:  # r corresponds to one image\n    image_path = r.path\n    pred_class_index = int(r.probs.top1)       # Use .top1\n    pred_class_conf = float(r.probs.top1conf) # Use .top1conf\n    \n    predictions.append({\n        'image_path': image_path,\n        'predicted_class': pred_class_index,\n        'confidence': pred_class_conf\n    })\n\nimport pandas as pd\ndf_preds = pd.DataFrame(predictions)\nprint(df_preds.head())\ndf_preds.to_csv(\"/kaggle/working/test_predictions.csv\", index=False)\nprint(\"Predictions saved!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-24T12:29:24.188079Z","iopub.execute_input":"2025-09-24T12:29:24.188769Z","iopub.status.idle":"2025-09-24T12:29:24.249669Z","shell.execute_reply.started":"2025-09-24T12:29:24.188744Z","shell.execute_reply":"2025-09-24T12:29:24.249047Z"}},"outputs":[],"execution_count":null}]}