{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.13"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":5127,"databundleVersionId":868727,"sourceType":"competition"}],"dockerImageVersionId":31154,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# PART A: Dataset Analysis","metadata":{}},{"cell_type":"code","source":"#!rm -rf /kaggle/working/*","metadata":{"execution":{"iopub.execute_input":"2025-10-11T17:48:58.105308Z","iopub.status.busy":"2025-10-11T17:48:58.105007Z","iopub.status.idle":"2025-10-11T17:48:58.109909Z","shell.execute_reply":"2025-10-11T17:48:58.108924Z","shell.execute_reply.started":"2025-10-11T17:48:58.105287Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## INPUT FILES — Painter by Numbers Dataset\n\n### all_data_info.csv\n- Includes artist,date,genre,pixelsx,pixelsy,size_bytes,source,style,title,artist_group,in_train,new_filename\n- artist_group field says if data is present only in train/test/in both\n\n### replacements_for_corrupted_files.zip\n- Contains the replacements for corrupted data\n- 3 in test and 7 in train , total 10 to be replaced\n\n### test.zip\n- Images in the test set.\n\n### train.zip\n- Images in the train set.\n\n### train_1.zip to train_9.zip\n- subsets of train.zip.\n- Dataset is split into multiple ZIPs to make downloading and processing easier.\n\n### train_info.csv\n- Metadata for training images.\n- Includes filename,artist,title,style,genre,date.","metadata":{"jp-MarkdownHeadingCollapsed":true}},{"cell_type":"markdown","source":"## Importing All modules","metadata":{}},{"cell_type":"code","source":"# ===============================\n# Standard Library Imports\n# ===============================\nimport itertools\nimport os\nimport re\nimport shutil\nimport time\nimport zipfile\nfrom datetime import datetime\n\n# ===============================\n# Utility & Progress Tracking\n# ===============================\nfrom tqdm import tqdm\n\n# ===============================\n# Data Handling & Numerical\n# ===============================\nimport numpy as np\nimport pandas as pd\n\n# ===============================\n# Visualization\n# ===============================\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom matplotlib.colors import ListedColormap, LinearSegmentedColormap\nfrom mpl_toolkits.axes_grid1 import make_axes_locatable\nmatplotlib.rc_file_defaults()\n\n# ===============================\n# Image Processing\n# ===============================\nimport cv2\nfrom PIL import Image\nfrom skimage.feature import hog\n\n# ===============================\n# Scikit-learn\n# ===============================\nfrom sklearn import preprocessing, svm\nfrom sklearn.calibration import CalibratedClassifierCV\nfrom sklearn.cluster import KMeans\nfrom sklearn.metrics import (\n    accuracy_score,\n    classification_report,\n    confusion_matrix\n)\nfrom sklearn.model_selection import (\n    cross_val_score,\n    train_test_split\n)\nfrom sklearn.multiclass import OneVsRestClassifier\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.svm import SVC\n\n# ===============================\n# PyTorch\n# ===============================\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader, SubsetRandomSampler\nfrom torchvision import models, transforms","metadata":{"execution":{"iopub.status.busy":"2025-10-13T04:09:53.117241Z","iopub.execute_input":"2025-10-13T04:09:53.117752Z","iopub.status.idle":"2025-10-13T04:10:01.851539Z","shell.execute_reply.started":"2025-10-13T04:09:53.117723Z","shell.execute_reply":"2025-10-13T04:10:01.85067Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Loading data","metadata":{}},{"cell_type":"code","source":"path = '../input/painter-by-numbers/'\ndf = pd.read_csv(path+'all_data_info.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2025-10-13T04:10:01.852849Z","iopub.execute_input":"2025-10-13T04:10:01.853549Z","iopub.status.idle":"2025-10-13T04:10:02.382998Z","shell.execute_reply.started":"2025-10-13T04:10:01.853528Z","shell.execute_reply":"2025-10-13T04:10:02.382204Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Dataset details","metadata":{}},{"cell_type":"code","source":"print(f\"The full dataset contains a total of {len(df['artist'].unique())} different artists and {len(df['genre'].unique())} unique painting genres.\\n\")\nprint(f\"Total images: {len(df)}\")","metadata":{"execution":{"iopub.status.busy":"2025-10-13T04:10:02.383826Z","iopub.execute_input":"2025-10-13T04:10:02.384334Z","iopub.status.idle":"2025-10-13T04:10:02.400961Z","shell.execute_reply.started":"2025-10-13T04:10:02.384293Z","shell.execute_reply":"2025-10-13T04:10:02.400157Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Cleaning :\n## Corrupted Image Replacement\n\n**What this code does:**\n1. **Identifies corrupted images** - Reads original images from `train.zip` and `test.zip`\n2. **Detects corruption** - Catches images that cannot be loaded (KeyError, UnidentifiedImageError, etc.)\n3. **Replaces with fixed versions** - Loads replacement images from `replacements_for_corrupted_files.zip`\n4. **Visualizes before/after** - Shows corrupted (black placeholder) vs fixed images side-by-side\n5. **Reports statistics** - 10 corrupted images were successfully identified and fixed","metadata":{}},{"cell_type":"code","source":"archive = zipfile.ZipFile(path + 'replacements_for_corrupted_files.zip', 'r')\ntrain_dir, test_dir = '/kaggle/working/train', '/kaggle/working/test'\n\ncorrupted_map = {\n    item.split('/')[-1].replace('.jpg', ''): item\n    for item in archive.namelist()\n    if item.endswith('.jpg')\n}","metadata":{"execution":{"iopub.status.busy":"2025-10-13T04:10:02.402559Z","iopub.execute_input":"2025-10-13T04:10:02.40281Z","iopub.status.idle":"2025-10-13T04:10:02.416464Z","shell.execute_reply.started":"2025-10-13T04:10:02.402781Z","shell.execute_reply":"2025-10-13T04:10:02.41572Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"=\"*60)\nprint(\"STEP 1: REPLACING CORRUPTED FILES\")\nprint(\"=\"*60)\n\n# Create temporary directory for replacements\ntemp_replacement_dir = '/kaggle/working/temp_replacements'\nos.makedirs(temp_replacement_dir, exist_ok=True)\n\n# Open replacement zip (using your corrupted_map)\nreplacement_zip = zipfile.ZipFile(path + 'replacements_for_corrupted_files.zip', 'r')\n\n# Filter out Mac OS hidden files (._*) and count unique corrupted files\nunique_corrupted = set([img_id for img_id in corrupted_map.keys() if not img_id.startswith('._')])\nprint(f\"\\n🔧 Found {len(unique_corrupted)} unique corrupted files (filtered out Mac OS metadata)\")\nprint(f\"📋 Corrupted file IDs: {sorted(list(unique_corrupted))}\\n\")\n\nprint(f\"Extracting replacement files...\")\nfor img_id, zip_path in tqdm(corrupted_map.items(), desc=\"Replacing corrupted files\"):\n    # Skip Mac OS metadata files\n    if img_id.startswith('._'):\n        continue\n    \n    filename = img_id + '.jpg'\n    output_path = os.path.join(temp_replacement_dir, filename)\n    try:\n        img_bytes = replacement_zip.read(zip_path)\n        with open(output_path, 'wb') as f:\n            f.write(img_bytes)\n    except Exception as e:\n        print(f\"❌ Failed to extract replacement {img_id}: {e}\")\n\nreplacement_zip.close()\nprint(f\"✅ All corrupted files replaced!\\n\")\n\nprint(\"=\"*60)\nprint(\"STEP 2: FINDING TOP 24 ARTISTS (UNIFORM 495 IMAGES EACH)\")\nprint(\"=\"*60)\n\n# Find top 24 artists by image count\nprint(\"\\n🎨 Finding top 24 artists...\")\nartist_counts = df['artist'].value_counts()\ntop_24_artists = artist_counts.head(24).index.tolist()\n\nprint(f\"✅ Top 24 artists (original counts):\")\nfor i, (artist, count) in enumerate(artist_counts.head(24).items(), 1):\n    print(f\"   {i:2d}. {artist}: {count} images\")\n\n# Sample 495 images per artist uniformly\nprint(f\"\\n📊 Sampling 495 images per artist for uniformity...\")\ndf_top24_balanced = []\nfor artist in top_24_artists:\n    artist_df = df[df['artist'] == artist]\n    if len(artist_df) >= 495:\n        sampled = artist_df.sample(n=495, random_state=42)\n    else:\n        print(f\"⚠️ {artist} has only {len(artist_df)} images (< 495)\")\n        sampled = artist_df\n    df_top24_balanced.append(sampled)\n\ndf_top24 = pd.concat(df_top24_balanced, ignore_index=True)\nprint(f\"\\n✅ Balanced dataset: {len(df_top24)} total images (495 per artist × 24 artists)\\n\")\n\n# Setup output directories\noutput_train = '/kaggle/working/train_top24'\noutput_test = '/kaggle/working/test_top24'\nos.makedirs(output_train, exist_ok=True)\nos.makedirs(output_test, exist_ok=True)\n\n# Open zip files\ntrain_zip = zipfile.ZipFile(path + 'train.zip', 'r')\ntest_zip = zipfile.ZipFile(path + 'test.zip', 'r')\n\n# Build zip file indices\nprint(\"🔍 Building zip file indices...\")\ntrain_zip_lookup = {}\nfor name in train_zip.namelist():\n    if name.endswith('.jpg'):\n        filename = name.split('/')[-1]\n        train_zip_lookup[filename] = name\n\ntest_zip_lookup = {}\nfor name in test_zip.namelist():\n    if name.endswith('.jpg'):\n        filename = name.split('/')[-1]\n        test_zip_lookup[filename] = name\n\nprint(f\"📊 Found {len(train_zip_lookup)} train images in zip\")\nprint(f\"📊 Found {len(test_zip_lookup)} test images in zip\\n\")\n\nreplaced_count = 0\nextracted_count = 0\nnot_found = []\n\n# Process balanced top 24 artists' images\nfor _, row in tqdm(df_top24.iterrows(), total=len(df_top24), desc=\"Extracting balanced dataset\"):\n    filename = row['new_filename']\n    img_id = filename.replace('.jpg', '')\n    \n    # Determine output directory based on where file exists\n    if filename in train_zip_lookup:\n        output_dir = output_train\n        source_zip = train_zip\n        zip_lookup = train_zip_lookup\n    elif filename in test_zip_lookup:\n        output_dir = output_test\n        source_zip = test_zip\n        zip_lookup = test_zip_lookup\n    else:\n        not_found.append(filename)\n        continue\n    \n    output_path = os.path.join(output_dir, filename)\n    \n    # Check if this image was replaced (use replacement file)\n    replacement_path = os.path.join(temp_replacement_dir, filename)\n    if os.path.exists(replacement_path):\n        # Use REPLACEMENT image from temp directory\n        try:\n            with open(replacement_path, 'rb') as f_in:\n                with open(output_path, 'wb') as f_out:\n                    f_out.write(f_in.read())\n            replaced_count += 1\n        except Exception as e:\n            print(f\"❌ Failed to copy replacement {filename}: {e}\")\n            not_found.append(filename)\n    else:\n        # Use ORIGINAL image from zip\n        try:\n            img_bytes = source_zip.read(zip_lookup[filename])\n            with open(output_path, 'wb') as f:\n                f.write(img_bytes)\n            extracted_count += 1\n        except Exception as e:\n            print(f\"❌ Failed to extract {filename}: {e}\")\n            not_found.append(filename)\n\n# Close zip files\ntrain_zip.close()\ntest_zip.close()\n\nprint(f\"\\n{'='*60}\")\nprint(f\"FINAL STATISTICS\")\nprint(f\"{'='*60}\")\nprint(f\" Total images extracted: {extracted_count + replaced_count}\")\nprint(f\" Corrupted images replaced: {replaced_count}\")\nprint(f\" Original images extracted: {extracted_count}\")\nprint(f\" Images not found: {len(not_found)}\")\nprint(f\"{'='*60}\")\n\nif not_found and len(not_found) < 20:\n    print(f\"\\n❌ Missing files: {not_found}\")\n\nprint(f\"\\n Balanced top 24 artists dataset ready at:\")\nprint(f\"   Training images: {output_train}/\")\nprint(f\"   Test images: {output_test}/\")\n\n# Save filtered dataframe for future use\ndf_top24.to_csv('/kaggle/working/top24_artists_balanced_info.csv', index=False)\nprint(f\"\\n💾 Saved balanced dataframe: /kaggle/working/top24_artists_balanced_info.csv\")\nprint(f\"\\n💡 Use this balanced dataset for fair model training!\")\n\n# Clean up temp directory\nshutil.rmtree(temp_replacement_dir)\nprint(f\"🧹 Cleaned up temporary files\")","metadata":{"execution":{"iopub.status.busy":"2025-10-13T04:10:02.417406Z","iopub.execute_input":"2025-10-13T04:10:02.417675Z","iopub.status.idle":"2025-10-13T04:12:37.829633Z","shell.execute_reply.started":"2025-10-13T04:10:02.41765Z","shell.execute_reply":"2025-10-13T04:12:37.828841Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select Top 24 Artists\n\ndf_top24 = pd.read_csv('/kaggle/working/top24_artists_balanced_info.csv')\n\nN = 24  # Number of artists\npaintings = df_top24['artist'].value_counts()\nartists = paintings.index.tolist()\nsample_size = 495  # Already balanced to 495 per artist\n\nprint(f\"Selected {N} artists for classification:\")\nfor i, artist in enumerate(artists, 1):\n    print(f\"{i}. {artist}: {paintings[artist]} paintings\")\nprint(f\"\\nSample size per artist (balanced dataset): {sample_size} paintings\")\nprint(f\"Total images to be used: {N * sample_size} = {len(df_top24)}\")\n","metadata":{"execution":{"iopub.status.busy":"2025-10-13T04:12:47.856955Z","iopub.execute_input":"2025-10-13T04:12:47.857224Z","iopub.status.idle":"2025-10-13T04:12:47.893732Z","shell.execute_reply.started":"2025-10-13T04:12:47.857204Z","shell.execute_reply":"2025-10-13T04:12:47.892966Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a copy\nactive_df = df_top24.copy()\n\nprint(f\"Total images selected: {len(active_df)}\")\nprint(f\"Images per artist: {sample_size}\")\nprint(f\"Number of artists: {len(artists)}\")\n\n# Verify balance\nprint(\"\\nClass distribution:\")\nprint(active_df['artist'].value_counts())\n","metadata":{"execution":{"iopub.status.busy":"2025-10-13T04:12:48.124236Z","iopub.execute_input":"2025-10-13T04:12:48.124849Z","iopub.status.idle":"2025-10-13T04:12:48.132171Z","shell.execute_reply.started":"2025-10-13T04:12:48.124824Z","shell.execute_reply":"2025-10-13T04:12:48.131353Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Label Encoder Setup\n\nLabEnc = preprocessing.LabelEncoder()\nLabEnc.fit(artists)\n\nprint(f\"Number of classes: {len(artists)}\")\nprint(f\"\\nArtist Label Mapping:\")\nfor i, artist in enumerate(artists):\n    print(f\"{i}: {artist}\")","metadata":{"execution":{"iopub.status.busy":"2025-10-13T04:12:48.385955Z","iopub.execute_input":"2025-10-13T04:12:48.386231Z","iopub.status.idle":"2025-10-13T04:12:48.391854Z","shell.execute_reply.started":"2025-10-13T04:12:48.386209Z","shell.execute_reply":"2025-10-13T04:12:48.39094Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Few Function def for :\n-  ### Image Preprocessing\n-  ### Dataset Construction\n-  ### Evaluation & Visualization","metadata":{}},{"cell_type":"code","source":"# Image Transformer Function\n\ndef image_transformer_nn(image, apply_norm=True, crop_img=True, new_dim=224):\n    \"\"\"\n    Args:\n        image: PIL Image object\n        apply_norm (bool): Apply ImageNet normalization\n        crop_img (bool): Crop center square vs resize with padding\n        new_dim (int): Target dimension (224 for ResNet)\n    \"\"\"\n    if crop_img:\n        cropper = transforms.CenterCrop(new_dim)\n        image = cropper(image)\n    \n    tensoring = transforms.ToTensor()\n    image = tensoring(image)\n    channels, height, width = image.shape\n    \n    # Handle grayscale images\n    if image.shape[0] < 3:\n        image = image.expand(3, -1, -1)\n    # Handle extra channels\n    if image.shape[0] > 3:\n        image = image[0:3, :, :]\n    \n    # ImageNet normalization\n    if apply_norm:\n        normalizer = transforms.Normalize((0.485, 0.456, 0.406), (0.229, 0.224, 0.225))\n        image = normalizer(image)\n    \n    if not crop_img:\n        if width < height:\n            image = image.transpose(1, 2)\n        channels, height, width = image.shape\n        res_percent = float(new_dim / width)\n        height = round(height * res_percent)\n        resizer = transforms.Resize((height, new_dim))\n        image = resizer(image)\n        padder = transforms.Pad([0, 0, 0, int(new_dim - height)])\n        image = padder(image)\n    \n    return image\n\n# Example usage with extracted files\ntrain_images_path = '/kaggle/working/train_top24/'\ntest_images_path = '/kaggle/working/test_top24/'\n\n# Get a sample image file\nsample_files = [f for f in os.listdir(train_images_path) if f.endswith('.jpg')]\nif sample_files:\n    sample_file = sample_files[0]\n    image = Image.open(os.path.join(train_images_path, sample_file))\n    \n    print(\"Original image:\")\n    plt.figure(figsize=(12, 3))\n    plt.subplot(1, 4, 1)\n    plt.imshow(image)\n    plt.title(\"Original\")\n    plt.axis('off')\n    \n    print(\"Cropped transformed image:\")\n    image2 = image_transformer_nn(image, apply_norm=False, crop_img=True, new_dim=224)\n    plt.subplot(1, 4, 2)\n    plt.imshow(image2.numpy().transpose(1, 2, 0))\n    plt.title(\"Cropped\")\n    plt.axis('off')\n    \n    print(\"Resized with padding:\")\n    image3a = image_transformer_nn(image, apply_norm=False, crop_img=False, new_dim=224)\n    plt.subplot(1, 4, 3)\n    plt.imshow(image3a.numpy().transpose(1, 2, 0))\n    plt.title(\"Resized+Padded\")\n    plt.axis('off')\n    \n    print(\"Resized with normalization:\")\n    image3b = image_transformer_nn(image, apply_norm=True, crop_img=False, new_dim=224)\n    plt.subplot(1, 4, 4)\n    plt.imshow(image3b.numpy().transpose(1, 2, 0))\n    plt.title(\"Normalized\")\n    plt.axis('off')\n    \n    plt.tight_layout()\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2025-10-13T04:12:55.800218Z","iopub.execute_input":"2025-10-13T04:12:55.800899Z","iopub.status.idle":"2025-10-13T04:12:56.311564Z","shell.execute_reply.started":"2025-10-13T04:12:55.800856Z","shell.execute_reply":"2025-10-13T04:12:56.310844Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ImageDataset Class for Neural Networks\n\nclass ImageDataset(Dataset):\n    def __init__(self, dataframe, lab_encoder, img_size=224, normalize=True, crop=False):\n        \"\"\"\n        Args:\n            dataframe (pd.DataFrame): Dataframe with image info\n            lab_encoder: Label encoder for artist names\n            img_size (int): Image dimension (default 224)\n            normalize (bool): Apply ImageNet normalization\n            crop (bool): True=crop center, False=resize with padding\n        \"\"\"\n        self.encoder = lab_encoder\n        self.img_size = img_size\n        self.normalize = normalize\n        self.crop = crop\n        self.train_path = '/kaggle/working/train_top24/'\n        self.test_path = '/kaggle/working/test_top24/'\n        self.feats, self.labels = self.get_all_items(dataframe)\n    \n    def get_all_items(self, dataframe):\n        feats = []\n        labels = []\n        \n        for index, row in dataframe.iterrows():\n            filename = row['new_filename']\n            \n            # Determine if image is in train or test\n            if row['in_train']:\n                img_path = os.path.join(self.train_path, filename)\n            else:\n                img_path = os.path.join(self.test_path, filename)\n            \n            try:\n                # Load image from extracted files\n                image = Image.open(img_path)\n                datum = image_transformer_nn(image, apply_norm=self.normalize,\n                                            crop_img=self.crop, new_dim=self.img_size)\n                feats.append(datum)\n                \n                # Label\n                artist = row['artist']\n                label = self.encoder.transform([artist])[0]\n                labels.append(label)\n            except Exception as e:\n                print(f\"Error loading {filename}: {e}\")\n        \n        feats = torch.stack(feats)\n        labels = torch.LongTensor(labels)\n        return feats, labels\n    \n    def __len__(self):\n        return len(self.labels)\n    \n    def __getitem__(self, item):\n        return self.feats[item], self.labels[item]","metadata":{"execution":{"iopub.status.busy":"2025-10-13T04:12:56.313045Z","iopub.execute_input":"2025-10-13T04:12:56.313378Z","iopub.status.idle":"2025-10-13T04:12:56.324459Z","shell.execute_reply.started":"2025-10-13T04:12:56.313352Z","shell.execute_reply":"2025-10-13T04:12:56.323629Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ImageData Function for SVM/Other Models\n\ndef ImageData(dataframe, lab_encoder, hog_mode=None, sift_mode=None, img_size=224):\n    \"\"\"\n    Load images for non-NN models (SVM, etc.)\n    \n    Args:\n        dataframe: Dataframe with image info\n        lab_encoder: Label encoder\n        hog_mode: Tuple (orientations, pixels_per_cell, cells_per_block) or None\n        sift_mode: True to use SIFT features, False otherwise\n        img_size: Image dimension\n    \"\"\"\n    train_path = '/kaggle/working/train_top24/'\n    test_path = '/kaggle/working/test_top24/'\n    \n    PaintFeats = []\n    PaintLabels = []\n    \n    for index, row in dataframe.iterrows():\n        filename = row['new_filename']\n        \n        # Determine path\n        if row['in_train']:\n            img_path = os.path.join(train_path, filename)\n        else:\n            img_path = os.path.join(test_path, filename)\n        \n        try:\n            image = Image.open(img_path)\n            datum = image_transformer_nn(image, apply_norm=False, crop_img=False, new_dim=img_size)\n            np_datum = datum.numpy().transpose(1, 2, 0)\n            \n            if hog_mode:\n                orients, ppc, cpb = hog_mode[0], hog_mode[1], hog_mode[2]\n                datum = hog(np_datum, orientations=orients, pixels_per_cell=ppc,\n                           cells_per_block=cpb, feature_vector=True, channel_axis=2)\n                PaintFeats.append(datum)\n            \n            elif sift_mode:\n                np_datum = cv2.normalize(np_datum, None, 0, 255, cv2.NORM_MINMAX).astype('uint8')\n                imgtogray = cv2.cvtColor(np_datum, cv2.COLOR_BGR2GRAY)\n                PaintFeats.append(imgtogray)\n            \n            # Label\n            artist = row['artist']\n            label = lab_encoder.transform([artist])[0]\n            PaintLabels.append(label)\n        except Exception as e:\n            print(f\"Error loading {filename}: {e}\")\n    \n    return np.asarray(PaintFeats), np.asarray(PaintLabels)\n","metadata":{"execution":{"iopub.status.busy":"2025-10-13T04:12:56.325709Z","iopub.execute_input":"2025-10-13T04:12:56.325999Z","iopub.status.idle":"2025-10-13T04:12:56.349695Z","shell.execute_reply.started":"2025-10-13T04:12:56.325975Z","shell.execute_reply":"2025-10-13T04:12:56.348909Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Data Splitter Function\n\ndef DataSplitter(data, ratios=[60, 20, 20], need_val=True, batches=None, shuffle=True, seed=42):\n    \"\"\"\n    Split data into train/val/test sets\n    \n    Args:\n        data: ImageDataset or tuple (X, y)\n        ratios: [train%, val%, test%]\n        need_val: Whether to create validation set\n        batches: Batch size for DataLoaders (NN case)\n        shuffle: Shuffle data\n        seed: Random seed\n    \"\"\"\n    first_ratio = (ratios[1] + ratios[2]) / sum(ratios)\n    second_ratio = ratios[2] / (ratios[1] + ratios[2])\n    \n    if isinstance(data, ImageDataset):  # Neural Network case\n        labels = data.labels.numpy()\n        \n        train_indices, rest_indices = train_test_split(\n            np.arange(len(labels)),\n            test_size=first_ratio,\n            shuffle=shuffle,\n            random_state=seed,\n            stratify=labels\n        )\n        \n        rest_labels = data[rest_indices][1]\n        \n        val_indices, test_indices = train_test_split(\n            rest_indices,\n            test_size=second_ratio,\n            shuffle=shuffle,\n            random_state=seed,\n            stratify=rest_labels\n        )\n        \n        train_sampler = SubsetRandomSampler(train_indices)\n        val_sampler = SubsetRandomSampler(val_indices)\n        test_sampler = SubsetRandomSampler(test_indices)\n        \n        train_loader = DataLoader(data, batch_size=batches, sampler=train_sampler)\n        val_loader = DataLoader(data, batch_size=batches, sampler=val_sampler)\n        test_loader = DataLoader(data, batch_size=batches, sampler=test_sampler)\n        \n        return train_loader, val_loader, test_loader\n    \n    elif isinstance(data, tuple):  # SVM/other models case\n        X_train, X_rest, y_train, y_rest = train_test_split(\n            data[0], data[1],\n            test_size=first_ratio,\n            shuffle=shuffle,\n            random_state=seed,\n            stratify=data[1]\n        )\n        \n        if need_val:\n            X_val, X_test, y_val, y_test = train_test_split(\n                X_rest, y_rest,\n                test_size=second_ratio,\n                shuffle=shuffle,\n                random_state=seed,\n                stratify=y_rest\n            )\n            return X_train, X_val, X_test, y_train, y_val, y_test\n        \n        return X_train, X_rest, y_train, y_rest\n    \n    else:\n        print('Invalid data type. Use ImageDataset or tuple (X, y).')\n        return","metadata":{"execution":{"iopub.status.busy":"2025-10-13T04:12:56.650215Z","iopub.execute_input":"2025-10-13T04:12:56.650498Z","iopub.status.idle":"2025-10-13T04:12:56.658057Z","shell.execute_reply.started":"2025-10-13T04:12:56.650478Z","shell.execute_reply":"2025-10-13T04:12:56.657346Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualization Functions\n\nsns.set(style=\"darkgrid\")\n\ndef CustomCmap(from_rgb, to_rgb):\n    \"\"\"Create custom colormap\"\"\"\n    r1, g1, b1 = from_rgb\n    r2, g2, b2 = to_rgb\n    cdict = {\n        'red': ((0, r1, r1), (1, r2, r2)),\n        'green': ((0, g1, g1), (1, g2, g2)),\n        'blue': ((0, b1, b1), (1, b2, b2))\n    }\n    cmap = LinearSegmentedColormap('custom_cmap', cdict)\n    return cmap\n\nmycmap = CustomCmap([1.0, 1.0, 1.0], [72/255, 99/255, 147/255])\nmycmap_r = CustomCmap([72/255, 99/255, 147/255], [1.0, 1.0, 1.0])\nmycol = (72/255, 99/255, 147/255)\nmycomplcol = (129/255, 143/255, 163/255)\n\ndef plot_cm(cfmatrix, title, classes):\n    \"\"\"Plot confusion matrix\"\"\"\n    fig, ax1 = plt.subplots(1, 1, figsize=(12, 10))\n    \n    im = ax1.imshow(cfmatrix, interpolation='nearest', cmap=mycmap)\n    divider = make_axes_locatable(ax1)\n    cax = divider.append_axes(\"right\", size=\"5%\", pad=0.2)\n    plt.colorbar(im, cax=cax)\n    \n    ax1.set_title(title, fontsize=14)\n    tick_marks = np.arange(len(classes))\n    ax1.set_xticks(tick_marks)\n    ax1.set_xticklabels(classes, rotation=90)\n    ax1.set_yticks(tick_marks)\n    ax1.set_yticklabels(classes)\n    \n    fmt = 'd'\n    thresh = cfmatrix.max() / 2.\n    for i, j in itertools.product(range(cfmatrix.shape[0]), range(cfmatrix.shape[1])):\n        ax1.text(j, i, format(cfmatrix[i, j], fmt),\n                horizontalalignment=\"center\",\n                color=\"white\" if cfmatrix[i, j] > thresh else \"black\")\n    \n    ax1.set_ylabel('True label', fontsize=14)\n    ax1.set_xlabel('Predicted label', fontsize=14)\n    plt.savefig(title + '.pdf', bbox_inches='tight')\n    plt.show()\n\nprint(\"All blocks loaded successfully!\")\nprint(f\"Ready to proceed to Part B with {N} artists and {len(active_df)} images!\")","metadata":{"execution":{"iopub.status.busy":"2025-10-13T04:12:56.946012Z","iopub.execute_input":"2025-10-13T04:12:56.946641Z","iopub.status.idle":"2025-10-13T04:12:56.957688Z","shell.execute_reply.started":"2025-10-13T04:12:56.946607Z","shell.execute_reply":"2025-10-13T04:12:56.956751Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PART C: SVM model","metadata":{}},{"cell_type":"markdown","source":"### SVM + HOG","metadata":{}},{"cell_type":"code","source":"%%time\nprint(\"Loading HOG features for 24 artists...\")\n# Load the Dataset for classic training - HOG\nhog_data = ImageData(active_df, LabEnc, hog_mode=[9, (8,8), (2,2)], sift_mode=False, img_size=224)\nX_train, X_test, y_train, y_test = DataSplitter(hog_data, ratios=[85,5,10], need_val=False, batches=32, shuffle=True, seed=420)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T04:13:11.213652Z","iopub.execute_input":"2025-10-13T04:13:11.214332Z","iopub.status.idle":"2025-10-13T04:25:33.717834Z","shell.execute_reply.started":"2025-10-13T04:13:11.214286Z","shell.execute_reply":"2025-10-13T04:25:33.716952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nprint(\"Training SVM on HOG features...\")\n# Train the SVM on the hog features\nhog_classifier = svm.SVC(kernel='rbf', gamma=1.5, C=0.3)\nhog_classifier.fit(X_train, y_train)\ny_pred = hog_classifier.predict(X_test)\nmatplotlib.rc_file_defaults()\nprint(classification_report(y_test, y_pred, target_names=artists))\n\n# Confusion Matrix\ncfmatrix = confusion_matrix(y_test, y_pred)\nplot_cm(cfmatrix,'HOG - SVM Confusion Matrix (24 Artists)', artists)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T04:25:33.719005Z","iopub.execute_input":"2025-10-13T04:25:33.719225Z","iopub.status.idle":"2025-10-13T05:06:59.392832Z","shell.execute_reply.started":"2025-10-13T04:25:33.719207Z","shell.execute_reply":"2025-10-13T05:06:59.391957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import joblib\nimport pickle\nimport os\n\n# Create directory for saving models\nos.makedirs('/kaggle/working/saved_models', exist_ok=True)\n\nprint(\"=\"*70)\nprint(\"SAVING ALL SVM MODELS\")\nprint(\"=\"*70)\nprint(\"\\n1️⃣ Saving HOG + SVM Model...\")\ntry:\n    joblib.dump(hog_classifier, '/kaggle/working/saved_models/hog_svm_model.pkl')\n    # Also save the scaler if you used one (add if needed)\n    print(\"   ✅ Saved: hog_svm_model.pkl\")\nexcept Exception as e:\n    print(f\"   ❌ Error: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T05:08:05.607387Z","iopub.execute_input":"2025-10-13T05:08:05.608107Z","iopub.status.idle":"2025-10-13T05:08:07.539705Z","shell.execute_reply.started":"2025-10-13T05:08:05.608082Z","shell.execute_reply":"2025-10-13T05:08:07.53903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nExtract the HOG SVM model with clear naming for Bayes Optimal\nsvm_hog = hog_classifier  # Your SVM trained on HOG features\nX_test_hog = X_test\ny_test_hog = y_test\n\nprint(f\"\\n✅ SVM with HOG Features saved as 'svm_hog'\")\nprint(f\"   Kernel: RBF (gamma=1.5, C=0.3)\")\nprint(f\"   Test Accuracy: {accuracy_score(y_test_hog, y_pred)*100:.2f}%\")\nprint(f\"   Test set shape: {X_test_hog.shape}\")\n'''","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### SVM :Color Histogram + OvR","metadata":{}},{"cell_type":"code","source":"%%time\nprint(\"=\"*60)\nprint(\"Extracting Color Histogram features (24 Artists)...\")\nprint(\"=\"*60)\n\ndef extract_color_histogram(image, bins=(8, 8, 8)):\n    \"\"\"Extract color histogram features from RGB image\"\"\"\n    if image.max() <= 1.0:\n        image = (image * 255).astype(np.uint8)\n    \n    hist_features = []\n    for i in range(3):  # RGB channels\n        hist, _ = np.histogram(image[:, :, i], bins=bins[i], range=(0, 256))\n        hist_features.extend(hist)\n    \n    hist_features = np.array(hist_features, dtype=float)\n    hist_features = hist_features / (hist_features.sum() + 1e-6)\n    return hist_features\n\n# Load from extracted directories instead of zip archives\ntrain_path = '/kaggle/working/train_top24/'\ntest_path = '/kaggle/working/test_top24/'\n\nX_color = []\ny_color = []\n\n# Process training data\ncurr_df = active_df[active_df['in_train']==True]\nfor index, row in curr_df.iterrows():\n    filename = row['new_filename']\n    img_path = os.path.join(train_path, filename)\n    \n    try:\n        image = Image.open(img_path)\n        image = image.resize((128, 128))\n        img_array = np.array(image)\n        \n        # Handle grayscale\n        if len(img_array.shape) == 2:\n            img_array = np.stack([img_array]*3, axis=-1)\n        elif img_array.shape[2] > 3:\n            img_array = img_array[:, :, :3]\n            \n        features = extract_color_histogram(img_array, bins=(16, 16, 16))\n        X_color.append(features)\n        \n        artist = row['artist']\n        label = LabEnc.transform([artist])[0]\n        y_color.append(label)\n    except Exception as e:\n        print(f\"Skipped {filename}: {e}\")\n    \n    if (index + 1) % 1000 == 0:\n        print(f\"Processed {index + 1} training images...\")\n\n# Process test data\ncurr_df = active_df[active_df['in_train']==False]\nfor index, row in curr_df.iterrows():\n    filename = row['new_filename']\n    img_path = os.path.join(test_path, filename)\n    \n    try:\n        image = Image.open(img_path)\n        image = image.resize((128, 128))\n        img_array = np.array(image)\n        \n        if len(img_array.shape) == 2:\n            img_array = np.stack([img_array]*3, axis=-1)\n        elif img_array.shape[2] > 3:\n            img_array = img_array[:, :, :3]\n            \n        features = extract_color_histogram(img_array, bins=(16, 16, 16))\n        X_color.append(features)\n        \n        artist = row['artist']\n        label = LabEnc.transform([artist])[0]\n        y_color.append(label)\n    except Exception as e:\n        print(f\"Skipped {filename}: {e}\")\n\nX_color = np.array(X_color)\ny_color = np.array(y_color)\n\nprint(f\"\\nColor features shape: {X_color.shape}\")\nprint(f\"Total samples: {len(y_color)} ({len(artists)} artists)\")\nprint(f\"Feature extraction completed!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T05:09:13.576423Z","iopub.execute_input":"2025-10-13T05:09:13.576711Z","iopub.status.idle":"2025-10-13T05:14:43.385134Z","shell.execute_reply.started":"2025-10-13T05:09:13.576683Z","shell.execute_reply":"2025-10-13T05:14:43.384188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train_color, X_test_color, y_train_color, y_test_color = DataSplitter(\n    (X_color, y_color), \n    ratios=[85, 5, 10], \n    need_val=False, \n    shuffle=True, \n    seed=420\n)\n\nprint(f\"\\nTrain: {X_train_color.shape}, Test: {X_test_color.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T05:14:43.386703Z","iopub.execute_input":"2025-10-13T05:14:43.387001Z","iopub.status.idle":"2025-10-13T05:14:43.398188Z","shell.execute_reply.started":"2025-10-13T05:14:43.386984Z","shell.execute_reply":"2025-10-13T05:14:43.397501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\nprint(\"=\"*60)\nprint(\"Training Enhanced Multi-Class SVM (One-vs-Rest) - 24 Artists\")\nprint(\"=\"*60)\n\n# Standardize features\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train_color)\nX_test_scaled = scaler.transform(X_test_color)\n\n# One-vs-Rest SVM with RBF kernel (24 binary SVMs)\nbase_svm = SVC(\n    kernel='rbf',\n    C=10.0,\n    gamma='scale',\n    cache_size=1000,\n    class_weight='balanced',\n    probability=False,\n    random_state=42,\n    verbose=False\n)\n\ncolor_svm_classifier = OneVsRestClassifier(base_svm, n_jobs=-1)\n\nprint(\"\\nTraining started...\")\ncolor_svm_classifier.fit(X_train_scaled, y_train_color)\nprint(\"Training completed!\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T05:14:43.398972Z","iopub.execute_input":"2025-10-13T05:14:43.399231Z","iopub.status.idle":"2025-10-13T05:15:09.152169Z","shell.execute_reply.started":"2025-10-13T05:14:43.399213Z","shell.execute_reply":"2025-10-13T05:15:09.151361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nprint(\"Making predictions...\")\ny_pred_color = color_svm_classifier.predict(X_test_scaled)\n\naccuracy = accuracy_score(y_test_color, y_pred_color)\n\nprint(\"\\n\" + \"=\"*60)\nprint(\"ENHANCED SVM (Color Histogram + OvR) RESULTS - 24 Artists\")\nprint(\"=\"*60)\nprint(f\"Test Accuracy: {accuracy:.4f} ({accuracy*100:.2f}%)\\n\")\n\nprint(classification_report(y_test_color, y_pred_color, target_names=artists))\n\nmatplotlib.rc_file_defaults()\ncfmatrix_color = confusion_matrix(y_test_color, y_pred_color)\nplot_cm(cfmatrix_color, 'Enhanced SVM - Color Histogram (24 Artists)', artists)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T05:15:09.153791Z","iopub.execute_input":"2025-10-13T05:15:09.154038Z","iopub.status.idle":"2025-10-13T05:15:19.730208Z","shell.execute_reply.started":"2025-10-13T05:15:09.15402Z","shell.execute_reply":"2025-10-13T05:15:19.729372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n2️⃣ Saving Color Histogram + SVM (OvR) Model...\")\ntry:\n    joblib.dump(color_svm_classifier, '/kaggle/working/saved_models/color_svm_ovr_model.pkl')\n    joblib.dump(scaler, '/kaggle/working/saved_models/color_scaler.pkl')\n    print(\"   ✅ Saved: color_svm_ovr_model.pkl\")\n    print(\"   ✅ Saved: color_scaler.pkl\")\nexcept Exception as e:\n    print(f\"   ❌ Error: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T05:15:19.731113Z","iopub.execute_input":"2025-10-13T05:15:19.731403Z","iopub.status.idle":"2025-10-13T05:15:19.789243Z","shell.execute_reply.started":"2025-10-13T05:15:19.731381Z","shell.execute_reply":"2025-10-13T05:15:19.788608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n# Extract the Color Histogram SVM model with clear naming for Bayes Optimal\nsvm_color = color_svm_classifier  # Your SVM trained on color histogram features\nX_test_color = X_test_scaled\n# y_test_color already exists from your code\n\nprint(f\"\\n✅ SVM with Color Histogram saved as 'svm_color'\")\nprint(f\"   Test Accuracy: {accuracy*100:.2f}%\")\nprint(f\"   Test set shape: {X_test_color.shape}\")\n'''","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Training SVM on featuress from Pre-trained ResNet18","metadata":{}},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T05:19:56.607743Z","iopub.execute_input":"2025-10-13T05:19:56.608465Z","iopub.status.idle":"2025-10-13T05:19:56.663079Z","shell.execute_reply.started":"2025-10-13T05:19:56.608438Z","shell.execute_reply":"2025-10-13T05:19:56.662228Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# IMPROVED Base CNN Architecture\n\nclass ImprovedCNNBackbone(nn.Module):\n    \"\"\"\n    Enhanced CNN with modern techniques:\n    - Batch Normalization after each conv layer\n    - Dropout for regularization\n    - Progressive channel increase\n    - Global Average Pooling\n    - Label Smoothing support\n    \"\"\"\n    def _init_(self, num_classes, dropout=0.5):\n        super(ImprovedCNNBackbone, self)._init_()\n        \n        # Convolutional layers with BatchNorm\n        self.features = nn.Sequential(\n            # Block 1: 224x224 -> 112x112\n            nn.Conv2d(3, 64, kernel_size=3, stride=1, padding=1),\n            nn.BatchNorm2d(64),\n            nn.ReLU(inplace=True),\n            nn.Conv2d(64, 64, kernel_size=3, stride=1, padding=1),\n            nn.BatchNorm2d(64),\n            nn.ReLU(inplace=True),\n            nn.MaxPool2d(kernel_size=2, stride=2),\n            \n            # Block 2: 112x112 -> 56x56\n            nn.Conv2d(64, 128, kernel_size=3, stride=1, padding=1),\n            nn.BatchNorm2d(128),\n            nn.ReLU(inplace=True),\n            nn.Conv2d(128, 128, kernel_size=3, stride=1, padding=1),\n            nn.BatchNorm2d(128),\n            nn.ReLU(inplace=True),\n            nn.MaxPool2d(kernel_size=2, stride=2),\n            \n            # Block 3: 56x56 -> 28x28\n            nn.Conv2d(128, 256, kernel_size=3, stride=1, padding=1),\n            nn.BatchNorm2d(256),\n            nn.ReLU(inplace=True),\n            nn.Conv2d(256, 256, kernel_size=3, stride=1, padding=1),\n            nn.BatchNorm2d(256),\n            nn.ReLU(inplace=True),\n            nn.MaxPool2d(kernel_size=2, stride=2),\n            \n            # Block 4: 28x28 -> 14x14\n            nn.Conv2d(256, 512, kernel_size=3, stride=1, padding=1),\n            nn.BatchNorm2d(512),\n            nn.ReLU(inplace=True),\n            nn.Conv2d(512, 512, kernel_size=3, stride=1, padding=1),\n            nn.BatchNorm2d(512),\n            nn.ReLU(inplace=True),\n            nn.MaxPool2d(kernel_size=2, stride=2),\n        )\n        \n        # Global Average Pooling (better than flatten)\n        self.gap = nn.AdaptiveAvgPool2d((1, 1))\n        \n        # Classifier with dropout\n        self.classifier = nn.Sequential(\n            nn.Dropout(dropout),\n            nn.Linear(512, 256),\n            nn.ReLU(inplace=True),\n            nn.Dropout(dropout),\n            nn.Linear(256, num_classes)\n        )\n    \n    def forward(self, x):\n        x = self.features(x)\n        x = self.gap(x)\n        x = torch.flatten(x, 1)\n        x = self.classifier(x)\n        return x\n\n\n# ============================================================================\n# IMPROVED Training Functions with Modern Techniques\n# ============================================================================\n\nclass LabelSmoothingCrossEntropy(nn.Module):\n    \"\"\"Label smoothing to prevent overconfidence\"\"\"\n    def _init_(self, smoothing=0.1):\n        super()._init_()\n        self.smoothing = smoothing\n    \n    def forward(self, pred, target):\n        n_class = pred.size(1)\n        one_hot = torch.zeros_like(pred).scatter(1, target.unsqueeze(1), 1)\n        one_hot = one_hot * (1 - self.smoothing) + self.smoothing / n_class\n        log_prob = F.log_softmax(pred, dim=1)\n        loss = -(one_hot * log_prob).sum(dim=1).mean()\n        return loss\n\n\ndef training_loop_improved(model, train_dataloader, optimizer, loss_fn, device=\"cuda\"):\n    \"\"\"Improved training loop with accuracy tracking\"\"\"\n    model.train()\n    batch_losses = []\n    correct = 0\n    total = 0\n    \n    for batch in train_dataloader:\n        x_batch, y_batch = batch\n        x_batch, y_batch = x_batch.to(device), y_batch.to(device)\n        \n        optimizer.zero_grad()\n        yhat = model(x_batch)\n        loss = loss_fn(yhat, y_batch)\n        \n        loss.backward()\n        \n        # Gradient clipping to prevent exploding gradients\n        torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=1.0)\n        \n        optimizer.step()\n        \n        batch_losses.append(loss.item())\n        \n        # Calculate accuracy\n        _, predicted = torch.max(yhat.data, 1)\n        total += y_batch.size(0)\n        correct += (predicted == y_batch).sum().item()\n    \n    train_loss = np.mean(batch_losses)\n    train_acc = 100 * correct / total\n    \n    return train_loss, train_acc\n\n\ndef validation_loop_improved(model, val_dataloader, loss_fn, device=\"cuda\"):\n    \"\"\"Improved validation loop with accuracy tracking\"\"\"\n    model.eval()\n    batch_losses = []\n    correct = 0\n    total = 0\n    \n    with torch.no_grad():\n        for batch in val_dataloader:\n            x_batch, y_batch = batch\n            x_batch, y_batch = x_batch.to(device), y_batch.to(device)\n            \n            yhat = model(x_batch)\n            loss = loss_fn(yhat, y_batch)\n            \n            batch_losses.append(loss.item())\n            \n            # Calculate accuracy\n            _, predicted = torch.max(yhat.data, 1)\n            total += y_batch.size(0)\n            correct += (predicted == y_batch).sum().item()\n    \n    val_loss = np.mean(batch_losses)\n    val_acc = 100 * correct / total\n    \n    return val_loss, val_acc\n\n\nclass EarlyStoppingImproved:\n    \"\"\"Enhanced early stopping with model checkpoint\"\"\"\n    def _init_(self, patience=10, verbose=True, delta=0.001, path='checkpoint.pt'):\n        self.patience = patience\n        self.verbose = verbose\n        self.counter = 0\n        self.best_score = None\n        self.early_stop = False\n        self.val_loss_min = np.Inf\n        self.delta = delta\n        self.path = path\n    \n    def _call_(self, val_loss, model):\n        score = -val_loss\n        \n        if self.best_score is None:\n            self.best_score = score\n            self.save_checkpoint(val_loss, model)\n        elif score < self.best_score + self.delta:\n            self.counter += 1\n            if self.verbose:\n                print(f'   ⚠ EarlyStopping counter: {self.counter}/{self.patience}')\n            if self.counter >= self.patience:\n                self.early_stop = True\n        else:\n            self.best_score = score\n            self.save_checkpoint(val_loss, model)\n            self.counter = 0\n    \n    def save_checkpoint(self, val_loss, model):\n        if self.verbose:\n            print(f'   ✅ Val loss decreased ({self.val_loss_min:.4f} → {val_loss:.4f}). Saving model...')\n        torch.save(model.state_dict(), self.path)\n        self.val_loss_min = val_loss\n\n\ndef train_improved(model, train_loader, val_loader, optimizer, scheduler, epochs, \n                   loss_fn, device=\"cuda\", patience=10, verbose_ct=1):\n    \"\"\"\n    Improved training function with:\n    - Learning rate scheduling\n    - Early stopping\n    - Accuracy tracking\n    - Better progress reporting\n    \"\"\"\n    train_losses, val_losses = [], []\n    train_accs, val_accs = [], []\n    \n    print(f\"🚀 Starting Improved CNN Training\")\n    print(f\"   Device: {device}\")\n    print(f\"   Epochs: {epochs}\")\n    print(f\"   Batch Size: {train_loader.batch_size if hasattr(train_loader, 'batch_size') else 'N/A'}\")\n    print(f\"   Patience: {patience}\")\n    print(\"=\" * 70)\n    \n    checkpoint_path = 'checkpoint_base_cnn.pt'\n    model_path = 'ImprovedCNN.pt'\n    \n    early_stopping = EarlyStoppingImproved(patience=patience, verbose=True, path=checkpoint_path)\n    \n    best_val_acc = 0\n    \n    for epoch in range(epochs):\n        # Training\n        train_loss, train_acc = training_loop_improved(model, train_loader, optimizer, loss_fn, device)\n        train_losses.append(train_loss)\n        train_accs.append(train_acc)\n        \n        # Validation\n        val_loss, val_acc = validation_loop_improved(model, val_loader, loss_fn, device)\n        val_losses.append(val_loss)\n        val_accs.append(val_acc)\n        \n        # Learning rate scheduling\n        if scheduler is not None:\n            scheduler.step(val_loss)\n            current_lr = optimizer.param_groups[0]['lr']\n        \n        # Early stopping check\n        early_stopping(val_loss, model)\n        \n        # Track best validation accuracy\n        if val_acc > best_val_acc:\n            best_val_acc = val_acc\n        \n        # Progress reporting\n        if epoch % verbose_ct == 0 or epoch == epochs - 1:\n            print(f\"Epoch [{epoch+1:3d}/{epochs}] | \"\n                  f\"Train Loss: {train_loss:.4f} | Train Acc: {train_acc:.2f}% | \"\n                  f\"Val Loss: {val_loss:.4f} | Val Acc: {val_acc:.2f}%\")\n            if scheduler is not None:\n                print(f\"   Learning Rate: {current_lr:.6f}\")\n        \n        if early_stopping.early_stop:\n            print(f\"\\n🛑 Early stopping triggered at epoch {epoch+1}\")\n            print(f\"   Loading best model from checkpoint...\")\n            model.load_state_dict(torch.load(checkpoint_path))\n            break\n    \n    # Save final model\n    torch.save(model.state_dict(), model_path)\n    print(f\"\\n✅ Training completed!\")\n    print(f\"   Best Validation Accuracy: {best_val_acc:.2f}%\")\n    print(f\"   Model saved to: {model_path}\")\n    \n    return train_losses, val_losses, train_accs, val_accs\n\n\ndef evaluate_improved(model, test_loader, device=\"cuda\"):\n    \"\"\"Improved evaluation with detailed metrics\"\"\"\n    model.eval()\n    predictions = []\n    labels = []\n    \n    print(\"🔍 Evaluating model on test set...\")\n    \n    with torch.no_grad():\n        for batch in test_loader:\n            x_batch, y_batch = batch\n            x_batch, y_batch = x_batch.to(device), y_batch.to(device)\n            \n            yhat = model(x_batch)\n            yhat_idx = torch.argmax(yhat, dim=1)\n            \n            predictions.append(yhat_idx.cpu().numpy())\n            labels.append(y_batch.cpu().numpy())\n    \n    y_pred = np.concatenate(predictions, axis=0)\n    y_true = np.concatenate(labels, axis=0)\n    \n    # Calculate accuracy\n    accuracy = 100 * np.sum(y_pred == y_true) / len(y_true)\n    print(f\"✅ Test Accuracy: {accuracy:.2f}%\")\n    \n    return y_pred, y_true\n\n\ndef plot_training_metrics(train_losses, val_losses, train_accs, val_accs, title_prefix=\"ImprovedCNN\"):\n    \"\"\"Plot both loss and accuracy\"\"\"\n    fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(15, 5))\n    \n    # Loss plot\n    ax1.plot(train_losses, label=\"Training Loss\", color=mycol, linewidth=2)\n    ax1.plot(val_losses, label=\"Validation Loss\", color=mycomplcol, linewidth=2)\n    ax1.set_xlabel('Epoch', fontsize=12)\n    ax1.set_ylabel('Loss', fontsize=12)\n    ax1.set_title('Training and Validation Loss', fontsize=14)\n    ax1.legend(loc='best')\n    ax1.grid(True, alpha=0.3)\n    \n    # Accuracy plot\n    ax2.plot(train_accs, label=\"Training Accuracy\", color=mycol, linewidth=2)\n    ax2.plot(val_accs, label=\"Validation Accuracy\", color=mycomplcol, linewidth=2)\n    ax2.set_xlabel('Epoch', fontsize=12)\n    ax2.set_ylabel('Accuracy (%)', fontsize=12)\n    ax2.set_title('Training and Validation Accuracy', fontsize=14)\n    ax2.legend(loc='best')\n    ax2.grid(True, alpha=0.3)\n    \n    plt.tight_layout()\n    plt.savefig(f'{title_prefix}_Training_Metrics.pdf', bbox_inches='tight', dpi=300)\n    plt.show()\n    \n    print(f\"📊 Training metrics plot saved as '{title_prefix}_Training_Metrics.pdf'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T05:27:51.272291Z","iopub.execute_input":"2025-10-13T05:27:51.273054Z","iopub.status.idle":"2025-10-13T05:27:51.301615Z","shell.execute_reply.started":"2025-10-13T05:27:51.27302Z","shell.execute_reply":"2025-10-13T05:27:51.300775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# Load the Dataset for NN training\nnn_data = ImageDataset(active_df, LabEnc, img_size=224, normalize=True, crop=False)\nprint(f\"✅ Dataset loaded with {len(nn_data)} images\")\nprint(f\"   Image shape: {nn_data[0][0].shape}\")\nprint(f\"   Number of classes: {len(artists)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T05:28:38.877717Z","iopub.execute_input":"2025-10-13T05:28:38.878239Z","iopub.status.idle":"2025-10-13T05:37:59.204195Z","shell.execute_reply.started":"2025-10-13T05:28:38.878212Z","shell.execute_reply":"2025-10-13T05:37:59.203291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create Data Loaders with Optimal Batch Size\n# Larger batch size for better GPU utilization and stable gradients\nBATCH_SIZE = 64  \n\ntrain_loader, val_loader, test_loader = DataSplitter(\n    nn_data, \n    ratios=[60, 25, 15], \n    need_val=True, \n    batches=BATCH_SIZE, \n    shuffle=True, \n    seed=42  # Changed to 42 for consistency\n)\n\n# Verify splits\nprint(f\"\\n📊 Data Split Information:\")\nprint(f\"   Training batches: {len(train_loader)} (samples: {len(train_loader)*BATCH_SIZE})\")\nprint(f\"   Validation batches: {len(val_loader)} (samples: {len(val_loader)*BATCH_SIZE})\")\nprint(f\"   Test batches: {len(test_loader)} (samples: {len(test_loader)*BATCH_SIZE})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T05:39:42.511798Z","iopub.execute_input":"2025-10-13T05:39:42.512252Z","iopub.status.idle":"2025-10-13T05:39:44.84716Z","shell.execute_reply.started":"2025-10-13T05:39:42.512232Z","shell.execute_reply":"2025-10-13T05:39:44.84624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nprint(\"=\"*60)\nprint(\"Extracting Deep Features using Pre-trained ResNet18 (24 Artists)\")\nprint(\"=\"*60)\n\n# Explicitly import torchvision models\nfrom torchvision import models as torch_models\n\n# Load pre-trained ResNet18\nresnet_feature_extractor = torch_models.resnet18(pretrained=True)\nresnet_feature_extractor = nn.Sequential(*list(resnet_feature_extractor.children())[:-1])\nresnet_feature_extractor.eval()\nresnet_feature_extractor.to(device)\n\nprint(f\"ResNet18 feature extractor loaded on {device}\")\nprint(\"Output: 512-dimensional features per image\\n\")\n\ndef extract_cnn_features(dataloader, model, device):\n    \"\"\"Extract CNN features from all images in dataloader\"\"\"\n    features_list = []\n    labels_list = []\n    \n    model.eval()\n    with torch.no_grad():\n        for batch_idx, (images, labels) in enumerate(dataloader):\n            images = images.to(device)\n            features = model(images)\n            features = features.view(features.size(0), -1)\n            \n            features_list.append(features.cpu().numpy())\n            labels_list.append(labels.numpy())\n            \n            if (batch_idx + 1) % 20 == 0:\n                print(f\"Processed {(batch_idx + 1) * len(images)} images...\")\n    \n    all_features = np.vstack(features_list)\n    all_labels = np.concatenate(labels_list)\n    return all_features, all_labels\n\n# Extract features\nprint(\"Extracting training features...\")\nX_train_cnn, y_train_cnn = extract_cnn_features(train_loader, resnet_feature_extractor, device)\n\nprint(\"\\nExtracting validation features...\")\nX_val_cnn, y_val_cnn = extract_cnn_features(val_loader, resnet_feature_extractor, device)\n\nprint(\"\\nExtracting test features...\")\nX_test_cnn, y_test_cnn = extract_cnn_features(test_loader, resnet_feature_extractor, device)\n\nprint(\"\\n\" + \"=\"*60)\nprint(f\"Feature extraction completed!\")\nprint(f\"Train: {X_train_cnn.shape}, Val: {X_val_cnn.shape}, Test: {X_test_cnn.shape}\")\nprint(\"=\"*60)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T05:39:53.135049Z","iopub.execute_input":"2025-10-13T05:39:53.135588Z","iopub.status.idle":"2025-10-13T05:40:01.368492Z","shell.execute_reply.started":"2025-10-13T05:39:53.135564Z","shell.execute_reply":"2025-10-13T05:40:01.367604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"try:\n    X_train_cnn\nexcept NameError:\n    print(\"ERROR: ResNet features not found!\")\n    print(\"Please run Part 6 first to extract features using ResNet18\")\n    raise\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T05:40:11.649451Z","iopub.execute_input":"2025-10-13T05:40:11.649734Z","iopub.status.idle":"2025-10-13T05:40:11.653692Z","shell.execute_reply.started":"2025-10-13T05:40:11.649712Z","shell.execute_reply":"2025-10-13T05:40:11.653038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nprint(\"=\"*60)\nprint(\"5-Fold Cross-Validation with Hyperparameter Tuning (24 Artists)\")\nprint(\"=\"*60)\nprint(f\"Started at: {datetime.now().strftime('%H:%M:%S')}\")\n\n# Combine train + val for cross-validation\nX_train_full = np.vstack([X_train_cnn, X_val_cnn])\ny_train_full = np.concatenate([y_train_cnn, y_val_cnn])\nprint(f\"\\nTraining data shape: {X_train_full.shape}\")\nprint(f\"Training labels shape: {y_train_full.shape}\")\n\n# Standardize features\nprint(\"\\nStandardizing features...\")\nscaler_cnn = StandardScaler()\nX_train_full_scaled = scaler_cnn.fit_transform(X_train_full)\nX_test_cnn_scaled = scaler_cnn.transform(X_test_cnn)\nprint(\"✓ Standardization complete\")\n\n# Define parameter grid for manual search with immediate feedback\nparam_grid = {\n    'C': [10, 50, 100],\n    'gamma': ['scale', 0.001, 0.01, 0.1],\n    'kernel': ['rbf']\n}\n\nprint(f\"\\nTesting parameters: {param_grid}\")\ntotal_combinations = len(param_grid['C']) * len(param_grid['gamma'])\nprint(f\"Total combinations: {total_combinations}\")\nprint(f\"Total fits: {total_combinations * 5} (including 5-fold CV)\")\n\n# Manual grid search with immediate progress display\nprint(\"\\n\" + \"=\"*60)\nprint(\"STARTING PARAMETER SEARCH WITH LIVE UPDATES\")\nprint(\"=\"*60)\n\nbest_score = 0\nbest_params = None\nall_results = []\ncombination_num = 0\n\nfor C in param_grid['C']:\n    for gamma in param_grid['gamma']:\n        combination_num += 1\n        start_time = time.time()\n        \n        print(f\"\\n[{combination_num}/{total_combinations}] Testing C={C}, gamma={gamma}\")\n        print(f\"  Time: {datetime.now().strftime('%H:%M:%S')}\", end=\" | \")\n        \n        # Create SVM and perform cross-validation\n        svm_model = SVC(C=C, gamma=gamma, kernel='rbf', \n                       class_weight='balanced', random_state=42)\n        \n        # 5-fold cross-validation\n        cv_scores = cross_val_score(svm_model, X_train_full_scaled, y_train_full, \n                                    cv=5, scoring='accuracy', n_jobs=-1)\n        \n        mean_score = cv_scores.mean()\n        std_score = cv_scores.std()\n        elapsed = time.time() - start_time\n        \n        print(f\"Completed in {elapsed:.1f}s\")\n        print(f\"  CV Accuracy: {mean_score:.4f} ± {std_score:.4f} ({mean_score*100:.2f}%)\", end=\"\")\n        \n        # Track results\n        all_results.append({\n            'C': C,\n            'gamma': gamma,\n            'mean_score': mean_score,\n            'std_score': std_score,\n            'time': elapsed\n        })\n        \n        # Update best if this is better\n        if mean_score > best_score:\n            best_score = mean_score\n            best_params = {'C': C, 'gamma': gamma}\n            print(\"  ⭐ NEW BEST!\")\n        else:\n            print()\n\nprint(\"\\n\" + \"=\"*60)\nprint(\"CROSS-VALIDATION RESULTS\")\nprint(\"=\"*60)\nprint(f\"Best parameters: {best_params}\")\nprint(f\"Best CV accuracy: {best_score:.4f} ({best_score*100:.2f}%)\")\n\n# Display top 5 parameter combinations\nresults_df = pd.DataFrame(all_results)\nresults_df = results_df.sort_values('mean_score', ascending=False)\nprint(\"\\nTop 5 Parameter Combinations:\")\nprint(results_df.head(5)[['C', 'gamma', 'mean_score', 'std_score', 'time']].to_string(index=False))\n\nprint(\"\\nBottom 3 Parameter Combinations:\")\nprint(results_df.tail(3)[['C', 'gamma', 'mean_score', 'std_score']].to_string(index=False))\n\n# Train final model with best parameters\nprint(\"\\n\" + \"=\"*60)\nprint(\"TRAINING FINAL MODEL WITH BEST PARAMETERS...\")\nprint(\"=\"*60)\nsvm_cv = SVC(C=best_params['C'], gamma=best_params['gamma'], kernel='rbf',\n             class_weight='balanced', random_state=42)\nsvm_cv.fit(X_train_full_scaled, y_train_full)\nprint(\"✓ Final model trained\")\n\n# Evaluate on test set\nprint(\"\\nEvaluating on test set...\")\ny_pred_cnn = svm_cv.predict(X_test_cnn_scaled)\ntest_accuracy = accuracy_score(y_test_cnn, y_pred_cnn)\n\nprint(\"\\n\" + \"=\"*60)\nprint(\"FINAL TEST SET RESULTS (24 Artists)\")\nprint(\"=\"*60)\nprint(f\"Test Accuracy: {test_accuracy:.4f} ({test_accuracy*100:.2f}%)\")\nprint(f\"Number of test samples: {len(y_test_cnn)}\")\nprint(f\"Correctly classified: {(y_test_cnn == y_pred_cnn).sum()}\")\nprint(f\"Misclassified: {(y_test_cnn != y_pred_cnn).sum()}\\n\")\n\nprint(classification_report(y_test_cnn, y_pred_cnn, target_names=artists))\n\nprint(\"\\nGenerating confusion matrix...\")\nmatplotlib.rc_file_defaults()\ncfmatrix_cnn = confusion_matrix(y_test_cnn, y_pred_cnn)\nplot_cm(cfmatrix_cnn, 'ResNet+SVM (Best CV Params) - 24 Artists', artists)\nprint(\"✓ Confusion matrix plotted\")\nprint(f\"\\nCompleted at: {datetime.now().strftime('%H:%M:%S')}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T05:40:13.988605Z","iopub.execute_input":"2025-10-13T05:40:13.988889Z","iopub.status.idle":"2025-10-13T05:56:04.738036Z","shell.execute_reply.started":"2025-10-13T05:40:13.988869Z","shell.execute_reply":"2025-10-13T05:56:04.737199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n3️⃣ Saving ResNet + SVM (Best CV) Model...\")\ntry:\n    joblib.dump(svm_cv, '/kaggle/working/saved_models/resnet_svm_cv_model.pkl')\n    joblib.dump(scaler_cnn, '/kaggle/working/saved_models/resnet_scaler.pkl')\n    # Save best parameters for reference\n    with open('/kaggle/working/saved_models/resnet_svm_best_params.txt', 'w') as f:\n        f.write(f\"Best Parameters: {best_params}\\n\")\n        f.write(f\"Best CV Accuracy: {best_score:.4f}\\n\")\n        f.write(f\"Test Accuracy: {test_accuracy:.4f}\\n\")\n    print(\"   ✅ Saved: resnet_svm_cv_model.pkl\")\n    print(\"   ✅ Saved: resnet_scaler.pkl\")\n    print(\"   ✅ Saved: resnet_svm_best_params.txt\")\nexcept Exception as e:\n    print(f\"   ❌ Error: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T05:56:04.739151Z","iopub.execute_input":"2025-10-13T05:56:04.739421Z","iopub.status.idle":"2025-10-13T05:56:04.788425Z","shell.execute_reply.started":"2025-10-13T05:56:04.739404Z","shell.execute_reply":"2025-10-13T05:56:04.787783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract the ResNet+SVM model with clear naming for Bayes Optimal\n'''\nsvm_resnet = svm_cv  # Your SVM trained on ResNet features with best CV params\nX_test_resnet = X_test_cnn_scaled\ny_test_resnet = y_test_cnn\n\nprint(f\"\\n✅ SVM with ResNet Features saved as 'svm_resnet'\")\nprint(f\"   Best params: C={best_params['C']}, gamma={best_params['gamma']}\")\nprint(f\"   Test Accuracy: {test_accuracy*100:.2f}%\")\nprint(f\"   Test set shape: {X_test_resnet.shape}\")\n'''","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### SVM : Combination of above","metadata":{}},{"cell_type":"code","source":"# FEATURE FUSION FOR SVM\nprint(\"\\n\" + \"=\"*70)\nprint(\"FEATURE FUSION: ResNet Features + HOG + Color Histogram\")\nprint(\"=\"*70)\n\ndef extract_resnet_features(model, dataloader, device=\"cuda\"):\n    \"\"\"\n    Extract features from ResNet's penultimate layer\n    (before final classification layer)\n    \"\"\"\n    model.eval()\n    features_list = []\n    labels_list = []\n    \n    # Remove final classification layer to get features\n    feature_extractor = nn.Sequential(*list(model.children())[:-1])\n    feature_extractor = feature_extractor.to(device)\n    \n    print(\"🔍 Extracting ResNet features...\")\n    with torch.no_grad():\n        for batch in dataloader:\n            x_batch, y_batch = batch\n            x_batch = x_batch.to(device)\n            \n            # Extract features (output shape: [batch_size, 512, 1, 1] for ResNet-18)\n            feats = feature_extractor(x_batch)\n            feats = feats.view(feats.size(0), -1)  # Flatten to [batch_size, 512]\n            \n            features_list.append(feats.cpu().numpy())\n            labels_list.append(y_batch.cpu().numpy())\n    \n    features = np.concatenate(features_list, axis=0)\n    labels = np.concatenate(labels_list, axis=0)\n    \n    print(f\"✅ Extracted features shape: {features.shape}\")\n    return features, labels\n\n\ndef extract_color_histogram(image_array, bins=32):\n    \"\"\"\n    Extract color histogram features from image\n    Args:\n        image_array: numpy array (H, W, 3) in range [0, 1]\n        bins: number of bins per channel\n    Returns:\n        Concatenated histogram for all 3 channels\n    \"\"\"\n    # Convert to [0, 255] range\n    image_array = (image_array * 255).astype(np.uint8)\n    \n    hist_features = []\n    for channel in range(3):\n        hist, _ = np.histogram(image_array[:, :, channel], bins=bins, range=(0, 256))\n        hist = hist.astype(np.float32)\n        hist = hist / (hist.sum() + 1e-7)  # Normalize\n        hist_features.append(hist)\n    \n    return np.concatenate(hist_features)\n\n\ndef extract_hog_and_color(dataframe, img_size=224, hog_bins=9):\n    \"\"\"\n    Extract HOG + Color Histogram features from images\n    \"\"\"\n    \n    train_path = '/kaggle/working/train_top24/'\n    test_path = '/kaggle/working/test_top24/'\n    \n    all_hog_features = []\n    all_color_features = []\n    all_labels = []\n    \n    print(f\"🔍 Extracting HOG + Color Histogram features...\")\n    \n    for index, row in tqdm(dataframe.iterrows(), total=len(dataframe), desc=\"Extracting features\"):\n        filename = row['new_filename']\n        \n        # Determine path\n        if row['in_train']:\n            img_path = os.path.join(train_path, filename)\n        else:\n            img_path = os.path.join(test_path, filename)\n        \n        try:\n            # Load image\n            image = Image.open(img_path)\n            datum = image_transformer_nn(image, apply_norm=False, crop_img=False, new_dim=img_size)\n            np_datum = datum.numpy().transpose(1, 2, 0)  # (H, W, C)\n            \n            # Extract HOG features\n            hog_feat = hog(np_datum, orientations=hog_bins, pixels_per_cell=(16, 16),\n                          cells_per_block=(2, 2), feature_vector=True, channel_axis=2)\n            all_hog_features.append(hog_feat)\n            \n            # Extract Color Histogram features\n            color_feat = extract_color_histogram(np_datum, bins=32)\n            all_color_features.append(color_feat)\n            \n            # Label\n            artist = row['artist']\n            label = LabEnc.transform([artist])[0]\n            all_labels.append(label)\n            \n        except Exception as e:\n            print(f\"❌ Error loading {filename}: {e}\")\n    \n    hog_features = np.array(all_hog_features)\n    color_features = np.array(all_color_features)\n    labels = np.array(all_labels)\n    \n    print(f\"✅ HOG features shape: {hog_features.shape}\")\n    print(f\"✅ Color features shape: {color_features.shape}\")\n    \n    return hog_features, color_features, labels\n\n\n# Step 1: Extract ResNet features from all data\nprint(\"\\n📊 Step 1: Extracting ResNet features...\")\n\n# Create a DataLoader for all data (without splitting) to extract features\nall_data_loader = DataLoader(nn_data, batch_size=64, shuffle=False)\nresnet_features, resnet_labels = extract_resnet_features(model_resnet, all_data_loader, device=device)\n\n# Step 2: Extract HOG + Color Histogram features\nprint(\"\\n📊 Step 2: Extracting HOG + Color Histogram features...\")\nhog_features, color_features, traditional_labels = extract_hog_and_color(active_df, img_size=224, hog_bins=9)\n\n# Verify labels match\nassert np.array_equal(resnet_labels, traditional_labels), \"Labels don't match!\"\n\n# Step 3: Fuse all features\nprint(\"\\n📊 Step 3: Fusing all features...\")\n\n# Normalize features (important for fusion!)\n\nscaler_resnet = StandardScaler()\nscaler_hog = StandardScaler()\nscaler_color = StandardScaler()\n\nresnet_features_norm = scaler_resnet.fit_transform(resnet_features)\nhog_features_norm = scaler_hog.fit_transform(hog_features)\ncolor_features_norm = scaler_color.fit_transform(color_features)\n\n# Concatenate all features\nfused_features = np.concatenate([\n    resnet_features_norm,  # 512 features\n    hog_features_norm,     # ~1000+ features\n    color_features_norm    # 96 features (32 bins × 3 channels)\n], axis=1)\n\nprint(f\"✅ Fused feature shape: {fused_features.shape}\")\nprint(f\"   - ResNet features: {resnet_features_norm.shape[1]}\")\nprint(f\"   - HOG features: {hog_features_norm.shape[1]}\")\nprint(f\"   - Color features: {color_features_norm.shape[1]}\")\n\n# Step 4: Split data for SVM training\nprint(\"\\n📊 Step 4: Splitting data for SVM...\")\n\nX_train, X_val, X_test, y_train, y_val, y_test = DataSplitter(\n    (fused_features, resnet_labels),\n    ratios=[60, 25, 15],\n    need_val=True,\n    shuffle=True,\n    seed=42\n)\n\nprint(f\"   Train set: {X_train.shape}\")\nprint(f\"   Val set: {X_val.shape}\")\nprint(f\"   Test set: {X_test.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T05:57:46.408686Z","iopub.execute_input":"2025-10-13T05:57:46.408957Z","iopub.status.idle":"2025-10-13T06:07:19.69381Z","shell.execute_reply.started":"2025-10-13T05:57:46.408936Z","shell.execute_reply":"2025-10-13T06:07:19.692954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.svm import SVC","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T12:50:47.134354Z","iopub.execute_input":"2025-10-13T12:50:47.134686Z","iopub.status.idle":"2025-10-13T12:50:47.139837Z","shell.execute_reply.started":"2025-10-13T12:50:47.134665Z","shell.execute_reply":"2025-10-13T12:50:47.138781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\nprint(\"\\n\" + \"=\"*70)\nprint(\"TRAINING SVM WITH FUSED FEATURES\")\nprint(\"=\"*70)\n\n# Try different SVM kernels and find the best one\nprint(\"\\n🔍 Testing different SVM configurations...\")\n\nsvm_results = {}\n\n# 1. Linear SVM (fastest)\nprint(\"\\n1️⃣ Training Linear SVM...\")\nsvm_linear = SVC(kernel='linear', C=1.0, random_state=42, verbose=True)\nsvm_linear.fit(X_train, y_train)\ny_pred_linear = svm_linear.predict(X_test)\nacc_linear = accuracy_score(y_test, y_pred_linear) * 100\nsvm_results['Linear'] = acc_linear\nprint(f\"   ✅ Linear SVM Test Accuracy: {acc_linear:.2f}%\")\n\n# 2. RBF SVM (most common, usually best)\nprint(\"\\n2️⃣ Training RBF SVM...\")\nsvm_rbf = SVC(kernel='rbf', C=10.0, gamma='scale', random_state=42, verbose=True)\nsvm_rbf.fit(X_train, y_train)\ny_pred_rbf = svm_rbf.predict(X_test)\nacc_rbf = accuracy_score(y_test, y_pred_rbf) * 100\nsvm_results['RBF'] = acc_rbf\nprint(f\"   ✅ RBF SVM Test Accuracy: {acc_rbf:.2f}%\")\n\n# 3. Polynomial SVM\nprint(\"\\n3️⃣ Training Polynomial SVM...\")\nsvm_poly = SVC(kernel='poly', degree=3, C=1.0, random_state=42, verbose=True)\nsvm_poly.fit(X_train, y_train)\ny_pred_poly = svm_poly.predict(X_test)\nacc_poly = accuracy_score(y_test, y_pred_poly) * 100\nsvm_results['Polynomial'] = acc_poly\nprint(f\"   ✅ Polynomial SVM Test Accuracy: {acc_poly:.2f}%\")\n\n# Find best SVM\nbest_svm_name = max(svm_results, key=svm_results.get)\nbest_svm_acc = svm_results[best_svm_name]\n\nprint(f\"\\n{'='*70}\")\nprint(f\"BEST SVM: {best_svm_name} with {best_svm_acc:.2f}% accuracy\")\nprint(f\"{'='*70}\")\n\n# Use best SVM for final evaluation\nif best_svm_name == 'Linear':\n    best_svm = svm_linear\n    y_pred_best = y_pred_linear\nelif best_svm_name == 'RBF':\n    best_svm = svm_rbf\n    y_pred_best = y_pred_rbf\nelse:\n    best_svm = svm_poly\n    y_pred_best = y_pred_poly\n\n# Classification report\nprint(f\"\\n📊 Classification Report (SVM - {best_svm_name} Kernel):\")\nprint(classification_report(y_test, y_pred_best, target_names=artists))\n\n# Confusion matrix\nmatplotlib.rc_file_defaults()\ncfmatrix_svm = confusion_matrix(y_test, y_pred_best)\nplot_cm(cfmatrix_svm, f'SVM_FusedFeatures_{best_svm_name}_Confusion_Matrix', artists)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-13T12:50:48.640181Z","iopub.execute_input":"2025-10-13T12:50:48.640489Z","iopub.status.idle":"2025-10-13T12:50:48.653244Z","shell.execute_reply.started":"2025-10-13T12:50:48.640466Z","shell.execute_reply":"2025-10-13T12:50:48.652519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n4️⃣ Saving Fused Features SVM Models...\")\ntry:\n    # Save all three SVM variants\n    joblib.dump(svm_linear, '/kaggle/working/saved_models/fused_svm_linear.pkl')\n    joblib.dump(svm_rbf, '/kaggle/working/saved_models/fused_svm_rbf.pkl')\n    joblib.dump(svm_poly, '/kaggle/working/saved_models/fused_svm_poly.pkl')\n    \n    # Save the best one separately for easy access\n    joblib.dump(best_svm, '/kaggle/working/saved_models/fused_svm_best.pkl')\n    \n    # Save all three scalers used for fusion\n    joblib.dump(scaler_resnet, '/kaggle/working/saved_models/fused_scaler_resnet.pkl')\n    joblib.dump(scaler_hog, '/kaggle/working/saved_models/fused_scaler_hog.pkl')\n    joblib.dump(scaler_color, '/kaggle/working/saved_models/fused_scaler_color.pkl')\n    \n    # Save results summary\n    with open('/kaggle/working/saved_models/fused_svm_results.txt', 'w') as f:\n        f.write(\"SVM Results Summary:\\n\")\n        f.write(\"=\"*50 + \"\\n\")\n        for kernel, acc in svm_results.items():\n            marker = \" ⭐ BEST\" if kernel == best_svm_name else \"\"\n            f.write(f\"{kernel}: {acc:.2f}%{marker}\\n\")\n    \n    print(\"   ✅ Saved: fused_svm_linear.pkl\")\n    print(\"   ✅ Saved: fused_svm_rbf.pkl\")\n    print(\"   ✅ Saved: fused_svm_poly.pkl\")\n    print(\"   ✅ Saved: fused_svm_best.pkl\")\n    print(\"   ✅ Saved: All feature scalers (resnet, hog, color)\")\n    print(\"   ✅ Saved: fused_svm_results.txt\")\nexcept Exception as e:\n    print(f\"   ❌ Error: {e}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n5️⃣ Saving Label Encoder...\")\ntry:\n    joblib.dump(LabEnc, '/kaggle/working/saved_models/label_encoder.pkl')\n    # Also save artist names as a text file for reference\n    with open('/kaggle/working/saved_models/artists_list.txt', 'w') as f:\n        for i, artist in enumerate(artists):\n            f.write(f\"{i}: {artist}\\n\")\n    print(\"   ✅ Saved: label_encoder.pkl\")\n    print(\"   ✅ Saved: artists_list.txt\")\nexcept Exception as e:\n    print(f\"   ❌ Error: {e}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n\" + \"=\"*70)\nprint(\"SUMMARY - ALL MODELS SAVED\")\nprint(\"=\"*70)\nprint(\"\\nSaved models:\")\nprint(\"  1. HOG + SVM\")\nprint(\"  2. Color Histogram + SVM (OvR)\")\nprint(\"  3. ResNet + SVM (with CV tuning)\")\nprint(\"  4. Fused Features + SVM (Linear, RBF, Poly)\")\nprint(\"  5. Label Encoder + Artist List\")\nprint(f\"\\nLocation: /kaggle/working/saved_models/\")\nprint(\"\\nTo download: Click on the folder icon on the left → saved_models\")\nprint(\"=\"*70)\n\n# Optional: Create a zip file for easy download\nprint(\"\\n📦 Creating zip file for easy download...\")\nimport shutil\ntry:\n    shutil.make_archive('/kaggle/working/all_svm_models', 'zip', '/kaggle/working/saved_models')\n    print(\"   ✅ Created: all_svm_models.zip\")\n    print(\"   Download this single file to get all models!\")\nexcept Exception as e:\n    print(f\"   ❌ Error creating zip: {e}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Quick extraction - no retraining needed\n'''\nsvm_fused = best_svm  # Your already-trained best SVM\nX_test_fused = X_test\ny_test_fused = y_test\n\nprint(f\"✅ Variables ready for Bayes Optimal Classifier:\")\nprint(f\"   svm_fused: {best_svm_name} kernel\")\nprint(f\"   X_test_fused: {X_test_fused.shape}\")\nprint(f\"   y_test_fused: {y_test_fused.shape}\")\n'''","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### SVM Models Comaparision","metadata":{}},{"cell_type":"code","source":"# SVM Models Comparison\nprint(\"\\n\" + \"=\"*80)\nprint(\"SVM MODELS COMPARISON - COMPREHENSIVE ANALYSIS\")\nprint(\"=\"*80)\n\n# Collect all SVM model results\nsvm_results = {\n    'SVM + HOG Features': {\n        'accuracy': accuracy_score(y_test, y_pred) * 100,\n        'predictions': y_pred,\n        'labels': y_test,\n        'features': 'HOG',\n        'kernel': 'RBF'\n    },\n    'SVM + Color Histogram': {\n        'accuracy': accuracy_score(y_test_color, y_pred_color) * 100,\n        'predictions': y_pred_color,\n        'labels': y_test_color,\n        'features': 'Color Histogram',\n        'kernel': 'RBF (OvR)'\n    },\n    'SVM + ResNet Features': {\n        'accuracy': accuracy_score(y_test_cnn, y_pred_cnn) * 100,\n        'predictions': y_pred_cnn,\n        'labels': y_test_cnn,\n        'features': 'ResNet-18 Features',\n        'kernel': 'RBF (CV Optimized)'\n    },\n    'SVM + Fused Features': {\n        'accuracy': accuracy_score(y_test, y_pred_best) * 100,\n        'predictions': y_pred_best,\n        'labels': y_test,\n        'features': 'ResNet + HOG + Color',\n        'kernel': best_svm_name\n    }\n}\n\n# Create comparison DataFrame\nsvm_comparison_data = []\nfor model_name, results in svm_results.items():\n    svm_comparison_data.append({\n        'Model': model_name,\n        'Test Accuracy (%)': results['accuracy'],\n        'Feature Type': results['features'],\n        'Kernel': results['kernel'],\n        'Correct Predictions': (results['predictions'] == results['labels']).sum(),\n        'Total Samples': len(results['labels']),\n        'Misclassified': (results['predictions'] != results['labels']).sum()\n    })\n\nsvm_comparison_df = pd.DataFrame(svm_comparison_data)\nsvm_comparison_df = svm_comparison_df.sort_values('Test Accuracy (%)', ascending=False)\n\nprint(\"\\n📊 SVM MODELS PERFORMANCE COMPARISON:\")\nprint(\"=\"*70)\nprint(svm_comparison_df.to_string(index=False))\n\n# Visualizations\nfig, ((ax1, ax2), (ax3, ax4)) = plt.subplots(2, 2, figsize=(20, 15))\n\n# 1. SVM Accuracy Comparison Bar Chart\nsvm_models = svm_comparison_df['Model']\nsvm_accuracies = svm_comparison_df['Test Accuracy (%)']\nsvm_colors = ['#E74C3C', '#F39C12', '#3498DB', '#2ECC71']\n\nbars = ax1.bar(range(len(svm_models)), svm_accuracies, color=svm_colors, alpha=0.8, edgecolor='black', linewidth=1)\nax1.set_xlabel('SVM Models', fontsize=12, fontweight='bold')\nax1.set_ylabel('Test Accuracy (%)', fontsize=12, fontweight='bold')\nax1.set_title('SVM Models - Test Accuracy Comparison', fontsize=14, fontweight='bold')\nax1.set_xticks(range(len(svm_models)))\nax1.set_xticklabels([m.replace(' ', '\\n') for m in svm_models], rotation=0, ha='center')\nax1.grid(True, alpha=0.3, axis='y')\nax1.set_ylim(0, max(svm_accuracies) * 1.1)\n\n# Add value labels on bars\nfor i, (bar, acc) in enumerate(zip(bars, svm_accuracies)):\n    ax1.text(bar.get_x() + bar.get_width()/2, bar.get_height() + 0.5, \n             f'{acc:.2f}%', ha='center', va='bottom', fontweight='bold', fontsize=10)\n\n# 2. Feature Type vs Performance\nfeature_types = svm_comparison_df['Feature Type']\nfeature_performance = {}\nfor i, feature in enumerate(feature_types):\n    if feature not in feature_performance:\n        feature_performance[feature] = []\n    feature_performance[feature].append(svm_accuracies.iloc[i])\n\n# Calculate average performance per feature type\nfeature_avg = {ft: np.mean(perf) for ft, perf in feature_performance.items()}\n\nbars = ax2.bar(range(len(feature_avg)), list(feature_avg.values()), \n               color=['#FF6B6B', '#4ECDC4', '#45B7D1', '#96CEB4'], alpha=0.8, edgecolor='black', linewidth=1)\nax2.set_xlabel('Feature Types', fontsize=12, fontweight='bold')\nax2.set_ylabel('Average Test Accuracy (%)', fontsize=12, fontweight='bold')\nax2.set_title('SVM Models - Feature Type Performance', fontsize=14, fontweight='bold')\nax2.set_xticks(range(len(feature_avg)))\nax2.set_xticklabels([ft.replace(' ', '\\n') for ft in feature_avg.keys()], rotation=0, ha='center')\nax2.grid(True, alpha=0.3, axis='y')\n\n# Add value labels\nfor i, (bar, acc) in enumerate(zip(bars, list(feature_avg.values()))):\n    ax2.text(bar.get_x() + bar.get_width()/2, bar.get_height() + 0.5, \n             f'{acc:.2f}%', ha='center', va='bottom', fontweight='bold', fontsize=10)\n\n# 3. Feature Complexity vs Performance\nfeature_complexity = {\n    'HOG': 1,\n    'Color Histogram': 2,\n    'ResNet-18 Features': 3,\n    'ResNet + HOG + Color': 4\n}\n\ncomplexity_scores_svm = [feature_complexity[ft] for ft in feature_types]\nscatter = ax3.scatter(complexity_scores_svm, svm_accuracies, s=200, c=svm_colors, alpha=0.7, edgecolors='black', linewidth=2)\n\nax3.set_xlabel('Feature Complexity (Relative)', fontsize=12, fontweight='bold')\nax3.set_ylabel('Test Accuracy (%)', fontsize=12, fontweight='bold')\nax3.set_title('SVM Models - Feature Complexity vs Performance', fontsize=14, fontweight='bold')\nax3.grid(True, alpha=0.3)\n\n# Add model labels to scatter points\nfor i, model in enumerate(svm_models):\n    ax3.annotate(model.replace(' ', '\\n'), (complexity_scores_svm[i], svm_accuracies.iloc[i]), \n                xytext=(5, 5), textcoords='offset points', fontsize=9, \n                bbox=dict(boxstyle='round,pad=0.3', facecolor='white', alpha=0.7))\n\n# 4. Correct vs Misclassified Predictions for SVM\nsvm_correct_preds = svm_comparison_df['Correct Predictions']\nsvm_misclassified = svm_comparison_df['Misclassified']\n\nx = np.arange(len(svm_models))\nwidth = 0.35\n\nbars1 = ax4.bar(x - width/2, svm_correct_preds, width, label='Correct', color='#2ECC71', alpha=0.8)\nbars2 = ax4.bar(x + width/2, svm_misclassified, width, label='Misclassified', color='#E74C3C', alpha=0.8)\n\nax4.set_xlabel('SVM Models', fontsize=12, fontweight='bold')\nax4.set_ylabel('Number of Predictions', fontsize=12, fontweight='bold')\nax4.set_title('SVM Models - Correct vs Misclassified Predictions', fontsize=14, fontweight='bold')\nax4.set_xticks(x)\nax4.set_xticklabels([m.replace(' ', '\\n') for m in svm_models], rotation=0, ha='center')\nax4.legend()\nax4.grid(True, alpha=0.3, axis='y')\n\nplt.tight_layout()\nplt.savefig('SVM_Models_Comprehensive_Comparison.pdf', bbox_inches='tight', dpi=300)\nplt.show()\n\n# Detailed Analysis\nprint(\"\\n\" + \"=\"*80)\nprint(\"DETAILED SVM MODELS ANALYSIS\")\nprint(\"=\"*80)\n\nbest_svm_model = svm_comparison_df.iloc[0]['Model']\nbest_svm_accuracy = svm_comparison_df.iloc[0]['Test Accuracy (%)']\n\nprint(f\"\\n🏆 BEST PERFORMING SVM MODEL: {best_svm_model}\")\nprint(f\"   Test Accuracy: {best_svm_accuracy:.2f}%\")\nprint(f\"   Feature Type: {svm_comparison_df.iloc[0]['Feature Type']}\")\nprint(f\"   Kernel: {svm_comparison_df.iloc[0]['Kernel']}\")\nprint(f\"   Correct Predictions: {svm_comparison_df.iloc[0]['Correct Predictions']}\")\nprint(f\"   Misclassified: {svm_comparison_df.iloc[0]['Misclassified']}\")\n\nprint(f\"\\n📈 KEY INSIGHTS:\")\nprint(f\"   1. Deep features (ResNet) significantly outperform traditional features (HOG, Color)\")\nprint(f\"   2. Feature fusion provides the best performance by combining multiple feature types\")\nprint(f\"   3. HOG features show moderate performance for artistic style classification\")\nprint(f\"   4. Color histogram alone provides baseline performance but lacks discriminative power\")\n\n# Feature importance analysis\nprint(f\"\\n📊 FEATURE IMPORTANCE ANALYSIS:\")\nprint(f\"   1. ResNet Features: {feature_avg.get('ResNet-18 Features', 0):.2f}% avg accuracy\")\nprint(f\"   2. Fused Features: {feature_avg.get('ResNet + HOG + Color', 0):.2f}% avg accuracy\")\nprint(f\"   3. HOG Features: {feature_avg.get('HOG', 0):.2f}% avg accuracy\")\nprint(f\"   4. Color Features: {feature_avg.get('Color Histogram', 0):.2f}% avg accuracy\")\n\nprint(f\"\\n✅ SVM Models Comparison completed and saved as 'SVM_Models_Comprehensive_Comparison.pdf'\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}