{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"import os, json\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nimport hashlib\nfrom PIL import Image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T00:18:15.900312Z","iopub.execute_input":"2026-07-04T00:18:15.900572Z","iopub.status.idle":"2026-07-04T00:18:19.367703Z","shell.execute_reply.started":"2026-07-04T00:18:15.900546Z","shell.execute_reply":"2026-07-04T00:18:19.366906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set base directrory\nbase_dir = \"/kaggle/input/competitions/cassava-leaf-disease-classification\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T00:18:31.551185Z","iopub.execute_input":"2026-07-04T00:18:31.551520Z","iopub.status.idle":"2026-07-04T00:18:31.556094Z","shell.execute_reply.started":"2026-07-04T00:18:31.551474Z","shell.execute_reply":"2026-07-04T00:18:31.555244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(\"/kaggle/input/competitions/cassava-leaf-disease-classification/label_num_to_disease_map.json\") as file:\n    print(\"yes\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T00:18:32.108444Z","iopub.execute_input":"2026-07-04T00:18:32.108773Z","iopub.status.idle":"2026-07-04T00:18:32.114614Z","shell.execute_reply.started":"2026-07-04T00:18:32.108743Z","shell.execute_reply":"2026-07-04T00:18:32.113883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# load and inspect label map\nwith open(os.path.join(base_dir, \"label_num_to_disease_map.json\")) as file:\n    map_classes = json.loads(file.read())\n    map_classes = {int(k): v for k, v in map_classes.items()}\n\n# Display the mapping\nprint(\"Class Mapping : \")\nprint(json.dumps(map_classes, indent=4))\n\n# Here there are 5 classes where shown in bellow.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T00:18:35.500029Z","iopub.execute_input":"2026-07-04T00:18:35.500361Z","iopub.status.idle":"2026-07-04T00:18:35.510390Z","shell.execute_reply.started":"2026-07-04T00:18:35.500332Z","shell.execute_reply":"2026-07-04T00:18:35.509551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"input_files = os.listdir(os.path.join(base_dir, \"train_images\"))\nprint(f\"Number of train images : {len(input_files)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T00:18:41.110590Z","iopub.execute_input":"2026-07-04T00:18:41.111598Z","iopub.status.idle":"2026-07-04T00:18:41.124354Z","shell.execute_reply.started":"2026-07-04T00:18:41.111556Z","shell.execute_reply":"2026-07-04T00:18:41.123454Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# step -3 : load train.csv and add a human-readable class name based on the mapping\ndf_train = pd.read_csv(os.path.join(base_dir, \"train.csv\"))\ndf_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T00:20:33.911308Z","iopub.execute_input":"2026-07-04T00:20:33.912158Z","iopub.status.idle":"2026-07-04T00:20:33.977996Z","shell.execute_reply.started":"2026-07-04T00:20:33.912102Z","shell.execute_reply":"2026-07-04T00:20:33.976813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train[\"class_name\"] = df_train[\"label\"].map(map_classes)\ndf_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T00:21:49.718788Z","iopub.execute_input":"2026-07-04T00:21:49.719152Z","iopub.status.idle":"2026-07-04T00:21:49.734507Z","shell.execute_reply.started":"2026-07-04T00:21:49.719123Z","shell.execute_reply":"2026-07-04T00:21:49.733721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train[\"class_name\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T00:23:11.186739Z","iopub.execute_input":"2026-07-04T00:23:11.187120Z","iopub.status.idle":"2026-07-04T00:23:11.200128Z","shell.execute_reply.started":"2026-07-04T00:23:11.187092Z","shell.execute_reply":"2026-07-04T00:23:11.199225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check distribution \nclass_distribution = df_train[\"class_name\"].value_counts()\n\n# plot the distribution\nplt.figure(figsize=(10,6))\nclass_distribution.plot(kind=\"bar\")\nplt.title(\"class Distribution of Cassava Leaf Disese\")\nplt.ylabel(\"Number of Images\")\nplt.xlabel(\"Diesease Class\")\nplt.xticks(rotation=45)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T00:27:12.158785Z","iopub.execute_input":"2026-07-04T00:27:12.159691Z","iopub.status.idle":"2026-07-04T00:27:12.492904Z","shell.execute_reply.started":"2026-07-04T00:27:12.159658Z","shell.execute_reply":"2026-07-04T00:27:12.491986Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"When we use techniques like transfer learning, you dont need to balance the data, Internally model will handle. (In pre trained model)\n\nbut when you build the model scratch then we need to somthing to do for Imbalance data","metadata":{}},{"cell_type":"code","source":"# countplot visualization\nplt.figure(figsize=(8,4))\nsns.countplot(y=\"class_name\", data=df_train)\nplt.title(\"Class Distribution (seaborn)\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T00:31:17.761570Z","iopub.execute_input":"2026-07-04T00:31:17.761934Z","iopub.status.idle":"2026-07-04T00:31:17.977654Z","shell.execute_reply.started":"2026-07-04T00:31:17.761893Z","shell.execute_reply":"2026-07-04T00:31:17.976905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Show data info and summary statistics\nprint(\"Dataset Info : \")\nprint(df_train.info())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T00:33:48.993284Z","iopub.execute_input":"2026-07-04T00:33:48.993600Z","iopub.status.idle":"2026-07-04T00:33:49.008947Z","shell.execute_reply.started":"2026-07-04T00:33:48.993572Z","shell.execute_reply":"2026-07-04T00:33:49.008097Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Image\n\nLets check the shape of images.\n\nImages - 50*50 (Very small pixels)\n\nIt doesn't make any sense to upscale to 224*224 (we mess up information)\n\nIf images 600*600 resize to may be 224*224\n\n500*500\n\nImage size distribution you need to tune you resize image size","metadata":{}},{"cell_type":"code","source":"# Analyze image shapes (size diemensions) for a sample of 1000 images\n# Dictionary to store image shapes and their counts\nimg_shapes = {}\n\nfor image_name in os.listdir(os.path.join(base_dir, \"train_images\"))[:1000]:\n    image = cv2.imread(os.path.join(base_dir, \"train_images\", image_name))\n    img_shapes[image.shape] = img_shapes.get(image.shape, 0) + 1\n\n# dispay image shape\nprint(\"Sample Image shapes and their frequencies (from 1000 images):\")\nprint(img_shapes)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T00:49:39.569272Z","iopub.execute_input":"2026-07-04T00:49:39.569601Z","iopub.status.idle":"2026-07-04T00:50:38.420926Z","shell.execute_reply.started":"2026-07-04T00:49:39.569572Z","shell.execute_reply":"2026-07-04T00:50:38.420061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nfrom PIL import Image\nfrom collections import Counter\nfrom tqdm import tqdm\n\n# Dataset Path\nimage_folder = \"/kaggle/input/competitions/cassava-leaf-disease-classification/train_images\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T01:08:00.012292Z","iopub.execute_input":"2026-07-04T01:08:00.013021Z","iopub.status.idle":"2026-07-04T01:08:00.027542Z","shell.execute_reply.started":"2026-07-04T01:08:00.012969Z","shell.execute_reply":"2026-07-04T01:08:00.026633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check Image Formats\nformats = Counter()\n\nfor image_name in tqdm(os.listdir(image_folder), desc=\"Checking Image Formats\"):\n\n    image_path = os.path.join(image_folder, image_name)\n\n    try:\n        with Image.open(image_path) as img:\n            formats[img.format] += 1\n    except Exception:\n        pass\n\nprint(\"Image Formats\")\nprint(\"-\" * 30)\n\nfor fmt, count in formats.items():\n    print(f\"{fmt}: {count}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T01:08:20.250796Z","iopub.execute_input":"2026-07-04T01:08:20.251249Z","iopub.status.idle":"2026-07-04T01:08:32.641833Z","shell.execute_reply.started":"2026-07-04T01:08:20.251216Z","shell.execute_reply":"2026-07-04T01:08:32.641001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check Corrupted Images\ncorrupted_images = []\n\nfor image_name in tqdm(os.listdir(image_folder), desc=\"Checking Corrupted Images\"):\n\n    image_path = os.path.join(image_folder, image_name)\n\n    try:\n        with Image.open(image_path) as img:\n            img.verify()\n\n    except Exception:\n        corrupted_images.append(image_name)\n\nprint(f\"Total Corrupted Images : {len(corrupted_images)}\")\n\nif corrupted_images:\n    print(corrupted_images[:10])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T01:08:43.769507Z","iopub.execute_input":"2026-07-04T01:08:43.770337Z","iopub.status.idle":"2026-07-04T01:08:56.451162Z","shell.execute_reply.started":"2026-07-04T01:08:43.770304Z","shell.execute_reply":"2026-07-04T01:08:56.450322Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Brightness Distribution\nbrightness = []\n\nfor image_name in tqdm(os.listdir(image_folder), desc=\"Calculating Brightness\"):\n\n    image_path = os.path.join(image_folder, image_name)\n\n    image = cv2.imread(image_path)\n\n    gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n\n    brightness.append(gray.mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T01:10:07.271791Z","iopub.execute_input":"2026-07-04T01:10:07.272500Z","iopub.status.idle":"2026-07-04T01:12:17.145762Z","shell.execute_reply.started":"2026-07-04T01:10:07.272464Z","shell.execute_reply":"2026-07-04T01:12:17.144962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot Brightness Distribution\nplt.figure(figsize=(10,5))\n\nplt.hist(brightness,\n         bins=30,\n         edgecolor=\"black\")\n\nplt.title(\"Brightness Distribution\")\nplt.xlabel(\"Average Brightness\")\nplt.ylabel(\"Number of Images\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T01:12:26.326530Z","iopub.execute_input":"2026-07-04T01:12:26.327239Z","iopub.status.idle":"2026-07-04T01:12:26.562474Z","shell.execute_reply.started":"2026-07-04T01:12:26.327207Z","shell.execute_reply":"2026-07-04T01:12:26.561590Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Blur Detection\ndef variance_of_laplacian(image):\n    return cv2.Laplacian(image, cv2.CV_64F).var()\n\n\nblur_scores = []\n\nfor image_name in tqdm(os.listdir(image_folder), desc=\"Calculating Blur Score\"):\n\n    image_path = os.path.join(image_folder, image_name)\n\n    gray = cv2.imread(image_path, cv2.IMREAD_GRAYSCALE)\n\n    blur_scores.append(variance_of_laplacian(gray))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T01:13:17.026687Z","iopub.execute_input":"2026-07-04T01:13:17.027289Z","iopub.status.idle":"2026-07-04T01:17:34.447081Z","shell.execute_reply.started":"2026-07-04T01:13:17.027257Z","shell.execute_reply":"2026-07-04T01:17:34.446216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot Blur Distribution\nplt.figure(figsize=(10,5))\n\nplt.hist(blur_scores,\n         bins=30,\n         edgecolor=\"black\")\n\nplt.title(\"Blur Score Distribution\")\nplt.xlabel(\"Variance of Laplacian\")\nplt.ylabel(\"Number of Images\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T01:18:48.943655Z","iopub.execute_input":"2026-07-04T01:18:48.943978Z","iopub.status.idle":"2026-07-04T01:18:49.319353Z","shell.execute_reply.started":"2026-07-04T01:18:48.943950Z","shell.execute_reply":"2026-07-04T01:18:49.318479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"threshold = 100\n\nblurry_images = []\n\nfor image_name in tqdm(os.listdir(image_folder), desc=\"Finding Blurry Images\"):\n\n    image_path = os.path.join(image_folder, image_name)\n\n    gray = cv2.imread(image_path, cv2.IMREAD_GRAYSCALE)\n\n    score = variance_of_laplacian(gray)\n\n    if score < threshold:\n        blurry_images.append((image_name, score))\n\nprint(f\"Total Blurry Images : {len(blurry_images)}\")\n\nprint(\"\\nFirst 10 Blurry Images:\")\n\nfor img_name, score in blurry_images[:10]:\n    print(f\"{img_name} --> {score:.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T01:25:39.839733Z","iopub.execute_input":"2026-07-04T01:25:39.840116Z","iopub.status.idle":"2026-07-04T01:26:57.753664Z","shell.execute_reply.started":"2026-07-04T01:25:39.840087Z","shell.execute_reply":"2026-07-04T01:26:57.752574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_sample_images(class_id, num_images=9):\n    \"\"\"\n    PLot sample images from a specific class in a 3x3 grid\n\n    parameters : \n    class_id (int): The class label to filter images.\n    num_images (int): The no. of images to plot\n    \"\"\"\n    # Filter images for specified class\n    class_images = df_train[df_train[\"label\"] == class_id]\n    num_images = min(len(class_images), num_images) # adjust if fewer images then \n\n    plt.figure(figsize=(10, 10))\n    images = class_images.sample(num_images)\n\n    # plot images on 3x3 grid\n    for i, (_, row) in enumerate(images.iterrows()):\n        img_path = os.path.join(base_dir, \"train_images\", row[\"image_id\"])\n        img = Image.open(img_path)\n        plt.subplot(3, 3, i+1)\n        plt.imshow(img)\n        plt.title(map_classes[class_id]) # use class name for the title\n        plt.axis(\"off\") # hide axis for better visualization\n\n    plt.tight_layout()\n    plt.show","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T01:43:39.437582Z","iopub.execute_input":"2026-07-04T01:43:39.437909Z","iopub.status.idle":"2026-07-04T01:43:39.444842Z","shell.execute_reply.started":"2026-07-04T01:43:39.437883Z","shell.execute_reply":"2026-07-04T01:43:39.443919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_sample_images(0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-04T01:44:10.624470Z","iopub.execute_input":"2026-07-04T01:44:10.624882Z","iopub.status.idle":"2026-07-04T01:44:12.178865Z","shell.execute_reply.started":"2026-07-04T01:44:10.624851Z","shell.execute_reply":"2026-07-04T01:44:12.177860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}