{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":5048,"databundleVersionId":868335,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-07T17:00:54.384150Z","iopub.execute_input":"2025-09-07T17:00:54.384588Z","iopub.status.idle":"2025-09-07T17:00:54.636944Z","shell.execute_reply.started":"2025-09-07T17:00:54.384556Z","shell.execute_reply":"2025-09-07T17:00:54.636164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport random\nfrom tensorflow.keras.preprocessing import image\n\n# Path to training data\ntrain_dir = \"/kaggle/input/state-farm-distracted-driver-detection/imgs/train\"   \n\nimg_size = (224, 224)   # resize for MobileNet or CNN\nx = []\ny = []\n\nclass_names = sorted(os.listdir(train_dir))\n\n# Loop through each class folder\nfor class_name in class_names:\n    class_path = os.path.join(train_dir, class_name)\n    if os.path.isdir(class_path):\n        img_files = [f for f in os.listdir(class_path) \n                     if f.lower().endswith(('.jpg', '.png', '.jpeg'))]\n\n        # Shuffle images randomly\n        random.shuffle(img_files)\n\n        # Drop 20% of images randomly\n        keep_count = int(0.65 * len(img_files))  # keep 80%\n        keep_files = img_files[:keep_count]\n\n        for img_file in keep_files:\n            img_path = os.path.join(class_path, img_file)\n            try:\n                # Load and preprocess image\n                img = image.load_img(img_path, target_size=img_size)\n                img_array = image.img_to_array(img) / 255.0\n                x.append(img_array)\n                y.append(class_name)   # store label\n            except:\n                print(f\"Skipped file: {img_path}\")\n\n# Convert to numpy arrays\nx = np.array(x)\ny = np.array(y)\n\n# Print dataset stats after random dropping\nprint(\"Total images after randomly dropping 35% per class:\", len(x))\nfor class_name in class_names:\n    count = sum(y == class_name)\n    print(f\"Class '{class_name}': {count} images\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-07T18:17:05.554192Z","iopub.execute_input":"2025-09-07T18:17:05.554773Z","iopub.status.idle":"2025-09-07T18:19:48.433788Z","shell.execute_reply.started":"2025-09-07T18:17:05.554749Z","shell.execute_reply":"2025-09-07T18:19:48.433002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# Shuffle dataset consistently\nindices = np.arange(len(x))\nnp.random.shuffle(indices)\n\nx = x[indices]\ny = y[indices]\n\nprint(\"After shuffling:\")\nprint(\"X shape:\", x.shape)\nprint(\"y shape:\", y.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-07T18:19:48.434928Z","iopub.execute_input":"2025-09-07T18:19:48.435791Z","iopub.status.idle":"2025-09-07T18:19:53.497224Z","shell.execute_reply.started":"2025-09-07T18:19:48.435767Z","shell.execute_reply":"2025-09-07T18:19:53.496559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n# stratify = y means to do balanced distribution of the labels \nX_train, X_test, y_train, y_test = train_test_split(\n    x, y, test_size=0.3, random_state=42, stratify=y\n)      \n\nprint(\"Train set:\", X_train.shape, y_train.shape)\nprint(\"Test set:\", X_test.shape, y_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-07T18:20:02.595107Z","iopub.execute_input":"2025-09-07T18:20:02.595380Z","iopub.status.idle":"2025-09-07T18:20:04.634762Z","shell.execute_reply.started":"2025-09-07T18:20:02.595358Z","shell.execute_reply":"2025-09-07T18:20:04.634126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# Function to count images per class\ndef count_classes(y_data, name=\"Dataset\"):\n    unique_classes, counts = np.unique(y_data, return_counts=True)\n    print(f\"\\n{name} class distribution:\")\n    for cls, cnt in zip(unique_classes, counts):\n        print(f\"Class '{cls}': {cnt} images\")\n\n# Check distribution\ncount_classes(y_train, \"Train set\")\ncount_classes(y_test, \"Test set\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-07T18:20:21.154548Z","iopub.execute_input":"2025-09-07T18:20:21.155136Z","iopub.status.idle":"2025-09-07T18:20:21.161375Z","shell.execute_reply.started":"2025-09-07T18:20:21.155111Z","shell.execute_reply":"2025-09-07T18:20:21.160727Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-07T18:20:25.830174Z","iopub.execute_input":"2025-09-07T18:20:25.830749Z","iopub.status.idle":"2025-09-07T18:20:25.841034Z","shell.execute_reply.started":"2025-09-07T18:20:25.830703Z","shell.execute_reply":"2025-09-07T18:20:25.840411Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install ultralytics\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-07T18:20:27.417920Z","iopub.execute_input":"2025-09-07T18:20:27.418197Z","iopub.status.idle":"2025-09-07T18:21:50.850985Z","shell.execute_reply.started":"2025-09-07T18:20:27.418170Z","shell.execute_reply":"2025-09-07T18:21:50.850205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-07T18:22:09.282016Z","iopub.execute_input":"2025-09-07T18:22:09.282309Z","iopub.status.idle":"2025-09-07T18:22:09.287984Z","shell.execute_reply.started":"2025-09-07T18:22:09.282285Z","shell.execute_reply":"2025-09-07T18:22:09.287398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom tensorflow.keras.preprocessing.image import array_to_img\n\ndef save_dataset(X, y, base_dir):\n    os.makedirs(base_dir, exist_ok=True)\n    for img_array, label in zip(X, y):\n        class_dir = os.path.join(base_dir, str(label))\n        os.makedirs(class_dir, exist_ok=True)\n        img = array_to_img(img_array)  # Convert numpy -> PIL image\n        img.save(os.path.join(class_dir, f\"{np.random.randint(1e9)}.jpg\"))\n\n# Save train and test sets\nsave_dataset(X_train, y_train, \"driver_dataset/train\")\nsave_dataset(X_test, y_test, \"driver_dataset/val\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-07T18:22:31.526962Z","iopub.execute_input":"2025-09-07T18:22:31.527241Z","iopub.status.idle":"2025-09-07T18:22:47.068574Z","shell.execute_reply.started":"2025-09-07T18:22:31.527221Z","shell.execute_reply":"2025-09-07T18:22:47.068030Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-07T18:23:10.802483Z","iopub.execute_input":"2025-09-07T18:23:10.803033Z","iopub.status.idle":"2025-09-07T18:23:10.807497Z","shell.execute_reply.started":"2025-09-07T18:23:10.803011Z","shell.execute_reply":"2025-09-07T18:23:10.806965Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from ultralytics import YOLO\n\n# Load YOLOv8 classification model (nano version for speed, you can try small/medium too)\nmodel = YOLO(\"yolov8n-cls.pt\")\n\n# Train\nmodel.train(\n    data=\"driver_dataset\",   # path with train/val folders\n    epochs=10,\n    imgsz=224,\n    batch=32,\n    patience = 4\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-07T18:24:56.603362Z","iopub.execute_input":"2025-09-07T18:24:56.604108Z","iopub.status.idle":"2025-09-07T18:30:15.702539Z","shell.execute_reply.started":"2025-09-07T18:24:56.604083Z","shell.execute_reply":"2025-09-07T18:30:15.701508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"metrics = model.val()\nprint(metrics)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-07T18:32:58.171898Z","iopub.execute_input":"2025-09-07T18:32:58.173027Z","iopub.status.idle":"2025-09-07T18:33:05.460439Z","shell.execute_reply.started":"2025-09-07T18:32:58.172989Z","shell.execute_reply":"2025-09-07T18:33:05.459666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-07T18:33:19.471663Z","iopub.execute_input":"2025-09-07T18:33:19.472357Z","iopub.status.idle":"2025-09-07T18:33:19.477354Z","shell.execute_reply.started":"2025-09-07T18:33:19.472328Z","shell.execute_reply":"2025-09-07T18:33:19.476606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix, classification_report\nimport cv2\n\ny_pred = []\n\nfor img in X_test:\n    # Ensure image is uint8 and correct shape\n    if img.dtype != np.uint8:\n        img = (img * 255).astype(np.uint8)  # if normalized [0,1]\n    \n    if img.shape[-1] != 3:  \n        img = cv2.cvtColor(img, cv2.COLOR_GRAY2RGB)  # if grayscale\n    \n    # Predict with YOLOv8\n    results = model.predict(img, imgsz=224, verbose=False)\n    pred_class = results[0].probs.top1\n    y_pred.append(pred_class)\n\n# Convert to numpy\ny_pred = np.array(y_pred)\n\n# Convert predicted indices back to class names\ny_pred_labels = [class_names[i] for i in y_pred]\n\n# Confusion Matrix\ncm = confusion_matrix(y_test, y_pred_labels, labels=class_names)\nplt.figure(figsize=(10, 8))\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\",\n            xticklabels=class_names,\n            yticklabels=class_names)\nplt.xlabel(\"Predicted\")\nplt.ylabel(\"True\")\nplt.title(\"Confusion Matrix - YOLOv8 Classification\")\nplt.show()\n\n# Classification Report\nprint(classification_report(y_test, y_pred_labels, target_names=class_names))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-07T18:36:37.182484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}