{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":5048,"databundleVersionId":868335,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.applications import MobileNetV2\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Dense\nfrom tensorflow.keras.models import Sequential\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T11:00:51.461353Z","iopub.execute_input":"2025-03-12T11:00:51.461746Z","iopub.status.idle":"2025-03-12T11:01:09.897001Z","shell.execute_reply.started":"2025-03-12T11:00:51.461706Z","shell.execute_reply":"2025-03-12T11:01:09.895883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ 1️⃣ Load Dataset Metadata\ntrain_dir = \"/kaggle/input/state-farm-distracted-driver-detection/imgs/train\"\nlabels_csv = \"/kaggle/input/state-farm-distracted-driver-detection/driver_imgs_list.csv\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T11:01:09.897992Z","iopub.execute_input":"2025-03-12T11:01:09.898555Z","iopub.status.idle":"2025-03-12T11:01:09.903151Z","shell.execute_reply.started":"2025-03-12T11:01:09.898528Z","shell.execute_reply":"2025-03-12T11:01:09.901819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels_df = pd.read_csv(labels_csv)\nlabels_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T11:01:09.904337Z","iopub.execute_input":"2025-03-12T11:01:09.904730Z","iopub.status.idle":"2025-03-12T11:01:10.037190Z","shell.execute_reply.started":"2025-03-12T11:01:09.904693Z","shell.execute_reply":"2025-03-12T11:01:10.036307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ 2️⃣ Extract Image Paths & Labels (Ensuring Correct Mapping)\nimage_paths = []\nlabels = []\n\nfor class_name in sorted(os.listdir(train_dir)):  # Ensure correct order\n    class_dir = os.path.join(train_dir, class_name)\n    if os.path.isdir(class_dir):  # Ensure it's a folder\n        for img_name in os.listdir(class_dir):\n            img_path = os.path.join(class_dir, img_name)\n            image_paths.append(img_path)\n            labels.append(int(class_name[1]))  # Convert 'c0' → 0, ..., 'c9' → 9\n\n# Convert to NumPy Arrays\nimage_paths = np.array(image_paths)\nlabels = np.array(labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T11:01:10.039122Z","iopub.execute_input":"2025-03-12T11:01:10.039406Z","iopub.status.idle":"2025-03-12T11:01:10.520254Z","shell.execute_reply.started":"2025-03-12T11:01:10.039384Z","shell.execute_reply":"2025-03-12T11:01:10.519405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ 3️⃣ Split Data (80% Train, 20% Validation)\ntrain_paths, val_paths, train_labels, val_labels = train_test_split(\n    image_paths, labels, test_size=0.2, random_state=42, stratify=labels\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T11:01:10.521430Z","iopub.execute_input":"2025-03-12T11:01:10.521755Z","iopub.status.idle":"2025-03-12T11:01:10.546780Z","shell.execute_reply.started":"2025-03-12T11:01:10.521729Z","shell.execute_reply":"2025-03-12T11:01:10.545660Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ 4️⃣ Preprocess Images Before Training (Save as NumPy)\ndef preprocess_and_save(image_paths, labels, save_path):\n    images = []\n    for path in image_paths:\n        image = tf.io.read_file(path)\n        image = tf.image.decode_jpeg(image, channels=3)\n        image = tf.image.resize(image, [224, 224]) / 255.0  # Normalize\n        images.append(image.numpy())  # Convert to NumPy\n    \n    images = np.array(images, dtype=np.float32)\n    labels = np.array(labels)\n    \n    np.savez_compressed(save_path, images=images, labels=labels)\n    print(f\"✅ Saved preprocessed dataset at {save_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T11:01:10.547977Z","iopub.execute_input":"2025-03-12T11:01:10.548289Z","iopub.status.idle":"2025-03-12T11:01:10.553813Z","shell.execute_reply.started":"2025-03-12T11:01:10.548260Z","shell.execute_reply":"2025-03-12T11:01:10.552745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocess and Save Data\npreprocess_and_save(train_paths, train_labels, \"train_data.npz\")\npreprocess_and_save(val_paths, val_labels, \"val_data.npz\")\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-03-12T12:04:09.182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ 5️⃣ Load Preprocessed Data\ntrain_data = np.load(\"/kaggle/working/train_data.npz\")\nval_data = np.load(\"/kaggle/working/val_data.npz\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T11:19:08.028195Z","iopub.execute_input":"2025-03-12T11:19:08.028596Z","iopub.status.idle":"2025-03-12T11:19:08.056773Z","shell.execute_reply.started":"2025-03-12T11:19:08.028564Z","shell.execute_reply":"2025-03-12T11:19:08.055822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_images, train_labels = train_data[\"images\"], train_data[\"labels\"]\nval_images, val_labels = val_data[\"images\"], val_data[\"labels\"]","metadata":{"trusted":true,"execution":{"execution_failed":"2025-03-12T12:04:09.182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ 6️⃣ Create TensorFlow Datasets (No Preprocessing Needed)\ntrain_dataset = tf.data.Dataset.from_tensor_slices((train_images, train_labels)).batch(32).prefetch(tf.data.AUTOTUNE)\nval_dataset = tf.data.Dataset.from_tensor_slices((val_images, val_labels)).batch(32).prefetch(tf.data.AUTOTUNE)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-03-12T12:04:09.182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ 7️⃣ Load Pretrained MobileNetV2\nbase_model = MobileNetV2(input_shape=(224, 224, 3), include_top=False, weights=\"imagenet\")\nbase_model.trainable = False  # Freeze base model","metadata":{"trusted":true,"execution":{"execution_failed":"2025-03-12T12:04:09.182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ 8️⃣ Build Custom Model\nmodel = Sequential([\n    base_model,\n    GlobalAveragePooling2D(),\n    Dense(128, activation=\"relu\"),\n    Dense(10, activation=\"softmax\")  # 10 classes\n])","metadata":{"trusted":true,"execution":{"execution_failed":"2025-03-12T12:04:09.182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ 9️⃣ Compile Model\nmodel.compile(optimizer=\"adam\", loss=\"sparse_categorical_crossentropy\", metrics=[\"accuracy\"])","metadata":{"trusted":true,"execution":{"execution_failed":"2025-03-12T12:04:09.182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ 🔟 Train the Model (Now Faster & Stable)\nhistory = model.fit(train_dataset, epochs=10, validation_data=val_dataset)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-03-12T12:04:09.182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}