{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"colab":{"provenance":[],"toc_visible":true},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":5048,"databundleVersionId":868335,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## ML Project\n# Distracted Driver Recognition Using Support Vector Machine with Convolutional Neural Networks for Feature Extraction and Principal Component Analysis for Feature Reduction. \n\n#### Project by: \nARIHANT DEBNATH [RA2211003011033] <br>\nDIVA MERJA [RA2211003011034] <br>\nMAKADIA YAKSH VIJAYAKUMAR[RA2211003011035] <br>\nKRISHNA WADHWANI[RA2211003011045] <br>\n\n<br>\nReference Paper: https://ejurnal.undana.ac.id/index.php/jicon/article/view/12658 <br>\nDataset: https://www.kaggle.com/competitions/state-farm-distracted-driver-detection/data <br>\nReference Code: https://www.kaggle.com/code/achmadyunedaalfajr/skripsi-05-22-2023-svm-pca-scrath-cnn-lib-v4 \n<br>","metadata":{}},{"cell_type":"code","source":"# IMPORTANT: SOME KAGGLE DATA SOURCES ARE PRIVATE\n# RUN THIS CELL IN ORDER TO IMPORT YOUR KAGGLE DATA SOURCES.\nimport kagglehub\nkagglehub.login()\n","metadata":{"id":"GMze2OTXJEzR","execution":{"iopub.status.busy":"2024-11-10T15:23:20.905396Z","iopub.execute_input":"2024-11-10T15:23:20.905800Z","iopub.status.idle":"2024-11-10T15:23:20.928723Z","shell.execute_reply.started":"2024-11-10T15:23:20.905765Z","shell.execute_reply":"2024-11-10T15:23:20.927802Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import dataset\nstate_farm_distracted_driver_detection_path = kagglehub.competition_download('state-farm-distracted-driver-detection')\n\nprint('Data source import complete.')\n","metadata":{"id":"_w5QMBWmJEzS","execution":{"iopub.status.busy":"2024-11-10T15:23:20.930649Z","iopub.execute_input":"2024-11-10T15:23:20.930962Z","iopub.status.idle":"2024-11-10T15:23:21.497256Z","shell.execute_reply.started":"2024-11-10T15:23:20.930930Z","shell.execute_reply":"2024-11-10T15:23:21.496248Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Importing Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\nimport itertools\nimport pandas as pd\nimport glob\nimport pickle\nimport os\nimport time\nfrom keras.models import Sequential, save_model\nfrom keras.layers import Conv2D, MaxPooling2D, Flatten, Dense\nfrom keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import check_random_state\nfrom tqdm import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","id":"W2SNiQwgJEzT","execution":{"iopub.status.busy":"2024-11-10T15:23:21.498649Z","iopub.execute_input":"2024-11-10T15:23:21.499117Z","iopub.status.idle":"2024-11-10T15:23:21.505967Z","shell.execute_reply.started":"2024-11-10T15:23:21.499069Z","shell.execute_reply":"2024-11-10T15:23:21.504803Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Loading the Dataset","metadata":{}},{"cell_type":"code","source":"main_path = '/kaggle/input/state-farm-distracted-driver-detection/imgs/train/c'\nclass_labels = []\nimages = []\n\n# Iterates for each class\nfor class_index in range(10):\n    class_path = main_path + str(class_index)  # Path to the current class directory\n    for root, dirs, files in os.walk(class_path):\n        # Process files in the current class\n        for filename in tqdm(files, desc='Processing class ' + str(class_index)):\n            image_path = os.path.join(class_path, filename)\n            img = cv2.imread(image_path)\n            img = cv2.resize(img, (100, 100)) / 255\n            images.append(img)\n            class_labels.append(class_index)","metadata":{"id":"JyIF9hbdJEzU","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:23:21.506994Z","iopub.execute_input":"2024-11-10T15:23:21.507357Z","iopub.status.idle":"2024-11-10T15:26:44.979715Z","shell.execute_reply.started":"2024-11-10T15:23:21.507325Z","shell.execute_reply":"2024-11-10T15:26:44.978780Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{}},{"cell_type":"markdown","source":"## Train Test Split","metadata":{}},{"cell_type":"code","source":"train_images, test_images, train_labels, test_labels = train_test_split(np.array(images), np.array(class_labels), test_size = 0.2, shuffle=True)","metadata":{"id":"OKCje16oJEzV","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:26:44.983088Z","iopub.execute_input":"2024-11-10T15:26:44.983384Z","iopub.status.idle":"2024-11-10T15:26:48.134938Z","shell.execute_reply.started":"2024-11-10T15:26:44.983353Z","shell.execute_reply":"2024-11-10T15:26:48.133844Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Extraction using a CNN","metadata":{}},{"cell_type":"code","source":"# Define the custom CNN architecture\ncnn = Sequential()\ncnn.add(Conv2D(32, (3, 3), activation='relu', input_shape=(100, 100, 3)))\ncnn.add(MaxPooling2D((2, 2)))\ncnn.add(Conv2D(64, (3, 3), activation='relu'))\ncnn.add(MaxPooling2D((2, 2)))\ncnn.add(Conv2D(128, (3, 3), activation='relu'))\ncnn.add(MaxPooling2D((2, 2)))\ncnn.add(Flatten())\ncnn.add(Dense(256, activation='relu'))\ncnn.add(Dense(128, activation='relu'))\ncnn.add(Dense(10, activation='softmax'))\n\n# Compile the model\ncnn.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])","metadata":{"id":"C15_STseJEzW","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:26:48.136046Z","iopub.execute_input":"2024-11-10T15:26:48.136360Z","iopub.status.idle":"2024-11-10T15:26:48.901936Z","shell.execute_reply.started":"2024-11-10T15:26:48.136327Z","shell.execute_reply":"2024-11-10T15:26:48.901144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cnn.summary()","metadata":{"id":"dCvwwoXZJEzW","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:26:48.903002Z","iopub.execute_input":"2024-11-10T15:26:48.903299Z","iopub.status.idle":"2024-11-10T15:26:48.928783Z","shell.execute_reply.started":"2024-11-10T15:26:48.903268Z","shell.execute_reply":"2024-11-10T15:26:48.927946Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## One Hot Encoding \n[Read more](https://machinelearningmastery.com/why-one-hot-encode-data-in-machine-learning/)","metadata":{}},{"cell_type":"code","source":"# Encode the target labels as one-hot vectors\ntrain_labels_encoded = to_categorical(train_labels, num_classes=10)\ntest_labels_encoded = to_categorical(test_labels, num_classes=10)\n\ncnn.fit(train_images, train_labels_encoded, epochs=10, batch_size=32, validation_data=(test_images, test_labels_encoded))","metadata":{"id":"8FiJIhAJJEzW","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:26:48.930012Z","iopub.execute_input":"2024-11-10T15:26:48.930369Z","iopub.status.idle":"2024-11-10T15:27:49.504765Z","shell.execute_reply.started":"2024-11-10T15:26:48.930335Z","shell.execute_reply":"2024-11-10T15:27:49.503924Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Removing the final Fully Connected Layer to get the 128 features extracted from the previous layer.**","metadata":{}},{"cell_type":"code","source":"# Remove the last layer from the CNN model\ncnn = Sequential(cnn.layers[:-1])\n\n# Preventing the weights from being updated\nfor layer in cnn.layers:\n    layer.trainable = False","metadata":{"id":"Rq-p5LjJJEzX","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:27:49.506041Z","iopub.execute_input":"2024-11-10T15:27:49.506445Z","iopub.status.idle":"2024-11-10T15:27:49.522437Z","shell.execute_reply.started":"2024-11-10T15:27:49.506400Z","shell.execute_reply":"2024-11-10T15:27:49.521550Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cnn.summary()","metadata":{"id":"qjWwoHK8JEzY","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:27:49.526379Z","iopub.execute_input":"2024-11-10T15:27:49.526708Z","iopub.status.idle":"2024-11-10T15:27:49.553993Z","shell.execute_reply.started":"2024-11-10T15:27:49.526675Z","shell.execute_reply":"2024-11-10T15:27:49.553024Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Extracted Features","metadata":{}},{"cell_type":"code","source":"train_features = cnn.predict(train_images)\ntest_features = cnn.predict(test_images)","metadata":{"id":"6amkcTN-JEzZ","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:27:49.555293Z","iopub.execute_input":"2024-11-10T15:27:49.555928Z","iopub.status.idle":"2024-11-10T15:27:59.232452Z","shell.execute_reply.started":"2024-11-10T15:27:49.555865Z","shell.execute_reply":"2024-11-10T15:27:59.231657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_features.shape)\nprint(test_features.shape)","metadata":{"id":"9-bFaH01JEzd","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:27:59.233721Z","iopub.execute_input":"2024-11-10T15:27:59.234040Z","iopub.status.idle":"2024-11-10T15:27:59.238818Z","shell.execute_reply.started":"2024-11-10T15:27:59.234007Z","shell.execute_reply":"2024-11-10T15:27:59.237843Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Reduction using PCA\n\n**Here, we will reduce the extracted features(128) to 16 features using Principle Component Analysis.**\n[Learn More](https://www.youtube.com/watch?v=FgakZw6K1QQ)","metadata":{}},{"cell_type":"markdown","source":"### PCA Implementation","metadata":{}},{"cell_type":"code","source":"class PCA:\n    def __init__(self, n_components):\n        self.n_components = n_components\n        self.components = None\n        self.mean = None\n\n    def fit(self, X):\n        # Calculate the mean of each feature\n        self.mean = np.mean(X, axis=0)\n\n        # Center the data by subtracting the mean from each feature\n        X = X - self.mean\n\n        # Calculate the covariance matrix\n        cov = np.cov(X.T)\n\n        # Calculate the eigenvalues and eigenvectors of the covariance matrix\n        eigenvalues, eigenvectors = np.linalg.eig(cov)\n\n        # Sort the eigenvectors by their corresponding eigenvalues in descending order\n        eigenvectors = eigenvectors.T\n        idxs = np.argsort(eigenvalues)[::-1]\n        eigenvectors = eigenvectors[idxs]\n        eigenvalues = eigenvalues[idxs]\n\n        # Store the first n_components eigenvectors as the components\n        self.components = eigenvectors[0:self.n_components]\n\n    def transform(self, X):\n        # Center the data by subtracting the mean from each feature\n        X = X - self.mean\n\n        # Project the data onto the components\n        return np.dot(X, self.components.T)\n\n    def fit_transform(self, X):\n        # Calculate the mean of each feature\n        self.mean = np.mean(X, axis=0)\n\n        # Center the data by subtracting the mean from each feature\n        X = X - self.mean\n\n        # Calculate the covariance matrix\n        cov = np.cov(X.T)\n\n        # Calculate the eigenvalues and eigenvectors of the covariance matrix\n        eigenvalues, eigenvectors = np.linalg.eig(cov)\n\n        # Sort the eigenvectors by their corresponding eigenvalues in descending order\n        eigenvectors = eigenvectors.T\n        idxs = np.argsort(eigenvalues)[::-1]\n        eigenvectors = eigenvectors[idxs]\n        eigenvalues = eigenvalues[idxs]\n\n        # Store the first n_components eigenvectors as the components\n        self.components = eigenvectors[0:self.n_components]\n\n        # Project the data onto the components\n        return np.dot(X, self.components.T)","metadata":{"id":"lF0S4mbHJEzd","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:27:59.240009Z","iopub.execute_input":"2024-11-10T15:27:59.240371Z","iopub.status.idle":"2024-11-10T15:27:59.254962Z","shell.execute_reply.started":"2024-11-10T15:27:59.240326Z","shell.execute_reply":"2024-11-10T15:27:59.254154Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Applying PCA","metadata":{}},{"cell_type":"code","source":"# Create a PCA object with 16 components\npca = PCA(n_components=16)\ntrain_features_reduced = pca.fit_transform(train_features)\ntest_features_reduced = pca.transform(test_features)","metadata":{"id":"SFeOUTanJEze","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:27:59.259248Z","iopub.execute_input":"2024-11-10T15:27:59.259562Z","iopub.status.idle":"2024-11-10T15:27:59.336994Z","shell.execute_reply.started":"2024-11-10T15:27:59.259524Z","shell.execute_reply":"2024-11-10T15:27:59.335570Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_features_reduced.shape)\nprint(test_features_reduced.shape)","metadata":{"id":"OlpqGhapJEze","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:27:59.338431Z","iopub.execute_input":"2024-11-10T15:27:59.342610Z","iopub.status.idle":"2024-11-10T15:27:59.350272Z","shell.execute_reply.started":"2024-11-10T15:27:59.342564Z","shell.execute_reply":"2024-11-10T15:27:59.349109Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Saving the reduced features","metadata":{}},{"cell_type":"code","source":"train_features_reduced_save = pd.DataFrame(train_features_reduced)\ntrain_features_reduced_save.to_csv('train_features_reduced_save.csv', index=False)","metadata":{"id":"8bOQJh4bJEze","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:27:59.353478Z","iopub.execute_input":"2024-11-10T15:27:59.359319Z","iopub.status.idle":"2024-11-10T15:27:59.962086Z","shell.execute_reply.started":"2024-11-10T15:27:59.359259Z","shell.execute_reply":"2024-11-10T15:27:59.961257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_features_reduced_save = pd.DataFrame(test_features_reduced)\ntest_features_reduced_save.to_csv('test_features_reduced_save.csv', index=False)","metadata":{"id":"V5oJgEgvJEze","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:27:59.963249Z","iopub.execute_input":"2024-11-10T15:27:59.963569Z","iopub.status.idle":"2024-11-10T15:28:00.106312Z","shell.execute_reply.started":"2024-11-10T15:27:59.963537Z","shell.execute_reply":"2024-11-10T15:28:00.105553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_tar_save = pd.DataFrame(train_labels)\ntrain_tar_save.to_csv('train_tar.csv', index=False)","metadata":{"id":"yXrsVjdiJEzf","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:28:00.107377Z","iopub.execute_input":"2024-11-10T15:28:00.107677Z","iopub.status.idle":"2024-11-10T15:28:00.124207Z","shell.execute_reply.started":"2024-11-10T15:28:00.107644Z","shell.execute_reply":"2024-11-10T15:28:00.123505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_tar_save = pd.DataFrame(test_labels)\ntest_tar_save.to_csv('test_tar.csv', index=False)","metadata":{"id":"x6BoIE--JEzf","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:28:00.125304Z","iopub.execute_input":"2024-11-10T15:28:00.126081Z","iopub.status.idle":"2024-11-10T15:28:00.134578Z","shell.execute_reply.started":"2024-11-10T15:28:00.126038Z","shell.execute_reply":"2024-11-10T15:28:00.133704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_features_reduced[0])","metadata":{"id":"hzuIGswmJEzf","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:28:00.135752Z","iopub.execute_input":"2024-11-10T15:28:00.136590Z","iopub.status.idle":"2024-11-10T15:28:00.142109Z","shell.execute_reply.started":"2024-11-10T15:28:00.136540Z","shell.execute_reply":"2024-11-10T15:28:00.141198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(test_features_reduced[0])","metadata":{"id":"qpIYOWrdJEzg","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:28:00.143317Z","iopub.execute_input":"2024-11-10T15:28:00.143683Z","iopub.status.idle":"2024-11-10T15:28:00.151153Z","shell.execute_reply.started":"2024-11-10T15:28:00.143640Z","shell.execute_reply":"2024-11-10T15:28:00.150230Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Classification","metadata":{}},{"cell_type":"markdown","source":"## 1. Support Vector Machine(SVM)","metadata":{}},{"cell_type":"markdown","source":"### 1.1 SVM - Our Implementation","metadata":{}},{"cell_type":"code","source":"class SVM:\n    def __init__(self, C=1, max_iter=50, tol=0.05,\n                 random_state=None, verbose=0):\n        self.C = C\n        self.max_iter = max_iter\n        self.tol = tol,\n        self.random_state = random_state\n        self.verbose = verbose\n\n    def projection_simplex(self, v, z=1):\n        n_features = v.shape[0]\n        u = np.sort(v)[::-1]\n        cssv = np.cumsum(u) - z\n        ind = np.arange(n_features) + 1\n        cond = u - cssv / ind > 0\n        rho = ind[cond][-1]\n        theta = cssv[cond][-1] / float(rho)\n        w = np.maximum(v - theta, 0)\n        return w\n\n    def _partial_gradient(self, X, y, i):\n        # Partial gradient for the ith sample.\n        g = np.dot(X[i], self.coef_.T) + 1\n        g[y[i]] -= 1\n        return g\n\n    def _violation(self, g, y, i):\n        # Optimality violation for the ith sample.\n        smallest = np.inf\n        for k in range(g.shape[0]):\n            if k == y[i] and self.dual_coef_[k, i] >= self.C:\n                continue\n            elif k != y[i] and self.dual_coef_[k, i] >= 0:\n                continue\n\n            smallest = min(smallest, g[k])\n\n        return g.max() - smallest\n\n    def _solve_subproblem(self, g, y, norms, i):\n        # Prepare inputs to the projection.\n        Ci = np.zeros(g.shape[0])\n        Ci[y[i]] = self.C\n        beta_hat = norms[i] * (Ci - self.dual_coef_[:, i]) + g / norms[i]\n        z = self.C * norms[i]\n\n        # Compute projection onto the simplex.\n        beta = self.projection_simplex(beta_hat, z)\n\n        return Ci - self.dual_coef_[:, i] - beta / norms[i]\n\n    def fit(self, X, y):\n        n_samples, n_features = X.shape\n\n        n_classes = np.unique(y).size\n        self.dual_coef_ = np.zeros((n_classes, n_samples), dtype=np.float64)\n        self.coef_ = np.zeros((n_classes, n_features))\n\n        # Pre-compute norms.\n        norms = np.sqrt(np.sum(X ** 2, axis=1))\n\n        # Shuffle sample indices.\n        rs = check_random_state(self.random_state)\n        ind = np.arange(n_samples)\n        rs.shuffle(ind)\n\n        violation_init = None\n        for it in range(self.max_iter):\n            violation_sum = 0\n\n            for ii in range(n_samples):\n                i = ind[ii]\n\n                # All-zero samples can be safely ignored.\n                if norms[i] == 0:\n                    continue\n\n                g = self._partial_gradient(X, y, i)\n                v = self._violation(g, y, i)\n                violation_sum += v\n\n                if v < 1e-12:\n                    continue\n\n                # Solve subproblem for the ith sample.\n                delta = self._solve_subproblem(g, y, norms, i)\n\n                # Update primal and dual coefficients.\n                self.coef_ = self.coef_.astype(np.complex128)\n                self.dual_coef_ = self.dual_coef_.astype(np.complex128)\n                delta = delta.astype(np.complex128)\n\n                self.coef_ += np.multiply(delta[:, np.newaxis], X[i][:, np.newaxis].conj().T)\n                self.dual_coef_[:, i] += delta\n\n            if it == 0:\n                violation_init = violation_sum\n\n            vratio = violation_sum / violation_init\n\n            if vratio < self.tol:\n                if self.verbose >= 1:\n                    print(\"Converged\")\n                break\n\n        return self\n\n    def predict(self, X):\n        decision = np.dot(X, self.coef_.T)\n        return decision.argmax(axis=1)","metadata":{"id":"2b72EtL6JEzg","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:28:00.152470Z","iopub.execute_input":"2024-11-10T15:28:00.152869Z","iopub.status.idle":"2024-11-10T15:28:00.174297Z","shell.execute_reply.started":"2024-11-10T15:28:00.152829Z","shell.execute_reply":"2024-11-10T15:28:00.173472Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### SVM with PCA","metadata":{}},{"cell_type":"code","source":"svm = SVM(C=1, tol=0.001, max_iter=100, random_state=0, verbose=1)\n\nstart_time = time.time()\nsvm.fit(train_features_reduced, train_labels)\nend_time = time.time()\n\ntraining_duration = end_time - start_time\nprint(\"Training duration:\", training_duration, \"seconds\")\n\npredictions = svm.predict(test_features_reduced)\naccuracy = np.mean(predictions == test_labels)\nprint(\"Accuracy:\", accuracy)","metadata":{"id":"MgOqouGGJEzg","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:28:00.175349Z","iopub.execute_input":"2024-11-10T15:28:00.175627Z","iopub.status.idle":"2024-11-10T15:28:55.867664Z","shell.execute_reply.started":"2024-11-10T15:28:00.175596Z","shell.execute_reply":"2024-11-10T15:28:55.866353Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### SVM without PCA","metadata":{}},{"cell_type":"code","source":"svm_no_pca = SVM(C=1, tol=0.001, max_iter=100, random_state=0, verbose=1)\n\nstart_time_no_pca = time.time()\nsvm_no_pca.fit(train_features, train_labels)\nend_time_no_pca = time.time()\n\ntraining_duration_no_pca = end_time_no_pca - start_time_no_pca\nprint(\"Training duration without PCA:\", training_duration_no_pca, \"seconds\")\n\npredictions_no_pca = svm_no_pca.predict(test_features)\naccuracy_no_pca = np.mean(predictions_no_pca == test_labels)\nprint(\"Accuracy without PCA:\", accuracy_no_pca)","metadata":{"id":"aLH77dbyJEzg","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:28:55.872577Z","iopub.execute_input":"2024-11-10T15:28:55.877067Z","iopub.status.idle":"2024-11-10T15:30:03.068314Z","shell.execute_reply.started":"2024-11-10T15:28:55.876998Z","shell.execute_reply":"2024-11-10T15:30:03.066927Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1.2 SVM from sklearn library \n[Documentation](https://scikit-learn.org/stable/modules/generated/sklearn.svm.SVC.html)","metadata":{}},{"cell_type":"markdown","source":"#### SVM with PCA","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import SVC\nfrom sklearn.metrics import accuracy_score\n\n# Create an SVM classifier with a linear kernel\nsvm_sklearn = SVC(kernel='linear', C=1, max_iter=100, random_state=0, verbose=0)\n\n# Train the SVM classifier on the reduced features (with PCA)\nstart_time_svm_sklearn = time.time()\nsvm_sklearn.fit(train_features_reduced, train_labels)\nend_time_svm_sklearn = time.time()\n\ntraining_duration_svm_sklearn = end_time_svm_sklearn- start_time_svm_sklearn\nprint(\"Training duration with PCA:\", training_duration_svm_sklearn, \"seconds\")\n\n# Predict the labels for the test set\npredictions_svm_sklearn = svm_sklearn.predict(test_features_reduced)\naccuracy_svm_sklearn = accuracy_score(test_labels, predictions_svm_sklearn)\nprint(\"Accuracy with PCA:\", accuracy_svm_sklearn)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:03.070315Z","iopub.execute_input":"2024-11-10T15:30:03.071410Z","iopub.status.idle":"2024-11-10T15:30:03.345052Z","shell.execute_reply.started":"2024-11-10T15:30:03.071348Z","shell.execute_reply":"2024-11-10T15:30:03.344089Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### SVM without PCA","metadata":{}},{"cell_type":"code","source":"# Train the SVM classifier on the original features (without PCA)\nstart_time_svm_sklearn_no_pca = time.time()\nsvm_sklearn.fit(train_features, train_labels)\nend_time_svm_sklearn_no_pca = time.time()\n\ntraining_duration_svm_sklearn_no_pca = end_time_svm_sklearn_no_pca - start_time_svm_sklearn_no_pca\nprint(\"Training duration without PCA:\", training_duration_svm_sklearn_no_pca, \"seconds\")\n\n# Predict the labels for the test set\npredictions_svm_sklearn_no_pca = svm_sklearn.predict(test_features)\naccuracy_svm_sklearn_no_pca = accuracy_score(test_labels, predictions_svm_sklearn_no_pca)\nprint(\"Accuracy without PCA:\", accuracy_svm_sklearn_no_pca)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:03.346511Z","iopub.execute_input":"2024-11-10T15:30:03.346820Z","iopub.status.idle":"2024-11-10T15:30:03.747902Z","shell.execute_reply.started":"2024-11-10T15:30:03.346787Z","shell.execute_reply":"2024-11-10T15:30:03.746911Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## SVM Model Evaluation","metadata":{}},{"cell_type":"markdown","source":"### 1. Confusion Matrix\n[Learn More](https://towardsdatascience.com/understanding-confusion-matrix-a9ad42dcfd62)","metadata":{}},{"cell_type":"code","source":"def plot_confusion_matrix(cm, classes,\n                          normalize=False,\n                          title='Confusion matrix',\n                          cmap=plt.cm.Blues):\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n        print(\"Normalized confusion matrix\")\n    else:\n        print('Confusion matrix')\n\n    #print(cm)\n    plt.figure(figsize = (10,10))\n    plt.imshow(cm, interpolation='nearest', cmap=cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation=45)\n    plt.yticks(tick_marks, classes)\n\n    fmt = '.2f' if normalize else 'd'\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, format(cm[i, j], fmt),\n                 horizontalalignment=\"center\",\n                 color=\"white\" if cm[i, j] > thresh else \"black\")\n\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    plt.tight_layout()","metadata":{"id":"DB73DWC5JEzh","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:03.749157Z","iopub.execute_input":"2024-11-10T15:30:03.749482Z","iopub.status.idle":"2024-11-10T15:30:03.759413Z","shell.execute_reply.started":"2024-11-10T15:30:03.749437Z","shell.execute_reply":"2024-11-10T15:30:03.758421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n# Compute confusion matrix\ncnf_matrix = confusion_matrix(test_labels, predictions)\nnp.set_printoptions(precision=2)\n\n# Plot non-normalized confusion matrix\nplt.figure()\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\n\nplot_confusion_matrix(cnf_matrix, classes=class_names,\n                      title='Confusion matrix with PCA')","metadata":{"id":"sUjWK5BpJEzi","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:03.760859Z","iopub.execute_input":"2024-11-10T15:30:03.761205Z","iopub.status.idle":"2024-11-10T15:30:04.536771Z","shell.execute_reply.started":"2024-11-10T15:30:03.761172Z","shell.execute_reply":"2024-11-10T15:30:04.535860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n# Compute confusion matrix\ncnf_matrix = confusion_matrix(test_labels, predictions_no_pca)\nnp.set_printoptions(precision=2)\n\n# Plot non-normalized confusion matrix\nplt.figure()\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\n\nplot_confusion_matrix(cnf_matrix, classes=class_names,\n                      title='Confusion matrix without PCA')","metadata":{"id":"oGfqgdZZJEzi","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:04.538172Z","iopub.execute_input":"2024-11-10T15:30:04.538601Z","iopub.status.idle":"2024-11-10T15:30:05.257939Z","shell.execute_reply.started":"2024-11-10T15:30:04.538554Z","shell.execute_reply":"2024-11-10T15:30:05.256999Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 2. Classification Report\nThis is the summary of the quality of classification made by the constructed ML model. The first column is the class label’s name and followed by Precision, Recall, F1-score, and Support. ","metadata":{}},{"cell_type":"code","source":"#classification report\nfrom sklearn.metrics import confusion_matrix, classification_report\ny_true = test_labels\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\nprint(classification_report(y_true, predictions, target_names = class_names))","metadata":{"id":"0Yxb9RDxJEzi","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:05.259166Z","iopub.execute_input":"2024-11-10T15:30:05.259481Z","iopub.status.idle":"2024-11-10T15:30:05.276031Z","shell.execute_reply.started":"2024-11-10T15:30:05.259447Z","shell.execute_reply":"2024-11-10T15:30:05.275177Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#classification report\nfrom sklearn.metrics import confusion_matrix, classification_report\ny_true = test_labels\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\nprint(classification_report(y_true, predictions_no_pca, target_names = class_names))","metadata":{"id":"YcHKCYv4JEzi","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:05.277067Z","iopub.execute_input":"2024-11-10T15:30:05.277371Z","iopub.status.idle":"2024-11-10T15:30:05.293028Z","shell.execute_reply.started":"2024-11-10T15:30:05.277339Z","shell.execute_reply":"2024-11-10T15:30:05.292181Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 3. AUC (Area Under the Curve)\nAUC represents the degree or measure of separability. It tells how much the model is capable of distinguishing between classes. Higher the AUC, the better the model is at predicting 0 classes as 0 and 1 classes as 1. <br>\n[Learn More](https://towardsdatascience.com/understanding-auc-roc-curve-68b2303cc9c5)","metadata":{}},{"cell_type":"code","source":"#AUC\nfrom sklearn.preprocessing import LabelBinarizer\nfrom sklearn.metrics import roc_auc_score\ndef multiclass_roc_auc_score(y_test, y_pred, average=\"macro\"):\n    lb = LabelBinarizer()\n    lb.fit(y_test)\n    y_test = lb.transform(y_test)\n    y_pred = lb.transform(y_pred)\n    return roc_auc_score(y_test, y_pred, average=average)\n\ny_true = test_labels\nmulticlass_roc_auc_score(y_true, predictions)","metadata":{"id":"TXQQmtC2JEzi","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:05.294080Z","iopub.execute_input":"2024-11-10T15:30:05.294374Z","iopub.status.idle":"2024-11-10T15:30:05.320906Z","shell.execute_reply.started":"2024-11-10T15:30:05.294342Z","shell.execute_reply":"2024-11-10T15:30:05.320055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#AUC\nfrom sklearn.preprocessing import LabelBinarizer\nfrom sklearn.metrics import roc_auc_score\ndef multiclass_roc_auc_score(y_test, y_pred, average=\"macro\"):\n    lb = LabelBinarizer()\n    lb.fit(y_test)\n    y_test = lb.transform(y_test)\n    y_pred = lb.transform(y_pred)\n    return roc_auc_score(y_test, y_pred, average=average)\n\ny_true = test_labels\nmulticlass_roc_auc_score(y_true, predictions_no_pca)","metadata":{"id":"1_kM265UJEzj","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:05.321804Z","iopub.execute_input":"2024-11-10T15:30:05.322076Z","iopub.status.idle":"2024-11-10T15:30:05.345565Z","shell.execute_reply.started":"2024-11-10T15:30:05.322047Z","shell.execute_reply":"2024-11-10T15:30:05.344740Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. K Nearest Neighbours (KNN) ","metadata":{}},{"cell_type":"markdown","source":"#### KNN with PCA","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\n# Create a KNN classifier with 5 neighbors\nknn = KNeighborsClassifier(n_neighbors=5)\n\n# Train the KNN classifier on the reduced features(with PCA)\nstart_time_knn = time.time()\nknn.fit(train_features_reduced, train_labels)\nend_time_knn = time.time() \n\ntraining_duration_knn = end_time_knn - start_time_knn\nprint(\"Training duration with PCA:\", training_duration_knn, \"seconds\")\n\n# Predict the labels for the test set\npredictions_knn = knn.predict(test_features_reduced)\naccuracy_knn = np.mean(predictions_knn == test_labels)\nprint(\"Accuracy with PCA:\", accuracy_knn)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:05.346588Z","iopub.execute_input":"2024-11-10T15:30:05.346926Z","iopub.status.idle":"2024-11-10T15:30:05.830699Z","shell.execute_reply.started":"2024-11-10T15:30:05.346861Z","shell.execute_reply":"2024-11-10T15:30:05.829724Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### KNN without PCA","metadata":{}},{"cell_type":"code","source":"# Train the KNN classifier on the original features(without PCA)\nstart_time_knn_no_pca = time.time()\nknn.fit(train_features, train_labels)\nend_time_knn_no_pca = time.time()\n\ntraining_duration_knn_no_pca = end_time_knn_no_pca - start_time_knn_no_pca\nprint(\"Training duration without PCA:\", training_duration_knn_no_pca, \"seconds\")\n\n# Predict the labels for the test set\npredictions_knn_no_pca = knn.predict(test_features)\naccuracy_knn_no_pca = np.mean(predictions_knn_no_pca == test_labels)\nprint(\"Accuracy without PCA:\", accuracy_knn_no_pca)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:05.831830Z","iopub.execute_input":"2024-11-10T15:30:05.832163Z","iopub.status.idle":"2024-11-10T15:30:06.386370Z","shell.execute_reply.started":"2024-11-10T15:30:05.832129Z","shell.execute_reply":"2024-11-10T15:30:06.385247Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## KNN Model Evaluation","metadata":{}},{"cell_type":"markdown","source":"### 1. Confusion Matrix","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n# Compute confusion matrix\ncnf_matrix = confusion_matrix(test_labels, predictions_knn)\nnp.set_printoptions(precision=2)\n\n# Plot non-normalized confusion matrix\nplt.figure()\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\n\nplot_confusion_matrix(cnf_matrix, classes=class_names,\n                      title='KNN Confusion matrix with PCA')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:06.387554Z","iopub.execute_input":"2024-11-10T15:30:06.387934Z","iopub.status.idle":"2024-11-10T15:30:07.044186Z","shell.execute_reply.started":"2024-11-10T15:30:06.387868Z","shell.execute_reply":"2024-11-10T15:30:07.043260Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n# Compute confusion matrix\ncnf_matrix = confusion_matrix(test_labels, predictions_knn_no_pca)\nnp.set_printoptions(precision=2)\n\n# Plot non-normalized confusion matrix\nplt.figure()\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\n\nplot_confusion_matrix(cnf_matrix, classes=class_names,\n                      title='KNN Confusion matrix without PCA')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:07.045345Z","iopub.execute_input":"2024-11-10T15:30:07.045655Z","iopub.status.idle":"2024-11-10T15:30:07.931248Z","shell.execute_reply.started":"2024-11-10T15:30:07.045622Z","shell.execute_reply":"2024-11-10T15:30:07.930274Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 2. Classification Report","metadata":{}},{"cell_type":"code","source":"#classification report\nfrom sklearn.metrics import confusion_matrix, classification_report\ny_true = test_labels\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\nprint(classification_report(y_true, predictions_knn, target_names = class_names))","metadata":{"id":"0Yxb9RDxJEzi","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:07.932504Z","iopub.execute_input":"2024-11-10T15:30:07.932898Z","iopub.status.idle":"2024-11-10T15:30:07.949932Z","shell.execute_reply.started":"2024-11-10T15:30:07.932839Z","shell.execute_reply":"2024-11-10T15:30:07.948924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#classification report\nfrom sklearn.metrics import confusion_matrix, classification_report\ny_true = test_labels\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\nprint(classification_report(y_true, predictions_knn_no_pca, target_names = class_names))","metadata":{"id":"YcHKCYv4JEzi","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:07.950926Z","iopub.execute_input":"2024-11-10T15:30:07.951188Z","iopub.status.idle":"2024-11-10T15:30:07.967296Z","shell.execute_reply.started":"2024-11-10T15:30:07.951159Z","shell.execute_reply":"2024-11-10T15:30:07.966396Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 3. AUC","metadata":{}},{"cell_type":"code","source":"#AUC\nfrom sklearn.preprocessing import LabelBinarizer\nfrom sklearn.metrics import roc_auc_score\ndef multiclass_roc_auc_score(y_test, y_pred, average=\"macro\"):\n    lb = LabelBinarizer()\n    lb.fit(y_test)\n    y_test = lb.transform(y_test)\n    y_pred = lb.transform(y_pred)\n    return roc_auc_score(y_test, y_pred, average=average)\n\ny_true = test_labels\nmulticlass_roc_auc_score(y_true, predictions_knn)","metadata":{"id":"TXQQmtC2JEzi","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:07.968690Z","iopub.execute_input":"2024-11-10T15:30:07.969135Z","iopub.status.idle":"2024-11-10T15:30:07.993729Z","shell.execute_reply.started":"2024-11-10T15:30:07.969092Z","shell.execute_reply":"2024-11-10T15:30:07.992838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#AUC\nfrom sklearn.preprocessing import LabelBinarizer\nfrom sklearn.metrics import roc_auc_score\ndef multiclass_roc_auc_score(y_test, y_pred, average=\"macro\"):\n    lb = LabelBinarizer()\n    lb.fit(y_test)\n    y_test = lb.transform(y_test)\n    y_pred = lb.transform(y_pred)\n    return roc_auc_score(y_test, y_pred, average=average)\n\ny_true = test_labels\nmulticlass_roc_auc_score(y_true, predictions_knn_no_pca)","metadata":{"id":"1_kM265UJEzj","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:08.001639Z","iopub.execute_input":"2024-11-10T15:30:08.001919Z","iopub.status.idle":"2024-11-10T15:30:08.026307Z","shell.execute_reply.started":"2024-11-10T15:30:08.001890Z","shell.execute_reply":"2024-11-10T15:30:08.025459Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Random Forest","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score\n\n# Create a Random Forest classifier\nrf = RandomForestClassifier(n_estimators=100, random_state=0)\n\n# Train the Random Forest classifier on the reduced features(with PCA)\nstart_time_rf = time.time()\nrf.fit(train_features_reduced, train_labels)\nend_time_rf = time.time()\n\ntraining_duration_rf = end_time_rf - start_time_rf\nprint(\"Training duration with PCA:\", training_duration_rf, \"seconds\")\n\n# Predict the labels for the test set\npredictions_rf = rf.predict(test_features_reduced)\naccuracy_rf = accuracy_score(test_labels, predictions_rf)\nprint(\"Accuracy with PCA:\", accuracy_rf)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:08.027460Z","iopub.execute_input":"2024-11-10T15:30:08.027819Z","iopub.status.idle":"2024-11-10T15:30:13.983039Z","shell.execute_reply.started":"2024-11-10T15:30:08.027776Z","shell.execute_reply":"2024-11-10T15:30:13.982065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the Random Forest classifier on the original features(without PCA)\nstart_time_rf_no_pca = time.time()\nrf.fit(train_features, train_labels)\nend_time_rf_no_pca = time.time()\n\ntraining_duration_rf_no_pca = end_time_rf_no_pca - start_time_rf_no_pca\nprint(\"Training duration without PCA:\", training_duration_rf_no_pca, \"seconds\")\n\n# Predict the labels for the test set\npredictions_rf_no_pca = rf.predict(test_features)\naccuracy_rf_no_pca = accuracy_score(test_labels, predictions_rf_no_pca)\nprint(\"Accuracy without PCA:\", accuracy_rf_no_pca)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:13.984377Z","iopub.execute_input":"2024-11-10T15:30:13.984804Z","iopub.status.idle":"2024-11-10T15:30:20.563299Z","shell.execute_reply.started":"2024-11-10T15:30:13.984760Z","shell.execute_reply":"2024-11-10T15:30:20.562355Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Random Forrest Model Evaluation","metadata":{}},{"cell_type":"markdown","source":"### 1. Confusion Matrix","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n# Compute confusion matrix\ncnf_matrix = confusion_matrix(test_labels, predictions_rf)\nnp.set_printoptions(precision=2)\n\n# Plot non-normalized confusion matrix\nplt.figure()\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\n\nplot_confusion_matrix(cnf_matrix, classes=class_names,\n                      title='Random Forest Confusion matrix with PCA')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:20.564440Z","iopub.execute_input":"2024-11-10T15:30:20.564754Z","iopub.status.idle":"2024-11-10T15:30:21.285020Z","shell.execute_reply.started":"2024-11-10T15:30:20.564722Z","shell.execute_reply":"2024-11-10T15:30:21.284079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n# Compute confusion matrix\ncnf_matrix = confusion_matrix(test_labels, predictions_rf_no_pca)\nnp.set_printoptions(precision=2)\n\n# Plot non-normalized confusion matrix\nplt.figure()\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\n\nplot_confusion_matrix(cnf_matrix, classes=class_names,\n                      title='Random Forest Confusion matrix without PCA')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:21.286062Z","iopub.execute_input":"2024-11-10T15:30:21.286340Z","iopub.status.idle":"2024-11-10T15:30:21.980127Z","shell.execute_reply.started":"2024-11-10T15:30:21.286302Z","shell.execute_reply":"2024-11-10T15:30:21.979164Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 2. Classification Report","metadata":{}},{"cell_type":"code","source":"#classification report\nfrom sklearn.metrics import confusion_matrix, classification_report\ny_true = test_labels\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\nprint(classification_report(y_true, predictions_rf, target_names = class_names))","metadata":{"id":"0Yxb9RDxJEzi","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:21.981499Z","iopub.execute_input":"2024-11-10T15:30:21.981918Z","iopub.status.idle":"2024-11-10T15:30:21.998862Z","shell.execute_reply.started":"2024-11-10T15:30:21.981855Z","shell.execute_reply":"2024-11-10T15:30:21.997847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#classification report\nfrom sklearn.metrics import confusion_matrix, classification_report\ny_true = test_labels\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\nprint(classification_report(y_true, predictions_rf_no_pca, target_names = class_names))","metadata":{"id":"YcHKCYv4JEzi","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:21.999990Z","iopub.execute_input":"2024-11-10T15:30:22.000261Z","iopub.status.idle":"2024-11-10T15:30:22.016383Z","shell.execute_reply.started":"2024-11-10T15:30:22.000231Z","shell.execute_reply":"2024-11-10T15:30:22.015600Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 3. AUC","metadata":{}},{"cell_type":"code","source":"#AUC\nfrom sklearn.preprocessing import LabelBinarizer\nfrom sklearn.metrics import roc_auc_score\ndef multiclass_roc_auc_score(y_test, y_pred, average=\"macro\"):\n    lb = LabelBinarizer()\n    lb.fit(y_test)\n    y_test = lb.transform(y_test)\n    y_pred = lb.transform(y_pred)\n    return roc_auc_score(y_test, y_pred, average=average)\n\ny_true = test_labels\nmulticlass_roc_auc_score(y_true, predictions_rf)","metadata":{"id":"TXQQmtC2JEzi","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:22.017301Z","iopub.execute_input":"2024-11-10T15:30:22.017573Z","iopub.status.idle":"2024-11-10T15:30:22.041839Z","shell.execute_reply.started":"2024-11-10T15:30:22.017543Z","shell.execute_reply":"2024-11-10T15:30:22.040978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#AUC\nfrom sklearn.preprocessing import LabelBinarizer\nfrom sklearn.metrics import roc_auc_score\ndef multiclass_roc_auc_score(y_test, y_pred, average=\"macro\"):\n    lb = LabelBinarizer()\n    lb.fit(y_test)\n    y_test = lb.transform(y_test)\n    y_pred = lb.transform(y_pred)\n    return roc_auc_score(y_test, y_pred, average=average)\n\ny_true = test_labels\nmulticlass_roc_auc_score(y_true, predictions_rf_no_pca)","metadata":{"id":"1_kM265UJEzj","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:22.042967Z","iopub.execute_input":"2024-11-10T15:30:22.043227Z","iopub.status.idle":"2024-11-10T15:30:22.066615Z","shell.execute_reply.started":"2024-11-10T15:30:22.043197Z","shell.execute_reply":"2024-11-10T15:30:22.065613Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Comparative Analysis ","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Model names\nmodels = ['SVM with PCA', 'SVM without PCA', 'KNN with PCA', 'KNN without PCA', 'RF with PCA', 'RF without PCA']\n\n# Accuracy values\naccuracies = [accuracy, accuracy_no_pca, accuracy_knn, accuracy_knn_no_pca, accuracy_rf, accuracy_rf_no_pca]\n\n# Training durations\ntraining_durations = [training_duration, training_duration_no_pca, training_duration_knn, training_duration_knn_no_pca, training_duration_rf, training_duration_rf_no_pca]\n\n# Plotting accuracies\nplt.figure(figsize=(10, 5))\nplt.bar(models, accuracies, color=['blue', 'blue', 'green', 'green', 'red', 'red'])\nplt.xlabel('Models')\nplt.ylabel('Accuracy')\nplt.title('Model Accuracy Comparison')\nplt.xticks(rotation=45)\nplt.ylim(0.98, 1)\n\n# Adding accuracy values on top of the bars\nfor i, v in enumerate(accuracies):\n    plt.text(i, v + 0.001, f\"{v:.4f}\", ha='center', va='bottom')\n    \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:46:51.710632Z","iopub.execute_input":"2024-11-10T15:46:51.711446Z","iopub.status.idle":"2024-11-10T15:46:52.012554Z","shell.execute_reply.started":"2024-11-10T15:46:51.711403Z","shell.execute_reply":"2024-11-10T15:46:52.011685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# AUC scores\nauc_scores = [\n    multiclass_roc_auc_score(test_labels, predictions),\n    multiclass_roc_auc_score(test_labels, predictions_no_pca),\n    multiclass_roc_auc_score(test_labels, predictions_knn),\n    multiclass_roc_auc_score(test_labels, predictions_knn_no_pca),\n    multiclass_roc_auc_score(test_labels, predictions_rf),\n    multiclass_roc_auc_score(test_labels, predictions_rf_no_pca)\n]\n\n# Plotting AUC scores\nplt.figure(figsize=(10, 5))\nplt.bar(models, auc_scores, color=['blue', 'blue', 'green', 'green', 'red', 'red'])\nplt.xlabel('Models')\nplt.ylabel('AUC Score')\nplt.title('Model AUC Score Comparison')\nplt.xticks(rotation=45)\nplt.ylim(0.99, 1)\n\n# Adding AUC values on top of the bars\nfor i, v in enumerate(auc_scores):\n    plt.text(i, v + 0.001, f\"{v:.4f}\", ha='center', va='bottom')\n    \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:46:58.419437Z","iopub.execute_input":"2024-11-10T15:46:58.419823Z","iopub.status.idle":"2024-11-10T15:46:58.711321Z","shell.execute_reply.started":"2024-11-10T15:46:58.419785Z","shell.execute_reply":"2024-11-10T15:46:58.710432Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Testing","metadata":{}},{"cell_type":"code","source":"# Define the class names\nclass_names_folder = ['c0', 'c1', 'c2', 'c3', 'c4', 'c5', 'c6', 'c7', 'c8', 'c9']\n\n# Number of images to display from each class\nnum_images_per_class = 2\n\n# Loop through each class\nfor class_index in range(len(class_names_folder)):\n    class_label = class_names_folder[class_index]\n\n    # Get all image file paths in the current class folder\n    image_paths = glob.glob(f'/kaggle/input/state-farm-distracted-driver-detection/imgs/train/{class_label}/*.jpg')\n\n    # Randomly select the specified number of images from the current class\n    selected_image_paths = np.random.choice(image_paths, size=num_images_per_class, replace=False)\n\n    # Loop through the selected images in the current class\n    for image_path in selected_image_paths:\n        # Load and preprocess the image\n        img = cv2.imread(image_path)\n        img_resized = cv2.resize(img, (100, 100))/256\n        images = np.array([img_resized])\n\n        # Perform prediction\n        images = cnn.predict(images)\n        images = pca.transform(images)\n        result = svm.predict(images)\n\n        # Display the original image\n        plt.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n        plt.axis('off')\n        plt.show()\n\n        # Print the result\n        print(f\"Image: {image_path}\")\n        print(f\"Predicted Class: {class_names[result[0]]}\")\n        print()","metadata":{"id":"hRs5-HjRJEzj","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:22.470625Z","iopub.execute_input":"2024-11-10T15:30:22.471021Z","iopub.status.idle":"2024-11-10T15:30:28.588574Z","shell.execute_reply.started":"2024-11-10T15:30:22.470978Z","shell.execute_reply":"2024-11-10T15:30:28.587381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the class names\nclass_names_folder = ['c0', 'c1', 'c2', 'c3', 'c4', 'c5', 'c6', 'c7', 'c8', 'c9']\n\n# Number of images to display from each class\nnum_images_per_class = 2\n\n# Loop through each class\nfor class_index in range(len(class_names_folder)):\n    class_label = class_names_folder[class_index]\n\n    # Get all image file paths in the current class folder\n    image_paths = glob.glob(f'/kaggle/input/state-farm-distracted-driver-detection/imgs/train/{class_label}/*.jpg')\n\n    # Randomly select the specified number of images from the current class\n    selected_image_paths = np.random.choice(image_paths, size=num_images_per_class, replace=False)\n\n    # Loop through the selected images in the current class\n    for image_path in selected_image_paths:\n        # Load and preprocess the image\n        img = cv2.imread(image_path)\n        img_resized = cv2.resize(img, (100, 100))/256\n        images = np.array([img_resized])\n\n        # Perform prediction\n        images = cnn.predict(images)\n        result = svm_no_pca.predict(images)\n\n        # Display the original image\n        plt.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n        plt.axis('off')\n        plt.show()\n\n        # Print the result\n        print(f\"Image: {image_path}\")\n        print(f\"Predicted Class without PCA: {class_names[result[0]]}\")\n        print()","metadata":{"id":"3GMzlCGkJEzj","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:28.589773Z","iopub.execute_input":"2024-11-10T15:30:28.590096Z","iopub.status.idle":"2024-11-10T15:30:34.446801Z","shell.execute_reply.started":"2024-11-10T15:30:28.590063Z","shell.execute_reply":"2024-11-10T15:30:34.445923Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Saving the Model Files","metadata":{}},{"cell_type":"code","source":"with open('svm_manual_bisa_v4.4.pkl', 'wb') as f:\n    pickle.dump(svm, f)\nwith open('svm_manual_bisa_v4.4_no_pca.pkl', 'wb') as f:\n    pickle.dump(svm_no_pca, f)\nwith open('pca_manual_bisa_v4.4.pkl', 'wb') as f:\n    pickle.dump(pca, f)\ncnn.save('/kaggle/working/cnn_bisa_v4.4.h5')","metadata":{"id":"f2P1jHztJEzj","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T15:30:34.448000Z","iopub.execute_input":"2024-11-10T15:30:34.448301Z","iopub.status.idle":"2024-11-10T15:30:34.500302Z","shell.execute_reply.started":"2024-11-10T15:30:34.448270Z","shell.execute_reply":"2024-11-10T15:30:34.499541Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}