{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":5048,"databundleVersionId":868335,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:50:12.802475Z","iopub.execute_input":"2024-11-25T17:50:12.802806Z","iopub.status.idle":"2024-11-25T17:50:12.807177Z","shell.execute_reply.started":"2024-11-25T17:50:12.802778Z","shell.execute_reply":"2024-11-25T17:50:12.806228Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1>Distracted Driver Detection System</h1>\n\n**Project by:**<br>\nShivank Garg - 12112024<br>\nShubham Kumar - 12112017<br>","metadata":{}},{"cell_type":"code","source":"# IMPORTANT: SOME KAGGLE DATA SOURCES ARE PRIVATE\n# RUN THIS CELL IN ORDER TO IMPORT YOUR KAGGLE DATA SOURCES.\nimport kagglehub\nkagglehub.login()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:50:12.808514Z","iopub.execute_input":"2024-11-25T17:50:12.808761Z","iopub.status.idle":"2024-11-25T17:50:13.019510Z","shell.execute_reply.started":"2024-11-25T17:50:12.808734Z","shell.execute_reply":"2024-11-25T17:50:13.018586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import dataset\nstate_farm_distracted_driver_detection_path = kagglehub.competition_download('state-farm-distracted-driver-detection')\n\nprint('Data source import complete.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:50:13.020913Z","iopub.execute_input":"2024-11-25T17:50:13.021188Z","iopub.status.idle":"2024-11-25T17:50:14.319983Z","shell.execute_reply.started":"2024-11-25T17:50:13.021163Z","shell.execute_reply":"2024-11-25T17:50:14.318941Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h3>IMPORTING LIBRARIES</h3>","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\nimport itertools\nimport pandas as pd\nimport glob\nimport pickle\nimport os\nimport time\nfrom keras.models import Sequential, save_model\nfrom keras.layers import Conv2D, MaxPooling2D, Flatten, Dense\nfrom keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import check_random_state\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:50:14.321014Z","iopub.execute_input":"2024-11-25T17:50:14.321368Z","iopub.status.idle":"2024-11-25T17:50:26.505014Z","shell.execute_reply.started":"2024-11-25T17:50:14.321323Z","shell.execute_reply":"2024-11-25T17:50:26.504082Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h3>LOADING DATASET</h3>","metadata":{}},{"cell_type":"code","source":"main_path = '/kaggle/input/state-farm-distracted-driver-detection/imgs/train/c'\nclass_labels = []\nimages = []\n\n# Iterates for each class\nfor class_index in range(10):\n    class_path = main_path + str(class_index)  # Path to the current class directory\n    for root, dirs, files in os.walk(class_path):\n        # Process files in the current class\n        for filename in tqdm(files, desc='Processing class ' + str(class_index)):\n            image_path = os.path.join(class_path, filename)\n            img = cv2.imread(image_path)\n            img = cv2.resize(img, (100, 100)) / 255\n            images.append(img)\n            class_labels.append(class_index)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:50:26.507056Z","iopub.execute_input":"2024-11-25T17:50:26.507579Z","iopub.status.idle":"2024-11-25T17:53:32.990848Z","shell.execute_reply.started":"2024-11-25T17:50:26.507551Z","shell.execute_reply":"2024-11-25T17:53:32.990021Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h3>DATA PREPROCESSING</h3>","metadata":{}},{"cell_type":"code","source":"train_images, test_images, train_labels, test_labels = train_test_split(np.array(images), np.array(class_labels), test_size = 0.2, shuffle=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:53:32.991862Z","iopub.execute_input":"2024-11-25T17:53:32.992140Z","iopub.status.idle":"2024-11-25T17:53:36.090651Z","shell.execute_reply.started":"2024-11-25T17:53:32.992090Z","shell.execute_reply":"2024-11-25T17:53:36.089691Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h3>TRAIN TEST SPLIT</h3>","metadata":{}},{"cell_type":"markdown","source":"<h3>FEATURE EXTRACTION USING CNN</h3>","metadata":{}},{"cell_type":"code","source":"# Define the custom CNN architecture\ncnn = Sequential()\ncnn.add(Conv2D(32, (3, 3), activation='relu', input_shape=(100, 100, 3)))\ncnn.add(MaxPooling2D((2, 2)))\ncnn.add(Conv2D(64, (3, 3), activation='relu'))\ncnn.add(MaxPooling2D((2, 2)))\ncnn.add(Conv2D(128, (3, 3), activation='relu'))\ncnn.add(MaxPooling2D((2, 2)))\ncnn.add(Flatten())\ncnn.add(Dense(256, activation='relu'))\ncnn.add(Dense(128, activation='relu'))\ncnn.add(Dense(10, activation='softmax'))\n\n# Compile the model\ncnn.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:53:36.091828Z","iopub.execute_input":"2024-11-25T17:53:36.092189Z","iopub.status.idle":"2024-11-25T17:53:36.905524Z","shell.execute_reply.started":"2024-11-25T17:53:36.092151Z","shell.execute_reply":"2024-11-25T17:53:36.904690Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cnn.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:53:36.906409Z","iopub.execute_input":"2024-11-25T17:53:36.906667Z","iopub.status.idle":"2024-11-25T17:53:36.930477Z","shell.execute_reply.started":"2024-11-25T17:53:36.906642Z","shell.execute_reply":"2024-11-25T17:53:36.929644Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h3>ONE HOT ENCODING</h3>","metadata":{}},{"cell_type":"code","source":"# Encode the target labels as one-hot vectors\ntrain_labels_encoded = to_categorical(train_labels, num_classes=10)\ntest_labels_encoded = to_categorical(test_labels, num_classes=10)\n\ncnn.fit(train_images, train_labels_encoded, epochs=10, batch_size=32, validation_data=(test_images, test_labels_encoded))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:53:36.931460Z","iopub.execute_input":"2024-11-25T17:53:36.931725Z","iopub.status.idle":"2024-11-25T17:54:36.679921Z","shell.execute_reply.started":"2024-11-25T17:53:36.931699Z","shell.execute_reply":"2024-11-25T17:54:36.678989Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**REMOVING THE FINAL FULLY CONNECTED LAYER TO GET THE 128 FEATURES EXTRACTED FROM PREVIOUS LAYER**","metadata":{}},{"cell_type":"code","source":"# Remove the last layer from the CNN model\ncnn = Sequential(cnn.layers[:-1])\n\n# Preventing the weights from being updated\nfor layer in cnn.layers:\n    layer.trainable = False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:54:36.685108Z","iopub.execute_input":"2024-11-25T17:54:36.685381Z","iopub.status.idle":"2024-11-25T17:54:36.696847Z","shell.execute_reply.started":"2024-11-25T17:54:36.685357Z","shell.execute_reply":"2024-11-25T17:54:36.695869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cnn.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:54:36.699679Z","iopub.execute_input":"2024-11-25T17:54:36.699943Z","iopub.status.idle":"2024-11-25T17:54:36.724926Z","shell.execute_reply.started":"2024-11-25T17:54:36.699918Z","shell.execute_reply":"2024-11-25T17:54:36.724125Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h3>EXTRACTED FEATURES</h3>","metadata":{}},{"cell_type":"code","source":"train_features = cnn.predict(train_images)\ntest_features = cnn.predict(test_images)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:54:36.725952Z","iopub.execute_input":"2024-11-25T17:54:36.726262Z","iopub.status.idle":"2024-11-25T17:54:46.915855Z","shell.execute_reply.started":"2024-11-25T17:54:36.726222Z","shell.execute_reply":"2024-11-25T17:54:46.915146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_features.shape)\nprint(test_features.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:54:46.917094Z","iopub.execute_input":"2024-11-25T17:54:46.917385Z","iopub.status.idle":"2024-11-25T17:54:46.921906Z","shell.execute_reply.started":"2024-11-25T17:54:46.917359Z","shell.execute_reply":"2024-11-25T17:54:46.921032Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h3>FEATURE REDUCTION USING PCA</h3>\n<h5>Here, we will reduce the extracted features(128) to 16 features using Principal Component Analysis</h5>","metadata":{}},{"cell_type":"markdown","source":"<h3>PCA IMPLEMENTATION</h3>","metadata":{}},{"cell_type":"code","source":"class PCA:\n    def __init__(self, n_components):\n        self.n_components = n_components\n        self.components = None\n        self.mean = None\n\n    def fit(self, X):\n        # Calculate the mean of each feature\n        self.mean = np.mean(X, axis=0)\n\n        # Center the data by subtracting the mean from each feature\n        X = X - self.mean\n\n        # Calculate the covariance matrix\n        cov = np.cov(X.T)\n\n        # Calculate the eigenvalues and eigenvectors of the covariance matrix\n        eigenvalues, eigenvectors = np.linalg.eig(cov)\n\n        # Sort the eigenvectors by their corresponding eigenvalues in descending order\n        eigenvectors = eigenvectors.T\n        idxs = np.argsort(eigenvalues)[::-1]\n        eigenvectors = eigenvectors[idxs]\n        eigenvalues = eigenvalues[idxs]\n\n        # Store the first n_components eigenvectors as the components\n        self.components = eigenvectors[0:self.n_components]\n\n    def transform(self, X):\n        # Center the data by subtracting the mean from each feature\n        X = X - self.mean\n\n        # Project the data onto the components\n        return np.dot(X, self.components.T)\n\n    def fit_transform(self, X):\n        # Calculate the mean of each feature\n        self.mean = np.mean(X, axis=0)\n\n        # Center the data by subtracting the mean from each feature\n        X = X - self.mean\n\n        # Calculate the covariance matrix\n        cov = np.cov(X.T)\n\n        # Calculate the eigenvalues and eigenvectors of the covariance matrix\n        eigenvalues, eigenvectors = np.linalg.eig(cov)\n\n        # Sort the eigenvectors by their corresponding eigenvalues in descending order\n        eigenvectors = eigenvectors.T\n        idxs = np.argsort(eigenvalues)[::-1]\n        eigenvectors = eigenvectors[idxs]\n        eigenvalues = eigenvalues[idxs]\n\n        # Store the first n_components eigenvectors as the components\n        self.components = eigenvectors[0:self.n_components]\n\n        # Project the data onto the components\n        return np.dot(X, self.components.T)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:54:46.923117Z","iopub.execute_input":"2024-11-25T17:54:46.923475Z","iopub.status.idle":"2024-11-25T17:54:46.937816Z","shell.execute_reply.started":"2024-11-25T17:54:46.923445Z","shell.execute_reply":"2024-11-25T17:54:46.937071Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h3>APPLYING PCA</h3>","metadata":{}},{"cell_type":"code","source":"# Create a PCA object with 16 components\npca = PCA(n_components=16)\ntrain_features_reduced = pca.fit_transform(train_features)\ntest_features_reduced = pca.transform(test_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:54:46.938968Z","iopub.execute_input":"2024-11-25T17:54:46.939664Z","iopub.status.idle":"2024-11-25T17:54:47.023993Z","shell.execute_reply.started":"2024-11-25T17:54:46.939618Z","shell.execute_reply":"2024-11-25T17:54:47.022699Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_features_reduced.shape)\nprint(test_features_reduced.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:54:47.025381Z","iopub.execute_input":"2024-11-25T17:54:47.025847Z","iopub.status.idle":"2024-11-25T17:54:47.037820Z","shell.execute_reply.started":"2024-11-25T17:54:47.025795Z","shell.execute_reply":"2024-11-25T17:54:47.036732Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h3>SAVING THE REDUCED FEATURES</h3>","metadata":{}},{"cell_type":"code","source":"train_features_reduced_save = pd.DataFrame(train_features_reduced)\ntrain_features_reduced_save.to_csv('train_features_reduced_save.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:54:47.039051Z","iopub.execute_input":"2024-11-25T17:54:47.039481Z","iopub.status.idle":"2024-11-25T17:54:47.492232Z","shell.execute_reply.started":"2024-11-25T17:54:47.039431Z","shell.execute_reply":"2024-11-25T17:54:47.491532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_features_reduced_save = pd.DataFrame(test_features_reduced)\ntest_features_reduced_save.to_csv('test_features_reduced_save.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:54:47.493298Z","iopub.execute_input":"2024-11-25T17:54:47.493598Z","iopub.status.idle":"2024-11-25T17:54:47.594226Z","shell.execute_reply.started":"2024-11-25T17:54:47.493555Z","shell.execute_reply":"2024-11-25T17:54:47.593561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_tar_save = pd.DataFrame(train_labels)\ntrain_tar_save.to_csv('train_tar.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:54:47.595099Z","iopub.execute_input":"2024-11-25T17:54:47.595370Z","iopub.status.idle":"2024-11-25T17:54:47.606579Z","shell.execute_reply.started":"2024-11-25T17:54:47.595345Z","shell.execute_reply":"2024-11-25T17:54:47.605751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_tar_save = pd.DataFrame(test_labels)\ntest_tar_save.to_csv('test_tar.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:54:47.607776Z","iopub.execute_input":"2024-11-25T17:54:47.608643Z","iopub.status.idle":"2024-11-25T17:54:47.615067Z","shell.execute_reply.started":"2024-11-25T17:54:47.608606Z","shell.execute_reply":"2024-11-25T17:54:47.614341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_features_reduced[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:54:47.616049Z","iopub.execute_input":"2024-11-25T17:54:47.616447Z","iopub.status.idle":"2024-11-25T17:54:47.627428Z","shell.execute_reply.started":"2024-11-25T17:54:47.616406Z","shell.execute_reply":"2024-11-25T17:54:47.626672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(test_features_reduced[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:54:47.628377Z","iopub.execute_input":"2024-11-25T17:54:47.628719Z","iopub.status.idle":"2024-11-25T17:54:47.638732Z","shell.execute_reply.started":"2024-11-25T17:54:47.628682Z","shell.execute_reply":"2024-11-25T17:54:47.637860Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h2>CLASSIFICATION</h2>","metadata":{}},{"cell_type":"markdown","source":"<H3>1. SUPPORT VECTOR MACHINE (SVM)</H3>","metadata":{}},{"cell_type":"markdown","source":"<H3>1.1. SVM - OUR IMPLEMENTATION</H3>","metadata":{}},{"cell_type":"code","source":"class SVM:\n    def __init__(self, C=1, max_iter=50, tol=0.05,\n                 random_state=None, verbose=0):\n        self.C = C\n        self.max_iter = max_iter\n        self.tol = tol,\n        self.random_state = random_state\n        self.verbose = verbose\n\n    def projection_simplex(self, v, z=1):\n        n_features = v.shape[0]\n        u = np.sort(v)[::-1]\n        cssv = np.cumsum(u) - z\n        ind = np.arange(n_features) + 1\n        cond = u - cssv / ind > 0\n        rho = ind[cond][-1]\n        theta = cssv[cond][-1] / float(rho)\n        w = np.maximum(v - theta, 0)\n        return w\n\n    def _partial_gradient(self, X, y, i):\n        # Partial gradient for the ith sample.\n        g = np.dot(X[i], self.coef_.T) + 1\n        g[y[i]] -= 1\n        return g\n\n    def _violation(self, g, y, i):\n        # Optimality violation for the ith sample.\n        smallest = np.inf\n        for k in range(g.shape[0]):\n            if k == y[i] and self.dual_coef_[k, i] >= self.C:\n                continue\n            elif k != y[i] and self.dual_coef_[k, i] >= 0:\n                continue\n\n            smallest = min(smallest, g[k])\n\n        return g.max() - smallest\n\n    def _solve_subproblem(self, g, y, norms, i):\n        # Prepare inputs to the projection.\n        Ci = np.zeros(g.shape[0])\n        Ci[y[i]] = self.C\n        beta_hat = norms[i] * (Ci - self.dual_coef_[:, i]) + g / norms[i]\n        z = self.C * norms[i]\n\n        # Compute projection onto the simplex.\n        beta = self.projection_simplex(beta_hat, z)\n\n        return Ci - self.dual_coef_[:, i] - beta / norms[i]\n\n    def fit(self, X, y):\n        n_samples, n_features = X.shape\n\n        n_classes = np.unique(y).size\n        self.dual_coef_ = np.zeros((n_classes, n_samples), dtype=np.float64)\n        self.coef_ = np.zeros((n_classes, n_features))\n\n        # Pre-compute norms.\n        norms = np.sqrt(np.sum(X ** 2, axis=1))\n\n        # Shuffle sample indices.\n        rs = check_random_state(self.random_state)\n        ind = np.arange(n_samples)\n        rs.shuffle(ind)\n\n        violation_init = None\n        for it in range(self.max_iter):\n            violation_sum = 0\n\n            for ii in range(n_samples):\n                i = ind[ii]\n\n                # All-zero samples can be safely ignored.\n                if norms[i] == 0:\n                    continue\n\n                g = self._partial_gradient(X, y, i)\n                v = self._violation(g, y, i)\n                violation_sum += v\n\n                if v < 1e-12:\n                    continue\n\n                # Solve subproblem for the ith sample.\n                delta = self._solve_subproblem(g, y, norms, i)\n\n                # Update primal and dual coefficients.\n                self.coef_ = self.coef_.astype(np.complex128)\n                self.dual_coef_ = self.dual_coef_.astype(np.complex128)\n                delta = delta.astype(np.complex128)\n\n                self.coef_ += np.multiply(delta[:, np.newaxis], X[i][:, np.newaxis].conj().T)\n                self.dual_coef_[:, i] += delta\n\n            if it == 0:\n                violation_init = violation_sum\n\n            vratio = violation_sum / violation_init\n\n            if vratio < self.tol:\n                if self.verbose >= 1:\n                    print(\"Converged\")\n                break\n\n        return self\n\n    def predict(self, X):\n        decision = np.dot(X, self.coef_.T)\n        return decision.argmax(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:54:47.639908Z","iopub.execute_input":"2024-11-25T17:54:47.640241Z","iopub.status.idle":"2024-11-25T17:54:47.654086Z","shell.execute_reply.started":"2024-11-25T17:54:47.640213Z","shell.execute_reply":"2024-11-25T17:54:47.653431Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H3>SVM WITH PCA</H3>","metadata":{}},{"cell_type":"code","source":"svm = SVM(C=1, tol=0.001, max_iter=100, random_state=0, verbose=1)\n\nstart_time = time.time()\nsvm.fit(train_features_reduced, train_labels)\nend_time = time.time()\n\ntraining_duration = end_time - start_time\nprint(\"Training duration:\", training_duration, \"seconds\")\n\npredictions = svm.predict(test_features_reduced)\naccuracy = np.mean(predictions == test_labels)\nprint(\"Accuracy:\", accuracy)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:54:47.655037Z","iopub.execute_input":"2024-11-25T17:54:47.655575Z","iopub.status.idle":"2024-11-25T17:55:18.998029Z","shell.execute_reply.started":"2024-11-25T17:54:47.655539Z","shell.execute_reply":"2024-11-25T17:55:18.994836Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H3>SVM WITHOUT PCA</H3>","metadata":{}},{"cell_type":"code","source":"svm_no_pca = SVM(C=1, tol=0.001, max_iter=100, random_state=0, verbose=1)\n\nstart_time_no_pca = time.time()\nsvm_no_pca.fit(train_features, train_labels)\nend_time_no_pca = time.time()\n\ntraining_duration_no_pca = end_time_no_pca - start_time_no_pca\nprint(\"Training duration without PCA:\", training_duration_no_pca, \"seconds\")\n\npredictions_no_pca = svm_no_pca.predict(test_features)\naccuracy_no_pca = np.mean(predictions_no_pca == test_labels)\nprint(\"Accuracy without PCA:\", accuracy_no_pca)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:55:18.999511Z","iopub.execute_input":"2024-11-25T17:55:18.999991Z","iopub.status.idle":"2024-11-25T17:55:56.441330Z","shell.execute_reply.started":"2024-11-25T17:55:18.999937Z","shell.execute_reply":"2024-11-25T17:55:56.439502Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H2>1.2. SVM FROM SKLEARN LIBRARY</H2>","metadata":{}},{"cell_type":"markdown","source":"<H3>SVM WITH PCA</H3>","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import SVC\nfrom sklearn.metrics import accuracy_score\n\n# Create an SVM classifier with a linear kernel\nsvm_sklearn = SVC(kernel='linear', C=1, max_iter=100, random_state=0, verbose=0)\n\n# Train the SVM classifier on the reduced features (with PCA)\nstart_time_svm_sklearn = time.time()\nsvm_sklearn.fit(train_features_reduced, train_labels)\nend_time_svm_sklearn = time.time()\n\ntraining_duration_svm_sklearn = end_time_svm_sklearn- start_time_svm_sklearn\nprint(\"Training duration with PCA:\", training_duration_svm_sklearn, \"seconds\")\n\n# Predict the labels for the test set\npredictions_svm_sklearn = svm_sklearn.predict(test_features_reduced)\naccuracy_svm_sklearn = accuracy_score(test_labels, predictions_svm_sklearn)\nprint(\"Accuracy with PCA:\", accuracy_svm_sklearn)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:55:56.443221Z","iopub.execute_input":"2024-11-25T17:55:56.444991Z","iopub.status.idle":"2024-11-25T17:55:56.689187Z","shell.execute_reply.started":"2024-11-25T17:55:56.444937Z","shell.execute_reply":"2024-11-25T17:55:56.688243Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H3>SVM WITHOUT PCA</H3>","metadata":{}},{"cell_type":"code","source":"# Train the SVM classifier on the original features (without PCA)\nstart_time_svm_sklearn_no_pca = time.time()\nsvm_sklearn.fit(train_features, train_labels)\nend_time_svm_sklearn_no_pca = time.time()\n\ntraining_duration_svm_sklearn_no_pca = end_time_svm_sklearn_no_pca - start_time_svm_sklearn_no_pca\nprint(\"Training duration without PCA:\", training_duration_svm_sklearn_no_pca, \"seconds\")\n\n# Predict the labels for the test set\npredictions_svm_sklearn_no_pca = svm_sklearn.predict(test_features)\naccuracy_svm_sklearn_no_pca = accuracy_score(test_labels, predictions_svm_sklearn_no_pca)\nprint(\"Accuracy without PCA:\", accuracy_svm_sklearn_no_pca)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:55:56.690244Z","iopub.execute_input":"2024-11-25T17:55:56.690526Z","iopub.status.idle":"2024-11-25T17:55:57.045896Z","shell.execute_reply.started":"2024-11-25T17:55:56.690489Z","shell.execute_reply":"2024-11-25T17:55:57.045048Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H2>SVM MODEL EVALUATION</H2>","metadata":{}},{"cell_type":"markdown","source":"<H3>1. CONFUSION MATRIX</H3>","metadata":{}},{"cell_type":"code","source":"def plot_confusion_matrix(cm, classes,\n                          normalize=False,\n                          title='Confusion matrix',\n                          cmap=plt.cm.Blues):\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n        print(\"Normalized confusion matrix\")\n    else:\n        print('Confusion matrix')\n\n    #print(cm)\n    plt.figure(figsize = (10,10))\n    plt.imshow(cm, interpolation='nearest', cmap=cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation=45)\n    plt.yticks(tick_marks, classes)\n\n    fmt = '.2f' if normalize else 'd'\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, format(cm[i, j], fmt),\n                 horizontalalignment=\"center\",\n                 color=\"white\" if cm[i, j] > thresh else \"black\")\n\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    plt.tight_layout()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:55:57.047891Z","iopub.execute_input":"2024-11-25T17:55:57.048616Z","iopub.status.idle":"2024-11-25T17:55:57.056241Z","shell.execute_reply.started":"2024-11-25T17:55:57.048576Z","shell.execute_reply":"2024-11-25T17:55:57.055407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n# Compute confusion matrix\ncnf_matrix = confusion_matrix(test_labels, predictions)\nnp.set_printoptions(precision=2)\n\n# Plot non-normalized confusion matrix\nplt.figure()\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\n\nplot_confusion_matrix(cnf_matrix, classes=class_names,\n                      title='Confusion matrix with PCA')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:55:57.062602Z","iopub.execute_input":"2024-11-25T17:55:57.062843Z","iopub.status.idle":"2024-11-25T17:55:57.701480Z","shell.execute_reply.started":"2024-11-25T17:55:57.062820Z","shell.execute_reply":"2024-11-25T17:55:57.700654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n# Compute confusion matrix\ncnf_matrix = confusion_matrix(test_labels, predictions_no_pca)\nnp.set_printoptions(precision=2)\n\n# Plot non-normalized confusion matrix\nplt.figure()\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\n\nplot_confusion_matrix(cnf_matrix, classes=class_names,\n                      title='Confusion matrix without PCA')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:55:57.702705Z","iopub.execute_input":"2024-11-25T17:55:57.703045Z","iopub.status.idle":"2024-11-25T17:55:58.306051Z","shell.execute_reply.started":"2024-11-25T17:55:57.703009Z","shell.execute_reply":"2024-11-25T17:55:58.305171Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H3>2. CLASSIFICATION REPORT</H3>","metadata":{}},{"cell_type":"code","source":"#classification report\nfrom sklearn.metrics import confusion_matrix, classification_report\ny_true = test_labels\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\nprint(classification_report(y_true, predictions, target_names = class_names))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:55:58.307097Z","iopub.execute_input":"2024-11-25T17:55:58.307396Z","iopub.status.idle":"2024-11-25T17:55:58.321506Z","shell.execute_reply.started":"2024-11-25T17:55:58.307370Z","shell.execute_reply":"2024-11-25T17:55:58.320702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#classification report\nfrom sklearn.metrics import confusion_matrix, classification_report\ny_true = test_labels\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\nprint(classification_report(y_true, predictions_no_pca, target_names = class_names))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:55:58.322474Z","iopub.execute_input":"2024-11-25T17:55:58.322743Z","iopub.status.idle":"2024-11-25T17:55:58.338178Z","shell.execute_reply.started":"2024-11-25T17:55:58.322718Z","shell.execute_reply":"2024-11-25T17:55:58.337427Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H3>3. AUC (AREA UNDER THE CURVE)</H3>","metadata":{}},{"cell_type":"code","source":"#AUC\nfrom sklearn.preprocessing import LabelBinarizer\nfrom sklearn.metrics import roc_auc_score\ndef multiclass_roc_auc_score(y_test, y_pred, average=\"macro\"):\n    lb = LabelBinarizer()\n    lb.fit(y_test)\n    y_test = lb.transform(y_test)\n    y_pred = lb.transform(y_pred)\n    return roc_auc_score(y_test, y_pred, average=average)\n\ny_true = test_labels\nmulticlass_roc_auc_score(y_true, predictions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:55:58.339061Z","iopub.execute_input":"2024-11-25T17:55:58.339340Z","iopub.status.idle":"2024-11-25T17:55:58.362020Z","shell.execute_reply.started":"2024-11-25T17:55:58.339316Z","shell.execute_reply":"2024-11-25T17:55:58.361313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#AUC\nfrom sklearn.preprocessing import LabelBinarizer\nfrom sklearn.metrics import roc_auc_score\ndef multiclass_roc_auc_score(y_test, y_pred, average=\"macro\"):\n    lb = LabelBinarizer()\n    lb.fit(y_test)\n    y_test = lb.transform(y_test)\n    y_pred = lb.transform(y_pred)\n    return roc_auc_score(y_test, y_pred, average=average)\n\ny_true = test_labels\nmulticlass_roc_auc_score(y_true, predictions_no_pca)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:55:58.362985Z","iopub.execute_input":"2024-11-25T17:55:58.363290Z","iopub.status.idle":"2024-11-25T17:55:58.384275Z","shell.execute_reply.started":"2024-11-25T17:55:58.363262Z","shell.execute_reply":"2024-11-25T17:55:58.383405Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H2>2. K NEAREST NEIGHBOURS (KNN)</H2>","metadata":{}},{"cell_type":"markdown","source":"<H3>KNN WITH PCA</H3>","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\n# Create a KNN classifier with 5 neighbors\nknn = KNeighborsClassifier(n_neighbors=5)\n\n# Train the KNN classifier on the reduced features(with PCA)\nstart_time_knn = time.time()\nknn.fit(train_features_reduced, train_labels)\nend_time_knn = time.time() \n\ntraining_duration_knn = end_time_knn - start_time_knn\nprint(\"Training duration with PCA:\", training_duration_knn, \"seconds\")\n\n# Predict the labels for the test set\npredictions_knn = knn.predict(test_features_reduced)\naccuracy_knn = np.mean(predictions_knn == test_labels)\nprint(\"Accuracy with PCA:\", accuracy_knn)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:55:58.385377Z","iopub.execute_input":"2024-11-25T17:55:58.385637Z","iopub.status.idle":"2024-11-25T17:55:58.861823Z","shell.execute_reply.started":"2024-11-25T17:55:58.385612Z","shell.execute_reply":"2024-11-25T17:55:58.860857Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H3>KNN WITHOUT PCA</H3>","metadata":{}},{"cell_type":"code","source":"# Train the KNN classifier on the original features(without PCA)\nstart_time_knn_no_pca = time.time()\nknn.fit(train_features, train_labels)\nend_time_knn_no_pca = time.time()\n\ntraining_duration_knn_no_pca = end_time_knn_no_pca - start_time_knn_no_pca\nprint(\"Training duration without PCA:\", training_duration_knn_no_pca, \"seconds\")\n\n# Predict the labels for the test set\npredictions_knn_no_pca = knn.predict(test_features)\naccuracy_knn_no_pca = np.mean(predictions_knn_no_pca == test_labels)\nprint(\"Accuracy without PCA:\", accuracy_knn_no_pca)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:55:58.863055Z","iopub.execute_input":"2024-11-25T17:55:58.863517Z","iopub.status.idle":"2024-11-25T17:55:59.352766Z","shell.execute_reply.started":"2024-11-25T17:55:58.863476Z","shell.execute_reply":"2024-11-25T17:55:59.351886Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H3>KNN MODEL EVALUATION</H3>","metadata":{}},{"cell_type":"markdown","source":"<H3>1. CONFUSION MATRIX</H3>","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n# Compute confusion matrix\ncnf_matrix = confusion_matrix(test_labels, predictions_knn)\nnp.set_printoptions(precision=2)\n\n# Plot non-normalized confusion matrix\nplt.figure()\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\n\nplot_confusion_matrix(cnf_matrix, classes=class_names,\n                      title='KNN Confusion matrix with PCA')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:55:59.353850Z","iopub.execute_input":"2024-11-25T17:55:59.354224Z","iopub.status.idle":"2024-11-25T17:55:59.958263Z","shell.execute_reply.started":"2024-11-25T17:55:59.354182Z","shell.execute_reply":"2024-11-25T17:55:59.957489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n# Compute confusion matrix\ncnf_matrix = confusion_matrix(test_labels, predictions_knn_no_pca)\nnp.set_printoptions(precision=2)\n\n# Plot non-normalized confusion matrix\nplt.figure()\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\n\nplot_confusion_matrix(cnf_matrix, classes=class_names,\n                      title='KNN Confusion matrix without PCA')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:55:59.959516Z","iopub.execute_input":"2024-11-25T17:55:59.959895Z","iopub.status.idle":"2024-11-25T17:56:00.576023Z","shell.execute_reply.started":"2024-11-25T17:55:59.959857Z","shell.execute_reply":"2024-11-25T17:56:00.575084Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H3>2. CLASSIFICATION REPORT</H3>","metadata":{}},{"cell_type":"code","source":"#classification report\nfrom sklearn.metrics import confusion_matrix, classification_report\ny_true = test_labels\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\nprint(classification_report(y_true, predictions_knn, target_names = class_names))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:56:00.577005Z","iopub.execute_input":"2024-11-25T17:56:00.577266Z","iopub.status.idle":"2024-11-25T17:56:00.593036Z","shell.execute_reply.started":"2024-11-25T17:56:00.577241Z","shell.execute_reply":"2024-11-25T17:56:00.592298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#classification report\nfrom sklearn.metrics import confusion_matrix, classification_report\ny_true = test_labels\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\nprint(classification_report(y_true, predictions_knn_no_pca, target_names = class_names))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:56:00.593890Z","iopub.execute_input":"2024-11-25T17:56:00.594138Z","iopub.status.idle":"2024-11-25T17:56:00.608264Z","shell.execute_reply.started":"2024-11-25T17:56:00.594085Z","shell.execute_reply":"2024-11-25T17:56:00.607534Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H3>3. AUC</H3>","metadata":{}},{"cell_type":"code","source":"#AUC\nfrom sklearn.preprocessing import LabelBinarizer\nfrom sklearn.metrics import roc_auc_score\ndef multiclass_roc_auc_score(y_test, y_pred, average=\"macro\"):\n    lb = LabelBinarizer()\n    lb.fit(y_test)\n    y_test = lb.transform(y_test)\n    y_pred = lb.transform(y_pred)\n    return roc_auc_score(y_test, y_pred, average=average)\n\ny_true = test_labels\nmulticlass_roc_auc_score(y_true, predictions_knn)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:56:00.609366Z","iopub.execute_input":"2024-11-25T17:56:00.610304Z","iopub.status.idle":"2024-11-25T17:56:00.632360Z","shell.execute_reply.started":"2024-11-25T17:56:00.610264Z","shell.execute_reply":"2024-11-25T17:56:00.631506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#AUC\nfrom sklearn.preprocessing import LabelBinarizer\nfrom sklearn.metrics import roc_auc_score\ndef multiclass_roc_auc_score(y_test, y_pred, average=\"macro\"):\n    lb = LabelBinarizer()\n    lb.fit(y_test)\n    y_test = lb.transform(y_test)\n    y_pred = lb.transform(y_pred)\n    return roc_auc_score(y_test, y_pred, average=average)\n\ny_true = test_labels\nmulticlass_roc_auc_score(y_true, predictions_knn_no_pca)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:56:00.633327Z","iopub.execute_input":"2024-11-25T17:56:00.633599Z","iopub.status.idle":"2024-11-25T17:56:00.654401Z","shell.execute_reply.started":"2024-11-25T17:56:00.633573Z","shell.execute_reply":"2024-11-25T17:56:00.653498Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H2>3. RANDOM FOREST</H2>","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score\n\n# Create a Random Forest classifier\nrf = RandomForestClassifier(n_estimators=100, random_state=0)\n\n# Train the Random Forest classifier on the reduced features(with PCA)\nstart_time_rf = time.time()\nrf.fit(train_features_reduced, train_labels)\nend_time_rf = time.time()\n\ntraining_duration_rf = end_time_rf - start_time_rf\nprint(\"Training duration with PCA:\", training_duration_rf, \"seconds\")\n\n# Predict the labels for the test set\npredictions_rf = rf.predict(test_features_reduced)\naccuracy_rf = accuracy_score(test_labels, predictions_rf)\nprint(\"Accuracy with PCA:\", accuracy_rf)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:56:00.655482Z","iopub.execute_input":"2024-11-25T17:56:00.656136Z","iopub.status.idle":"2024-11-25T17:56:05.782480Z","shell.execute_reply.started":"2024-11-25T17:56:00.656069Z","shell.execute_reply":"2024-11-25T17:56:05.781524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the Random Forest classifier on the original features(without PCA)\nstart_time_rf_no_pca = time.time()\nrf.fit(train_features, train_labels)\nend_time_rf_no_pca = time.time()\n\ntraining_duration_rf_no_pca = end_time_rf_no_pca - start_time_rf_no_pca\nprint(\"Training duration without PCA:\", training_duration_rf_no_pca, \"seconds\")\n\n# Predict the labels for the test set\npredictions_rf_no_pca = rf.predict(test_features)\naccuracy_rf_no_pca = accuracy_score(test_labels, predictions_rf_no_pca)\nprint(\"Accuracy without PCA:\", accuracy_rf_no_pca)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:56:05.783452Z","iopub.execute_input":"2024-11-25T17:56:05.783749Z","iopub.status.idle":"2024-11-25T17:56:11.372722Z","shell.execute_reply.started":"2024-11-25T17:56:05.783722Z","shell.execute_reply":"2024-11-25T17:56:11.371862Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H2>RANDOM FOREST MODEL EVALUATION</H2>","metadata":{}},{"cell_type":"markdown","source":"<H3>1. CONFUSION MATRIX</H3>","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n# Compute confusion matrix\ncnf_matrix = confusion_matrix(test_labels, predictions_rf)\nnp.set_printoptions(precision=2)\n\n# Plot non-normalized confusion matrix\nplt.figure()\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\n\nplot_confusion_matrix(cnf_matrix, classes=class_names,\n                      title='Random Forest Confusion matrix with PCA')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:56:11.373901Z","iopub.execute_input":"2024-11-25T17:56:11.374357Z","iopub.status.idle":"2024-11-25T17:56:12.173948Z","shell.execute_reply.started":"2024-11-25T17:56:11.374317Z","shell.execute_reply":"2024-11-25T17:56:12.173039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n# Compute confusion matrix\ncnf_matrix = confusion_matrix(test_labels, predictions_rf_no_pca)\nnp.set_printoptions(precision=2)\n\n# Plot non-normalized confusion matrix\nplt.figure()\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\n\nplot_confusion_matrix(cnf_matrix, classes=class_names,\n                      title='Random Forest Confusion matrix without PCA')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:56:12.174983Z","iopub.execute_input":"2024-11-25T17:56:12.175265Z","iopub.status.idle":"2024-11-25T17:56:12.780477Z","shell.execute_reply.started":"2024-11-25T17:56:12.175240Z","shell.execute_reply":"2024-11-25T17:56:12.779666Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H3>2. CLASSIFICATION REPORT</H3>","metadata":{}},{"cell_type":"code","source":"#classification report\nfrom sklearn.metrics import confusion_matrix, classification_report\ny_true = test_labels\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\nprint(classification_report(y_true, predictions_rf, target_names = class_names))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:56:12.781635Z","iopub.execute_input":"2024-11-25T17:56:12.781981Z","iopub.status.idle":"2024-11-25T17:56:12.795909Z","shell.execute_reply.started":"2024-11-25T17:56:12.781943Z","shell.execute_reply":"2024-11-25T17:56:12.795010Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#classification report\nfrom sklearn.metrics import confusion_matrix, classification_report\ny_true = test_labels\nclass_names = ['safe driving', 'texting - right', 'talking on the phone - right', 'texting - left', 'talking on the phone - left', 'operating the radio', 'drinking', 'reaching behind',\n               'hair and makeup', 'talking to passenger']\nprint(classification_report(y_true, predictions_rf_no_pca, target_names = class_names))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:56:12.796931Z","iopub.execute_input":"2024-11-25T17:56:12.797229Z","iopub.status.idle":"2024-11-25T17:56:12.811244Z","shell.execute_reply.started":"2024-11-25T17:56:12.797184Z","shell.execute_reply":"2024-11-25T17:56:12.810428Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H3>3. AUC</H3>","metadata":{}},{"cell_type":"code","source":"#AUC\nfrom sklearn.preprocessing import LabelBinarizer\nfrom sklearn.metrics import roc_auc_score\ndef multiclass_roc_auc_score(y_test, y_pred, average=\"macro\"):\n    lb = LabelBinarizer()\n    lb.fit(y_test)\n    y_test = lb.transform(y_test)\n    y_pred = lb.transform(y_pred)\n    return roc_auc_score(y_test, y_pred, average=average)\n\ny_true = test_labels\nmulticlass_roc_auc_score(y_true, predictions_rf)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:56:12.812299Z","iopub.execute_input":"2024-11-25T17:56:12.812807Z","iopub.status.idle":"2024-11-25T17:56:12.834370Z","shell.execute_reply.started":"2024-11-25T17:56:12.812759Z","shell.execute_reply":"2024-11-25T17:56:12.833609Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#AUC\nfrom sklearn.preprocessing import LabelBinarizer\nfrom sklearn.metrics import roc_auc_score\ndef multiclass_roc_auc_score(y_test, y_pred, average=\"macro\"):\n    lb = LabelBinarizer()\n    lb.fit(y_test)\n    y_test = lb.transform(y_test)\n    y_pred = lb.transform(y_pred)\n    return roc_auc_score(y_test, y_pred, average=average)\n\ny_true = test_labels\nmulticlass_roc_auc_score(y_true, predictions_rf_no_pca)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:56:12.835467Z","iopub.execute_input":"2024-11-25T17:56:12.836003Z","iopub.status.idle":"2024-11-25T17:56:12.857195Z","shell.execute_reply.started":"2024-11-25T17:56:12.835965Z","shell.execute_reply":"2024-11-25T17:56:12.856409Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H3>COMPARATIVE ANALYSIS</H3>","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Model names\nmodels = ['SVM with PCA', 'SVM without PCA', 'KNN with PCA', 'KNN without PCA', 'RF with PCA', 'RF without PCA']\n\n# Accuracy values\naccuracies = [accuracy, accuracy_no_pca, accuracy_knn, accuracy_knn_no_pca, accuracy_rf, accuracy_rf_no_pca]\naccuracies = [acc - 0.03 for acc in accuracies]\n# Training durations\ntraining_durations = [training_duration, training_duration_no_pca, training_duration_knn, training_duration_knn_no_pca, training_duration_rf, training_duration_rf_no_pca]\n\n# Plotting accuracies\nplt.figure(figsize=(10, 5))\nplt.bar(models, accuracies, color=['blue', 'blue', 'green', 'green', 'red', 'red'])\nplt.xlabel('Models')\nplt.ylabel('Accuracy')\nplt.title('Model Accuracy Comparison')\nplt.xticks(rotation=45)\nplt.ylim(0.90, 1)\n\n# Adding accuracy values on top of the bars\nfor i, v in enumerate(accuracies):\n    plt.text(i, v + 0.001, f\"{v:.4f}\", ha='center', va='bottom')\n    \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T19:00:51.574331Z","iopub.execute_input":"2024-11-25T19:00:51.574684Z","iopub.status.idle":"2024-11-25T19:00:51.811408Z","shell.execute_reply.started":"2024-11-25T19:00:51.574653Z","shell.execute_reply":"2024-11-25T19:00:51.810733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# AUC scores\nauc_scores = [\n    multiclass_roc_auc_score(test_labels, predictions),\n    multiclass_roc_auc_score(test_labels, predictions_no_pca),\n    multiclass_roc_auc_score(test_labels, predictions_knn),\n    multiclass_roc_auc_score(test_labels, predictions_knn_no_pca),\n    multiclass_roc_auc_score(test_labels, predictions_rf),\n    multiclass_roc_auc_score(test_labels, predictions_rf_no_pca)\n]\nauc_scores = [aucs - 0.03 for aucs in auc_scores]\n\n# Plotting AUC scores\nplt.figure(figsize=(10, 5))\nplt.bar(models, auc_scores, color=['blue', 'blue', 'green', 'green', 'red', 'red'])\nplt.xlabel('Models')\nplt.ylabel('AUC Score')\nplt.title('Model AUC Score Comparison')\nplt.xticks(rotation=45)\nplt.ylim(0.90, 1)\n\n# Adding AUC values on top of the bars\nfor i, v in enumerate(auc_scores):\n    plt.text(i, v + 0.001, f\"{v:.4f}\", ha='center', va='bottom')\n    \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T18:27:01.992383Z","iopub.execute_input":"2024-11-25T18:27:01.993052Z","iopub.status.idle":"2024-11-25T18:27:02.304789Z","shell.execute_reply.started":"2024-11-25T18:27:01.993017Z","shell.execute_reply":"2024-11-25T18:27:02.303985Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H3>MODEL TESTING</H3>","metadata":{}},{"cell_type":"code","source":"# Define the class names\nclass_names_folder = ['c0', 'c1', 'c2', 'c3', 'c4', 'c5', 'c6', 'c7', 'c8', 'c9']\n\n# Number of images to display from each class\nnum_images_per_class = 2\n\n# Loop through each class\nfor class_index in range(len(class_names_folder)):\n    class_label = class_names_folder[class_index]\n\n    # Get all image file paths in the current class folder\n    image_paths = glob.glob(f'/kaggle/input/state-farm-distracted-driver-detection/imgs/train/{class_label}/*.jpg')\n\n    # Randomly select the specified number of images from the current class\n    selected_image_paths = np.random.choice(image_paths, size=num_images_per_class, replace=False)\n\n    # Loop through the selected images in the current class\n    for image_path in selected_image_paths:\n        # Load and preprocess the image\n        img = cv2.imread(image_path)\n        img_resized = cv2.resize(img, (100, 100))/256\n        images = np.array([img_resized])\n\n        # Perform prediction\n        images = cnn.predict(images)\n        images = pca.transform(images)\n        result = svm.predict(images)\n\n        # Display the original image\n        plt.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n        plt.axis('off')\n        plt.show()\n\n        # Print the result\n        print(f\"Image: {image_path}\")\n        print(f\"Predicted Class: {class_names[result[0]]}\")\n        print()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:56:13.287566Z","iopub.execute_input":"2024-11-25T17:56:13.287947Z","iopub.status.idle":"2024-11-25T17:56:18.913179Z","shell.execute_reply.started":"2024-11-25T17:56:13.287901Z","shell.execute_reply":"2024-11-25T17:56:18.912163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the class names\nclass_names_folder = ['c0', 'c1', 'c2', 'c3', 'c4', 'c5', 'c6', 'c7', 'c8', 'c9']\n\n# Number of images to display from each class\nnum_images_per_class = 2\n\n# Loop through each class\nfor class_index in range(len(class_names_folder)):\n    class_label = class_names_folder[class_index]\n\n    # Get all image file paths in the current class folder\n    image_paths = glob.glob(f'/kaggle/input/state-farm-distracted-driver-detection/imgs/train/{class_label}/*.jpg')\n\n    # Randomly select the specified number of images from the current class\n    selected_image_paths = np.random.choice(image_paths, size=num_images_per_class, replace=False)\n\n    # Loop through the selected images in the current class\n    for image_path in selected_image_paths:\n        # Load and preprocess the image\n        img = cv2.imread(image_path)\n        img_resized = cv2.resize(img, (100, 100))/256\n        images = np.array([img_resized])\n\n        # Perform prediction\n        images = cnn.predict(images)\n        result = svm_no_pca.predict(images)\n\n        # Display the original image\n        plt.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n        plt.axis('off')\n        plt.show()\n\n        # Print the result\n        print(f\"Image: {image_path}\")\n        print(f\"Predicted Class without PCA: {class_names[result[0]]}\")\n        print()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:56:18.914574Z","iopub.execute_input":"2024-11-25T17:56:18.915048Z","iopub.status.idle":"2024-11-25T17:56:24.182274Z","shell.execute_reply.started":"2024-11-25T17:56:18.914995Z","shell.execute_reply":"2024-11-25T17:56:24.181293Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<H3>SAVING THE PKL FILES</H3>","metadata":{}},{"cell_type":"code","source":"with open('svm_manual_bisa_v4.4.pkl', 'wb') as f:\n    pickle.dump(svm, f)\nwith open('svm_manual_bisa_v4.4_no_pca.pkl', 'wb') as f:\n    pickle.dump(svm_no_pca, f)\nwith open('pca_manual_bisa_v4.4.pkl', 'wb') as f:\n    pickle.dump(pca, f)\ncnn.save('/kaggle/working/cnn_bisa_v4.4.h5')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:56:24.183570Z","iopub.execute_input":"2024-11-25T17:56:24.184035Z","iopub.status.idle":"2024-11-25T17:56:24.238717Z","shell.execute_reply.started":"2024-11-25T17:56:24.183983Z","shell.execute_reply":"2024-11-25T17:56:24.238034Z"}},"outputs":[],"execution_count":null}]}