{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n# visulization\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\n\nimport os\nimport gc # garbage collection\nimport glob # extract path via pattern matching\nfrom tqdm.notebook import tqdm # progressbar\nimport random\nimport math\nimport cv2 # read image\n# store to disk\n\nfrom sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"ROOT_DIR = '../input/state-farm-distracted-driver-detection/'\nTRAIN_DIR = ROOT_DIR + 'imgs/train/'\nTEST_DIR = ROOT_DIR + 'imgs/test/'\ndriver_imgs_list = pd.read_csv(ROOT_DIR + \"driver_imgs_list.csv\")\nsample_submission = pd.read_csv(ROOT_DIR + \"sample_submission.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"random_list = np.random.permutation(len(driver_imgs_list))[:50]\ndf_copy = driver_imgs_list.iloc[random_list]\nimage_paths = [TRAIN_DIR+row.classname+'/'+row.img \n                   for (index, row) in df_copy.iterrows()]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_path_list = []\nlabel_list = []\nfor index, row in driver_imgs_list.iterrows():\n    img_path_list.append('{0}{1}/{2}'.format(TRAIN_DIR, row.classname, row.img))\n    label_list.append(int(row.classname[1]))\n# One hot vector representation of labels\ny_labels = np.array(label_list, dtype=np.int8)\nx_img_path = np.array(img_path_list)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.save('x_img_path.npy', x_img_path)\nnp.save('y_labels.npy', y_labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.utils import shuffle\n\nx_img_path_shuffled, y_labels_shuffled = shuffle(x_img_path, y_labels)\n\n# saving the shuffled file.\n# you can load them later using np.load().\nnp.save('y_labels_shuffled.npy', y_labels_shuffled)\nnp.save('x_img_path_shuffled.npy', x_img_path_shuffled)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Used this line as our filename array is not a numpy array.\nx_img_path_shuffled_numpy = np.array(x_img_path_shuffled)\n\nX_train_filenames, X_val_filenames, y_train, y_val = train_test_split(\n    x_img_path_shuffled_numpy, y_labels_shuffled, test_size=0.2, random_state=1)\n\nprint(X_train_filenames.shape) # (3800,)\nprint(y_train.shape)           # (3800, 12)\n\nprint(X_val_filenames.shape)   # (950,)\nprint(y_val.shape)             # (950, 12)\n\n# You can save these files as well. As you will be using them later for training and validation of your model.\nnp.save('X_train_filenames.npy', X_train_filenames)\nnp.save('y_train.npy', y_train)\n\nnp.save('X_val_filenames.npy', X_val_filenames)\nnp.save('y_val.npy', y_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ORB_extractor = cv2.ORB_create(nfeatures=200)\nall_descriptors = []\nfor filepath in X_train_filenames:\n    img = cv2.imread(filepath, 0)\n    points, desc = ORB_extractor.detectAndCompute(img, None)\n    all_descriptors.append(desc)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"kmeans_features = np.vstack(tuple(all_descriptors[:2000]))\nall_features = np.vstack(tuple(all_descriptors))\nnp.save('kmeans_features.npy', kmeans_features)\nnp.save('all_features.npy', all_features)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.cluster import KMeans\nkmeans_model = KMeans(n_clusters = 50).fit(kmeans_features)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def extractFeatures(kmeans, descriptor_list, no_clusters):\n    image_count = len(descriptor_list)\n    im_features = np.array([np.zeros(no_clusters) for i in range(image_count)])\n    for i in range(image_count):\n        for j in range(len(descriptor_list[i])):\n            feature = descriptor_list[i][j]\n            feature = feature.reshape(1, 32)\n            idx = kmeans.predict(feature)\n            im_features[i][idx] += 1\n\n    return im_features","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"im_features = extractFeatures(kmeans_model, all_descriptors, 50)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nscale = StandardScaler().fit(im_features)        \nim_features_normed = scale.transform(im_features)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\nfrom sklearn.svm import SVC\ndef svcParamSelection(X, y, kernel, nfolds):\n    Cs = [0.5, 0.1, 0.15, 0.2, 0.3]\n    gammas = [0.1, 0.11, 0.095, 0.105]\n    param_grid = {'C': Cs, 'gamma' : gammas}\n    grid_search = GridSearchCV(SVC(kernel=kernel), param_grid, cv=nfolds)\n    grid_search.fit(X, y)\n    grid_search.best_params_\n    return grid_search.best_params_\n\ndef findSVM(im_features, train_labels, kernel):\n    features = im_features   \n    params = svcParamSelection(features, train_labels, kernel, 5)\n    C_param, gamma_param = params.get(\"C\"), params.get(\"gamma\")\n    print(C_param, gamma_param)\n    svm = SVC(kernel = kernel, C =  C_param, gamma = gamma_param)\n    svm.fit(features, train_labels)\n    return svm","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"svm_model = findSVM(im_features_normed, y_train, \"linear\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Testing"},{"metadata":{"trusted":true},"cell_type":"code","source":"def extractFeatures(kmeans_model, descriptor_list, image_count, no_clusters=50):\n    im_features = np.array([np.zeros(no_clusters) for i in range(image_count)])\n    for i in range(image_count):\n        for j in range(len(descriptor_list[i])):\n            feature = descriptor_list[i][j]\n            feature = feature.reshape(1, 32)\n            idx = kmeans_model.predict(feature)\n            im_features[i][idx] += 1\n    return im_features\n\ndef test_model(kmeans_model, svm_model, test_x, test_y):\n    ORB_extractor_test = cv2.ORB_create(nfeatures=200)\n    all_descriptors = []\n    count = 0\n    for filepath in test_x:\n        img = cv2.imread(filepath, 0)\n        _, desc = ORB_extractor_test.detectAndCompute(img, None)\n        if desc is not None:\n            all_descriptors.append(desc)\n            count += 1\n    test_features = extractFeatures(kmeans_model, all_descriptors, count, 50)\n    test_features = scale.transform(test_features)\n  \n    predictions = svm_model.predict(test_features)\n    return predictions\n    print(\"Test images classified.\")\n\n    #plotConfusions(true, predictions)\n    print(\"Confusion matrixes plotted.\")\n\n    #findAccuracy(true, predictions)\n    print(\"Accuracy calculated.\")\n    print(\"Execution done.\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"result = test_model(kmeans_model, svm_model, X_val_filenames, y_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"correct = sum(y_val == result)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"correct/y_val.shape[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}