{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"execution":{"iopub.status.busy":"2022-03-16T06:40:55.341884Z","iopub.execute_input":"2022-03-16T06:40:55.342719Z","iopub.status.idle":"2022-03-16T06:40:56.026848Z","shell.execute_reply.started":"2022-03-16T06:40:55.342595Z","shell.execute_reply":"2022-03-16T06:40:56.025967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cudf as pd\nimport cupy as np\nimport matplotlib.pyplot as plt\nfrom cuml.model_selection import train_test_split\nfrom tqdm import tqdm\nfrom keras.preprocessing import image\n\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:19:17.094033Z","iopub.execute_input":"2022-03-16T07:19:17.094292Z","iopub.status.idle":"2022-03-16T07:19:25.459297Z","shell.execute_reply.started":"2022-03-16T07:19:17.094263Z","shell.execute_reply":"2022-03-16T07:19:25.458549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/state-farm-distracted-driver-detection/driver_imgs_list.csv')    # reading the csv file\ntrain.head() ","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:19:25.460789Z","iopub.execute_input":"2022-03-16T07:19:25.461086Z","iopub.status.idle":"2022-03-16T07:19:30.488357Z","shell.execute_reply.started":"2022-03-16T07:19:25.461051Z","shell.execute_reply":"2022-03-16T07:19:30.487680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image dataset loading  \n### For ML models we put target image size as 64x64 across 3 channels (R,G,B) and flatten the matrix to give 1D array which ML Models expects.\n### Image api of Keras is used for dataset loading.","metadata":{}},{"cell_type":"code","source":"train_image = []\nfor i in tqdm(range(train.shape[0])):\n    img = image.load_img('../input/state-farm-distracted-driver-detection/imgs/train/'+train[\"classname\"][i]+\"/\"+train[\"img\"][i],target_size=(64,64,3))\n    img = image.img_to_array(img).flatten()\n    img = img/255\n    train_image.append(img)\nX = np.array(train_image)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:19:30.489535Z","iopub.execute_input":"2022-03-16T07:19:30.489940Z","iopub.status.idle":"2022-03-16T07:25:49.316885Z","shell.execute_reply.started":"2022-03-16T07:19:30.489905Z","shell.execute_reply":"2022-03-16T07:25:49.316054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Encoding Classnames","metadata":{}},{"cell_type":"code","source":"factor = pd.factorize(train['classname'])\ny = factor[0]\ndefinitions = factor[1]\nprint(y)\nprint(definitions)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:25:49.318999Z","iopub.execute_input":"2022-03-16T07:25:49.319217Z","iopub.status.idle":"2022-03-16T07:25:49.434815Z","shell.execute_reply.started":"2022-03-16T07:25:49.319190Z","shell.execute_reply":"2022-03-16T07:25:49.434001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Checking for class Imbalance in dataset\n","metadata":{}},{"cell_type":"code","source":"print(train['classname'].value_counts())\npd.DataFrame(train['classname'].value_counts()).to_pandas().plot(kind='bar')","metadata":{"execution":{"iopub.status.busy":"2022-03-16T06:47:33.577860Z","iopub.execute_input":"2022-03-16T06:47:33.578115Z","iopub.status.idle":"2022-03-16T06:47:33.849784Z","shell.execute_reply.started":"2022-03-16T06:47:33.578080Z","shell.execute_reply":"2022-03-16T06:47:33.849109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image Quality Assessment using Brisque Score","metadata":{}},{"cell_type":"code","source":"from libsvm import svmutil\n!pip install pybrisque\nfrom brisque import *","metadata":{"execution":{"iopub.status.busy":"2022-03-16T06:50:32.739309Z","iopub.execute_input":"2022-03-16T06:50:32.739738Z","iopub.status.idle":"2022-03-16T06:50:39.949627Z","shell.execute_reply.started":"2022-03-16T06:50:32.739702Z","shell.execute_reply":"2022-03-16T06:50:39.948772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"brisq = BRISQUE()","metadata":{"execution":{"iopub.status.busy":"2022-03-16T06:50:41.567424Z","iopub.execute_input":"2022-03-16T06:50:41.567828Z","iopub.status.idle":"2022-03-16T06:50:41.581037Z","shell.execute_reply.started":"2022-03-16T06:50:41.567793Z","shell.execute_reply":"2022-03-16T06:50:41.580371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nscores=[]\nl=[]\nfor i in tqdm(range(train.shape[0])):\n    temp=brisq.get_score('../input/state-farm-distracted-driver-detection/imgs/train/'+train[\"classname\"][i]+\"/\"+train[\"img\"][i])\n    l.append((train[\"img\"][i],temp))\n    scores.append(temp)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T06:50:42.099609Z","iopub.execute_input":"2022-03-16T06:50:42.099863Z","iopub.status.idle":"2022-03-16T07:13:14.805022Z","shell.execute_reply.started":"2022-03-16T06:50:42.099833Z","shell.execute_reply":"2022-03-16T07:13:14.804338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import statistics\nstatistics.mean(scores)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:13:14.807067Z","iopub.execute_input":"2022-03-16T07:13:14.807607Z","iopub.status.idle":"2022-03-16T07:13:14.836414Z","shell.execute_reply.started":"2022-03-16T07:13:14.807566Z","shell.execute_reply":"2022-03-16T07:13:14.835587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### since BRISQUE score is less so dataset images are of high quality.","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.hist(scores)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:13:14.837889Z","iopub.execute_input":"2022-03-16T07:13:14.838142Z","iopub.status.idle":"2022-03-16T07:13:15.151778Z","shell.execute_reply.started":"2022-03-16T07:13:14.838107Z","shell.execute_reply":"2022-03-16T07:13:15.151121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:13:15.153756Z","iopub.execute_input":"2022-03-16T07:13:15.154153Z","iopub.status.idle":"2022-03-16T07:13:15.159986Z","shell.execute_reply.started":"2022-03-16T07:13:15.154115Z","shell.execute_reply":"2022-03-16T07:13:15.159176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train-Test Split ","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=42, test_size=0.1)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:25:49.436258Z","iopub.execute_input":"2022-03-16T07:25:49.436509Z","iopub.status.idle":"2022-03-16T07:25:51.520614Z","shell.execute_reply.started":"2022-03-16T07:25:49.436475Z","shell.execute_reply":"2022-03-16T07:25:51.519884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from cuml.naive_bayes import GaussianNB\nfrom cuml.linear_model import LogisticRegression\nfrom cuml.svm import SVC","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:13:17.451328Z","iopub.execute_input":"2022-03-16T07:13:17.451570Z","iopub.status.idle":"2022-03-16T07:13:17.458430Z","shell.execute_reply.started":"2022-03-16T07:13:17.451538Z","shell.execute_reply":"2022-03-16T07:13:17.456563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Logistic Regression","metadata":{}},{"cell_type":"code","source":"clf_lr = LogisticRegression()\nclf_lr.fit(X_train, y_train)\nimport cuml\npreds= clf_lr.predict(X_test)\ncu_score = cuml.metrics.accuracy_score( y_test, preds )\nprint(cu_score)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:13:17.459910Z","iopub.execute_input":"2022-03-16T07:13:17.460215Z","iopub.status.idle":"2022-03-16T07:13:28.152099Z","shell.execute_reply.started":"2022-03-16T07:13:17.460179Z","shell.execute_reply":"2022-03-16T07:13:28.151336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Gaussian Naive Bias","metadata":{}},{"cell_type":"code","source":"clf_gnb = GaussianNB()\nclf_gnb.fit(X_train, y_train)\nimport cuml\npreds= clf_gnb.predict(X_test)\ncu_score = cuml.metrics.accuracy_score( y_test, preds )\nprint(cu_score)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:13:28.153460Z","iopub.execute_input":"2022-03-16T07:13:28.153874Z","iopub.status.idle":"2022-03-16T07:13:37.205838Z","shell.execute_reply.started":"2022-03-16T07:13:28.153835Z","shell.execute_reply":"2022-03-16T07:13:37.204327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Support Vector Classifier","metadata":{}},{"cell_type":"code","source":"clf_svc = SVC(probability=True)\nclf_svc.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:13:37.206964Z","iopub.execute_input":"2022-03-16T07:13:37.207204Z","iopub.status.idle":"2022-03-16T07:16:00.450393Z","shell.execute_reply.started":"2022-03-16T07:13:37.207168Z","shell.execute_reply":"2022-03-16T07:16:00.449572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predicting Probability of Each Class","metadata":{}},{"cell_type":"code","source":"preds_prob= clf_svc.predict_proba(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:16:00.453211Z","iopub.execute_input":"2022-03-16T07:16:00.453493Z","iopub.status.idle":"2022-03-16T07:16:30.997813Z","shell.execute_reply.started":"2022-03-16T07:16:00.453456Z","shell.execute_reply":"2022-03-16T07:16:30.997102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_prob[0]","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:16:30.999045Z","iopub.execute_input":"2022-03-16T07:16:30.999297Z","iopub.status.idle":"2022-03-16T07:16:31.005680Z","shell.execute_reply.started":"2022-03-16T07:16:30.999247Z","shell.execute_reply":"2022-03-16T07:16:31.004875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predicting Best class for Accuracy metric","metadata":{}},{"cell_type":"code","source":"import cuml\npreds= clf_svc.predict(X_test)\ncu_score = cuml.metrics.accuracy_score( y_test, preds )\nprint(cu_score)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:16:31.007531Z","iopub.execute_input":"2022-03-16T07:16:31.007798Z","iopub.status.idle":"2022-03-16T07:17:01.668825Z","shell.execute_reply.started":"2022-03-16T07:16:31.007762Z","shell.execute_reply":"2022-03-16T07:17:01.667162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:17:01.669985Z","iopub.execute_input":"2022-03-16T07:17:01.670673Z","iopub.status.idle":"2022-03-16T07:17:01.676159Z","shell.execute_reply.started":"2022-03-16T07:17:01.670632Z","shell.execute_reply":"2022-03-16T07:17:01.675502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Calculating Confusion Metric","metadata":{}},{"cell_type":"code","source":"from cuml.metrics import confusion_matrix\ncm=confusion_matrix(y_test.astype(\"int32\"),preds.astype(\"int32\"))","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:17:01.677406Z","iopub.execute_input":"2022-03-16T07:17:01.677806Z","iopub.status.idle":"2022-03-16T07:17:06.284506Z","shell.execute_reply.started":"2022-03-16T07:17:01.677766Z","shell.execute_reply":"2022-03-16T07:17:06.283763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport cupy as np\nsns.set(font_scale=1.0)\nsns.heatmap(np.asnumpy(cm),annot=True, cmap='Blues',fmt='g')","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:17:47.507006Z","iopub.execute_input":"2022-03-16T07:17:47.507292Z","iopub.status.idle":"2022-03-16T07:17:48.163020Z","shell.execute_reply.started":"2022-03-16T07:17:47.507248Z","shell.execute_reply":"2022-03-16T07:17:48.162341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## XGBoost ,CatBoost, LightGbm, Random Forest & their Ensemble","metadata":{}},{"cell_type":"code","source":"import xgboost as xgb\nimport cuml\nxgb_clf = xgb.XGBClassifier(use_label_encoder=False,tree_method='gpu_hist')\nxgb_clf.fit(X_train, y_train)\npreds_prob_xgb=xgb_clf.predict_proba(X_test)\npreds= xgb_clf.predict(X_test)\ncu_score = cuml.metrics.accuracy_score( y_test, preds )\nprint(cu_score)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:28:19.181478Z","iopub.execute_input":"2022-03-16T07:28:19.182015Z","iopub.status.idle":"2022-03-16T07:28:19.447309Z","shell.execute_reply.started":"2022-03-16T07:28:19.181978Z","shell.execute_reply":"2022-03-16T07:28:19.446528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostClassifier\ncgb_clf = CatBoostClassifier(iterations=500,learning_rate =0.01,\n                           task_type=\"GPU\",metric_period=100,\n                           random_seed=42)\ncgb_clf.fit(np.asnumpy(X_train),np.asnumpy(y_train))\npreds_prob_cgb=cgb_clf.predict_proba(np.asnumpy(X_test))\npreds= cgb_clf.predict(np.asnumpy(X_test))\ncu_score = cuml.metrics.accuracy_score( y_test, preds )\nprint(cu_score)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:28:58.486803Z","iopub.execute_input":"2022-03-16T07:28:58.487402Z","iopub.status.idle":"2022-03-16T07:33:01.263459Z","shell.execute_reply.started":"2022-03-16T07:28:58.487365Z","shell.execute_reply":"2022-03-16T07:33:01.262709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nlgb_clf = lgb.LGBMClassifier(boosting_type='dart',learning_rate=0.18, max_depth=7,\n               n_estimators=450,objective='binary',device='gpu',\n               random_state=42)\nlgb_clf.fit(np.asnumpy(X_train),np.asnumpy(y_train))\npreds_prob_lgb=lgb_clf.predict_proba(np.asnumpy(X_test))\npreds= lgb_clf.predict(np.asnumpy(X_test))\ncu_score = cuml.metrics.accuracy_score( y_test, preds )\nprint(cu_score)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T07:33:01.265420Z","iopub.execute_input":"2022-03-16T07:33:01.265886Z","iopub.status.idle":"2022-03-16T08:37:57.542771Z","shell.execute_reply.started":"2022-03-16T07:33:01.265805Z","shell.execute_reply":"2022-03-16T08:37:57.542119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from cuml.ensemble import RandomForestClassifier\nrdf_clf=RandomForestClassifier(n_estimators=600,random_state=42, verbose=0,warm_start=False)\nrdf_clf.fit(X_train, y_train)\npreds_prob_rdf=rdf_clf.predict_proba(X_test)\npreds= rdf_clf.predict(X_test)\ncu_score = cuml.metrics.accuracy_score( y_test, preds )\nprint(cu_score)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T08:37:57.546166Z","iopub.execute_input":"2022-03-16T08:37:57.547734Z","iopub.status.idle":"2022-03-16T08:39:10.819553Z","shell.execute_reply.started":"2022-03-16T08:37:57.547700Z","shell.execute_reply":"2022-03-16T08:39:10.818819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ensemble\n#### *Note - had memory allocation problems in ensembling can be implemented as below on gpu with greater memory ","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import  VotingClassifier\neclf1 = VotingClassifier(estimators=[('catboost', cgb_clf), ('xgboost', xgb_cl), ('lightgbm', lgb_clf),('randomforest', rdf_clf)], voting='soft',weights=[3,2,3,3],flatten_transform=True)\neclf1 = eclf1.fit(np.asnumpy(X_train),np.asnumpy(y_train))","metadata":{"execution":{"iopub.status.busy":"2022-03-16T08:40:41.300337Z","iopub.execute_input":"2022-03-16T08:40:41.300592Z","iopub.status.idle":"2022-03-16T09:51:28.449165Z","shell.execute_reply.started":"2022-03-16T08:40:41.300563Z","shell.execute_reply":"2022-03-16T09:51:28.448580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds= eclf1.predict(np.asnumpy(X_test))\ncu_score = cuml.metrics.accuracy_score( y_test, preds )\nprint(cu_score)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T09:58:34.052055Z","iopub.execute_input":"2022-03-16T09:58:34.052337Z","iopub.status.idle":"2022-03-16T09:58:37.323803Z","shell.execute_reply.started":"2022-03-16T09:58:34.052307Z","shell.execute_reply":"2022-03-16T09:58:37.322913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eclf1.predict_proba(np.asnumpy(X_test))[0]","metadata":{"execution":{"iopub.status.busy":"2022-03-16T09:59:17.494836Z","iopub.execute_input":"2022-03-16T09:59:17.495374Z","iopub.status.idle":"2022-03-16T09:59:20.649562Z","shell.execute_reply.started":"2022-03-16T09:59:17.495339Z","shell.execute_reply":"2022-03-16T09:59:20.648787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/state-farm-distracted-driver-detection/sample_submission.csv')    # reading the csv file\ntest.head() \n","metadata":{"execution":{"iopub.status.busy":"2022-03-16T10:27:00.664310Z","iopub.execute_input":"2022-03-16T10:27:00.664910Z","iopub.status.idle":"2022-03-16T10:27:00.769159Z","shell.execute_reply.started":"2022-03-16T10:27:00.664869Z","shell.execute_reply":"2022-03-16T10:27:00.768420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_image = []\nfor i in tqdm(range(test.shape[0])):\n    img = image.load_img('../input/state-farm-distracted-driver-detection/imgs/test/'+test[\"img\"][i],target_size=(64,64,3))\n    img = image.img_to_array(img).flatten()\n    img = img/255\n    test_image.append(img)\ntest_data = np.array(test_image)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T10:27:17.198244Z","iopub.execute_input":"2022-03-16T10:27:17.198731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds=eclf1.predict_proba(np.asnumpy(test_data))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}