{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport cv2\nfrom sklearn.svm import SVC\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, accuracy_score, roc_curve, auc\nimport matplotlib.pyplot as plt\nfrom joblib import Parallel, delayed\nfrom sklearn.preprocessing import StandardScaler","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-08T05:54:05.716049Z","iopub.execute_input":"2024-04-08T05:54:05.716549Z","iopub.status.idle":"2024-04-08T05:54:05.723741Z","shell.execute_reply.started":"2024-04-08T05:54:05.716515Z","shell.execute_reply":"2024-04-08T05:54:05.722563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocess and feature extraction functions\ndef preprocess_image(image_path, output_size=(224, 224)):\n    image = cv2.imread(image_path, cv2.IMREAD_GRAYSCALE)\n    image_resized = cv2.equalizeHist(cv2.resize(image, output_size))\n    return image_resized\n\ndef extract_features(image):\n    return np.histogram(image, bins=8)[0]\n\ndef process_image(index, row):\n    image_path = f\"{train_images_path}/{row['id_code']}.png\"\n    image = preprocess_image(image_path)\n    features = extract_features(image)\n    return features, row['diagnosis']","metadata":{"execution":{"iopub.status.busy":"2024-04-08T05:54:05.726671Z","iopub.execute_input":"2024-04-08T05:54:05.727238Z","iopub.status.idle":"2024-04-08T05:54:05.735982Z","shell.execute_reply.started":"2024-04-08T05:54:05.727191Z","shell.execute_reply":"2024-04-08T05:54:05.734980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_path = \"/kaggle/input/aptos2019-blindness-detection\"\ntrain_csv_path = f\"{base_path}/train.csv\"\ntrain_images_path = f\"{base_path}/train_images\"","metadata":{"execution":{"iopub.status.busy":"2024-04-08T05:54:05.737233Z","iopub.execute_input":"2024-04-08T05:54:05.738113Z","iopub.status.idle":"2024-04-08T05:54:05.746841Z","shell.execute_reply.started":"2024-04-08T05:54:05.738081Z","shell.execute_reply":"2024-04-08T05:54:05.745838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load and split data into train and test sets","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(train_csv_path)\n\n# Parallelize preprocessing and feature extraction\nresults = Parallel(n_jobs=-1)(delayed(process_image)(index, row) for index, row in df.iterrows())\n\nX = np.array([result[0] for result in results])\ny = np.array([result[1] for result in results])\n\n# Split dataset\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T05:54:05.748805Z","iopub.execute_input":"2024-04-08T05:54:05.749505Z","iopub.status.idle":"2024-04-08T05:57:09.297709Z","shell.execute_reply.started":"2024-04-08T05:54:05.749437Z","shell.execute_reply":"2024-04-08T05:57:09.293184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train SVM Classifier","metadata":{}},{"cell_type":"code","source":"svm_model = SVC(kernel='rbf', C=1.0, gamma='auto', probability=True)\nsvm_model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T05:57:09.302259Z","iopub.execute_input":"2024-04-08T05:57:09.302649Z","iopub.status.idle":"2024-04-08T05:57:14.467930Z","shell.execute_reply.started":"2024-04-08T05:57:09.302616Z","shell.execute_reply":"2024-04-08T05:57:14.466588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluate model on test data","metadata":{}},{"cell_type":"code","source":"y_pred = svm_model.predict(X_test)\nprint(\"Accuracy:\", accuracy_score(y_test, y_pred))\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-04-08T05:57:14.469683Z","iopub.execute_input":"2024-04-08T05:57:14.470006Z","iopub.status.idle":"2024-04-08T05:57:14.631930Z","shell.execute_reply.started":"2024-04-08T05:57:14.469980Z","shell.execute_reply":"2024-04-08T05:57:14.629829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Plotting ROC Curve","metadata":{}},{"cell_type":"code","source":"y_prob = svm_model.predict_proba(X_test)[:, 1]\nfpr, tpr, thresholds = roc_curve(y_test, y_prob, pos_label=1)\nroc_auc = auc(fpr, tpr)\n\nplt.figure()\nplt.plot(fpr, tpr, color='darkorange', lw=2, label='ROC curve (area = %0.2f)' % roc_auc)\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic')\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-08T05:57:14.634151Z","iopub.execute_input":"2024-04-08T05:57:14.634625Z","iopub.status.idle":"2024-04-08T05:57:15.101772Z","shell.execute_reply.started":"2024-04-08T05:57:14.634575Z","shell.execute_reply":"2024-04-08T05:57:15.100534Z"},"trusted":true},"execution_count":null,"outputs":[]}]}