{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\n\n# Load the dataset\ntrain_df = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/train.csv')\n\n# Assuming 'id_code' is the column with the image filenames and 'diagnosis' is the label\nX = train_df['id_code']\ny = train_df['diagnosis']\n\n# Split the dataset into training and validation sets\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-08T01:01:18.226308Z","iopub.execute_input":"2024-04-08T01:01:18.226730Z","iopub.status.idle":"2024-04-08T01:01:18.239602Z","shell.execute_reply.started":"2024-04-08T01:01:18.226702Z","shell.execute_reply":"2024-04-08T01:01:18.238546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\n\ndef preprocess_image(image_path, target_size=(224, 224)):\n    # Read the image\n    img = cv2.imread(image_path)\n    # Resize the image\n    img = cv2.resize(img, target_size)\n    # Convert to grayscale\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n    return img\n\n# Example usage\n# img = preprocess_image('path/to/image.png')\n","metadata":{"execution":{"iopub.status.busy":"2024-04-08T01:01:21.905857Z","iopub.execute_input":"2024-04-08T01:01:21.906263Z","iopub.status.idle":"2024-04-08T01:01:22.123285Z","shell.execute_reply.started":"2024-04-08T01:01:21.906235Z","shell.execute_reply":"2024-04-08T01:01:22.122146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_features(image):\n    # Assuming image is a preprocessed grayscale image\n    features = []\n    # Add basic statistical features\n    features.append(np.mean(image))\n    features.append(np.std(image))\n    # You can add more sophisticated features here\n    return features\n\n# Example usage\n# features = extract_features(preprocessed_img)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-08T01:01:25.010510Z","iopub.execute_input":"2024-04-08T01:01:25.010993Z","iopub.status.idle":"2024-04-08T01:01:25.017437Z","shell.execute_reply.started":"2024-04-08T01:01:25.010959Z","shell.execute_reply":"2024-04-08T01:01:25.016293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.svm import SVC\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\n\n# Initialize an SVM with a linear kernel\nsvm = make_pipeline(StandardScaler(), SVC(kernel='linear'))\n\n# For demonstration, let's assume you have a function to extract features for all images\nX_train_features = np.array([extract_features(preprocess_image(f'/kaggle/input/aptos2019-blindness-detection/train_images/{x}.png')) for x in X_train])\nX_val_features = np.array([extract_features(preprocess_image(f'/kaggle/input/aptos2019-blindness-detection/train_images/{x}.png')) for x in X_val])\n\n# Train the SVM\nsvm.fit(X_train_features, y_train)\n\n# Evaluate the model\nprint(\"Validation Accuracy:\", svm.score(X_val_features, y_val))\n","metadata":{"execution":{"iopub.status.busy":"2024-04-08T01:01:33.110449Z","iopub.execute_input":"2024-04-08T01:01:33.111046Z","iopub.status.idle":"2024-04-08T01:08:56.509860Z","shell.execute_reply.started":"2024-04-08T01:01:33.111004Z","shell.execute_reply":"2024-04-08T01:08:56.506084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications.efficientnet import preprocess_input\n\n# Load the dataset\ntrain_df = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/train.csv')\ntrain_df['id_code'] = train_df['id_code'].apply(lambda x: f\"{x}.png\")\ntrain_df['diagnosis'] = train_df['diagnosis'].astype(str)\n\n# Split the dataset\ntrain_df, val_df = train_test_split(train_df, test_size=0.2, random_state=42)\n\n# Data generators for training and validation sets\ntrain_datagen = ImageDataGenerator(preprocessing_function=preprocess_input)\nval_datagen = ImageDataGenerator(preprocessing_function=preprocess_input)\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    directory='/kaggle/input/aptos2019-blindness-detection/train_images',\n    x_col='id_code',\n    y_col='diagnosis',\n    target_size=(224, 224),\n    class_mode='raw',\n    batch_size=32\n)\n\nval_generator = val_datagen.flow_from_dataframe(\n    dataframe=val_df,\n    directory='/kaggle/input/aptos2019-blindness-detection/train_images',\n    x_col='id_code',\n    y_col='diagnosis',\n    target_size=(224, 224),\n    class_mode='raw',\n    batch_size=32\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-08T01:12:26.457994Z","iopub.execute_input":"2024-04-08T01:12:26.459329Z","iopub.status.idle":"2024-04-08T01:12:40.987461Z","shell.execute_reply.started":"2024-04-08T01:12:26.459275Z","shell.execute_reply":"2024-04-08T01:12:40.986432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.models import Model\n\n# Load the pre-trained EfficientNetB0 model\nbase_model = EfficientNetB0(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\n\n# Construct the feature extractor model\nmodel = Model(inputs=base_model.input, outputs=base_model.output)\n\n# Function to extract features from a batch of images\ndef extract_features(generator, model):\n    features = []\n    labels = []\n    for i in range(len(generator)):\n        imgs, lbls = generator[i]\n        feats = model.predict(imgs)\n        features.append(feats)\n        labels.append(lbls)\n    features = np.vstack(features)\n    labels = np.concatenate(labels)\n    return features, labels\n\n# Extract features for training and validation sets\nX_train, y_train = extract_features(train_generator, model)\nX_val, y_val = extract_features(val_generator, model)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-08T01:12:53.579106Z","iopub.execute_input":"2024-04-08T01:12:53.579524Z","iopub.status.idle":"2024-04-08T01:23:39.667371Z","shell.execute_reply.started":"2024-04-08T01:12:53.579492Z","shell.execute_reply":"2024-04-08T01:23:39.666302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.svm import SVC\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\n\n# Reshape the features\nX_train_flat = X_train.reshape(X_train.shape[0], -1)\nX_val_flat = X_val.reshape(X_val.shape[0], -1)\n\n# Initialize and train the SVM\nsvm = make_pipeline(StandardScaler(), SVC(kernel='linear'))\nsvm.fit(X_train_flat, y_train)\n\n# Evaluate the model\naccuracy = svm.score(X_val_flat, y_val)\nprint(f\"Validation Accuracy: {accuracy}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-04-08T01:26:49.544814Z","iopub.execute_input":"2024-04-08T01:26:49.545529Z","iopub.status.idle":"2024-04-08T01:30:10.391069Z","shell.execute_reply.started":"2024-04-08T01:26:49.545486Z","shell.execute_reply":"2024-04-08T01:30:10.389668Z"},"trusted":true},"execution_count":null,"outputs":[]}]}