{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":9250838,"sourceType":"datasetVersion","datasetId":5596611},{"sourceId":9543210,"sourceType":"datasetVersion","datasetId":5813719},{"sourceId":9544466,"sourceType":"datasetVersion","datasetId":5814640},{"sourceId":9573165,"sourceType":"datasetVersion","datasetId":5835544},{"sourceId":9583077,"sourceType":"datasetVersion","datasetId":5843572},{"sourceId":9641206,"sourceType":"datasetVersion","datasetId":5887412}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nfrom tensorflow.keras.applications import InceptionV3\nfrom tensorflow.keras.applications.inception_v3 import preprocess_input\nfrom tensorflow.keras.preprocessing import image\nfrom xgboost import XGBClassifier\nfrom sklearn.preprocessing import LabelEncoder\nimport joblib\n\n# Paths to InceptionV3 weights and data files\nweights_path = '/kaggle/input/inceptionv3/inception_v3_weights_tf_dim_ordering_tf_kernels_notop.h5'\ntrain_csv_path = '/kaggle/input/aptos2019-blindness-detection/train.csv' \ntrain_images_folder = '/kaggle/input/aptos2019-blindness-detection/train_images'  \ntest_csv_path = '/kaggle/input/aptos2019-blindness-detection/test.csv'  \ntest_images_folder = '/kaggle/input/aptos2019-blindness-detection/test_images'  \noutput_csv_path = '/kaggle/working/submission.csv'  # Output CSV for predictions\n\n# Load the pretrained InceptionV3 model without the top layer\nbase_model = InceptionV3(weights=weights_path, include_top=False, pooling='avg')\n\n# Function to extract features from an image\ndef extract_features(img_path, model, img_size=(299, 299)):\n    img = image.load_img(img_path, target_size=img_size)\n    img_array = image.img_to_array(img)\n    img_array = np.expand_dims(img_array, axis=0)\n    img_array = preprocess_input(img_array)\n    features = model.predict(img_array)\n    return features.flatten()\n\n# Function to load and extract features from training data\ndef load_and_extract_features(csv_path, images_folder, model):\n    data = pd.read_csv(csv_path)\n    features = []\n    labels = []\n    for index, row in data.iterrows():\n        image_name = row['id_code']\n        label = row['diagnosis']\n        image_path = f\"{images_folder}/{image_name}.png\"\n        feature = extract_features(image_path, model)\n        features.append(feature)\n        labels.append(label)\n    return np.array(features), np.array(labels)\n\n# Function to load and extract features from test data\ndef load_test_data_and_extract_features(csv_path, images_folder, model):\n    data = pd.read_csv(csv_path)\n    features = []\n    image_names = []\n    for index, row in data.iterrows():\n        image_name = row['id_code']\n        image_path = f\"{images_folder}/{image_name}.png\"\n        feature = extract_features(image_path, model)\n        features.append(feature)\n        image_names.append(image_name)\n    return np.array(features), image_names\n\n# Function to generate predictions and save them to CSV\ndef predict_and_generate_csv(model, test_features, image_names, output_csv_path):\n    predictions = model.predict(test_features)\n    output_df = pd.DataFrame({\n        'id_code': image_names,\n        'diagnosis': predictions\n    })\n    output_df.to_csv(output_csv_path, index=False)\n    print(f\"Predictions saved to {output_csv_path}\")\n\n# Main Program\nif __name__ == \"__main__\":\n    # 1. Load and extract features from the training data\n    print(\"Loading and extracting features from training data...\")\n    X, y = load_and_extract_features(train_csv_path, train_images_folder, base_model)\n    \n    # 2. Encode labels to numeric values\n    label_encoder = LabelEncoder()\n    y_encoded = label_encoder.fit_transform(y)\n    \n    # 3. Split data into training and validation sets\n    X_train, X_val, y_train, y_val = train_test_split(X, y_encoded, test_size=0.2, random_state=42)\n    \n    # 4. Train the XGBoost model\n    print(\"Training XGBoost model...\")\n    xgb_model = XGBClassifier(n_estimators=100, random_state=42, use_label_encoder=False, eval_metric='mlogloss')\n    xgb_model.fit(X_train, y_train)\n    \n    # 5. Validate the model\n    y_val_pred = xgb_model.predict(X_val)\n    accuracy_xgb = accuracy_score(y_val, y_val_pred)\n    print(f\"XGBoost Validation Accuracy: {accuracy_xgb * 100:.2f}%\")\n    \n    # 6. Save the trained XGBoost model\n    joblib.dump(xgb_model, '/kaggle/working/xgboost_classifier.joblib')\n    print(\"Model saved as 'xgboost_classifier.joblib'.\")\n    \n    # 7. Load and extract features from the test data\n    print(\"Loading and extracting features from test data...\")\n    test_features, image_names = load_test_data_and_extract_features(test_csv_path, test_images_folder, base_model)\n    \n    # 8. Predict test labels and generate CSV output\n    print(\"Generating predictions and saving to CSV...\")\n    predict_and_generate_csv(xgb_model, test_features, image_names, output_csv_path)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-06T05:41:00.412271Z","iopub.execute_input":"2024-11-06T05:41:00.412668Z","iopub.status.idle":"2024-11-06T05:41:26.654871Z","shell.execute_reply.started":"2024-11-06T05:41:00.412603Z","shell.execute_reply":"2024-11-06T05:41:26.653565Z"},"trusted":true},"outputs":[],"execution_count":null}]}