{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"},{"sourceId":104884,"sourceType":"datasetVersion","datasetId":54339}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom PIL import Image\nfrom keras.utils import to_categorical\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array, ImageDataGenerator\nimport os\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras import layers, Sequential\nimport random\nimport matplotlib.pyplot as plt\nimport PIL\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-15T16:07:04.31849Z","iopub.execute_input":"2024-02-15T16:07:04.318835Z","iopub.status.idle":"2024-02-15T16:07:19.374Z","shell.execute_reply.started":"2024-02-15T16:07:04.318811Z","shell.execute_reply":"2024-02-15T16:07:19.372908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Path to the HAM10000 dataset CSV file\ncsv_path = '/kaggle/input/skin-cancer-mnist-ham10000/HAM10000_metadata.csv'\n# Path to the HAM10000 dataset images folder\nimage_folder = '/kaggle/input/skin-cancer-mnist-ham10000/HAM10000_images_part_1'\nimage_folder2 = '/kaggle/input/skin-cancer-mnist-ham10000/HAM10000_images_part_2'\n\n\n# Path to the SIIM ISIC dataset CSV file\nsiim_csv_path = '/kaggle/input/siim-isic-melanoma-classification/test.csv'\n# Path to the SIIM ISIC dataset images folder\nsiim_image_folder = '/kaggle/input/siim-isic-melanoma-classification/jpeg/test'\n\n\nmetadata = pd.read_csv(csv_path)\nsiim_metadata = pd.read_csv(siim_csv_path)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-02-15T16:07:19.37593Z","iopub.execute_input":"2024-02-15T16:07:19.376505Z","iopub.status.idle":"2024-02-15T16:07:19.433293Z","shell.execute_reply.started":"2024-02-15T16:07:19.376478Z","shell.execute_reply":"2024-02-15T16:07:19.43219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dictionary to map classes to numeric labels\nclass_to_label = {name: index for index, name in enumerate(metadata['dx'].unique())}\n\n# Load and preprocess the images and labels\nimages = []\nlabels = []\n\nfor index, row in metadata.iterrows():\n    image_path = f\"{image_folder}/{row['image_id']}.jpg\"\n    if not os.path.exists(image_path):\n        image_path = f\"{image_folder2}/{row['image_id']}.jpg\"\n\n    image = load_img(image_path, target_size=(100, 100))  # Resize the image if necessary\n    image = np.array(image)\n    label = class_to_label[row['dx']]\n    \n    images.append(image)\n    labels.append(label)\n\nimages=images[:10000]\nlabels=labels[:10000]\n# Convert the lists into numpy arrays\nimages = np.array(images)\nlabels = np.array(labels)\n\n\n# Normalize the image data\nimages = images / 255.0\n\n# Convert the labels into one-hot encoded vectors\nnum_classes = len(class_to_label)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-15T16:07:19.434991Z","iopub.execute_input":"2024-02-15T16:07:19.43542Z","iopub.status.idle":"2024-02-15T16:10:08.565438Z","shell.execute_reply.started":"2024-02-15T16:07:19.435387Z","shell.execute_reply":"2024-02-15T16:10:08.564024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load and preprocess the SIIM ISIC images and labels\nsiim_images = []\nsiim_labels = []\n\nfor index, row in siim_metadata.iterrows():\n    image_name = row['image_name'] + '.jpg'\n    image_path = os.path.join(siim_image_folder, image_name)\n\n    image = load_img(image_path, target_size=(100, 100))  # Resize the image if necessary\n    image = np.array(image)\n    label = 1\n    \n    siim_images.append(image)\n    siim_labels.append(label)\n\nsiim_images = np.array(siim_images)\nsiim_labels = np.array(siim_labels)\n\n# Normalize the SIIM ISIC image data\nsiim_images = siim_images / 255.0\n","metadata":{"execution":{"iopub.status.busy":"2024-02-15T16:10:08.567427Z","iopub.execute_input":"2024-02-15T16:10:08.567743Z","iopub.status.idle":"2024-02-15T16:23:51.992197Z","shell.execute_reply.started":"2024-02-15T16:10:08.567717Z","shell.execute_reply":"2024-02-15T16:23:51.990066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = images\ny_train = labels\n\nx_test= siim_images[:len(images)]  \ny_test = siim_labels[:len(labels)]","metadata":{"execution":{"iopub.status.busy":"2024-02-15T16:23:51.995525Z","iopub.execute_input":"2024-02-15T16:23:51.996204Z","iopub.status.idle":"2024-02-15T16:23:52.005136Z","shell.execute_reply.started":"2024-02-15T16:23:51.996147Z","shell.execute_reply":"2024-02-15T16:23:52.003672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print (images.size)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T16:23:52.006533Z","iopub.execute_input":"2024-02-15T16:23:52.00688Z","iopub.status.idle":"2024-02-15T16:23:52.021909Z","shell.execute_reply.started":"2024-02-15T16:23:52.006854Z","shell.execute_reply":"2024-02-15T16:23:52.020206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-02-15T16:24:41.955659Z","iopub.execute_input":"2024-02-15T16:24:41.956107Z","iopub.status.idle":"2024-02-15T16:24:41.966453Z","shell.execute_reply.started":"2024-02-15T16:24:41.956075Z","shell.execute_reply":"2024-02-15T16:24:41.964652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"siim_csv_path1 = '/kaggle/input/siim-isic-melanoma-classification/sample_submission.csv'\nsiim_metadata1 = pd.read_csv(siim_csv_path1)\nx= siim_metadata1.drop(columns=[\"target\"])\ny = siim_metadata1[\"target\"]\n\n# Split the data into train and validation sets\nx_train, x_val, y_train, y_val = train_test_split(x, y, test_size=0.2, random_state=42)\n\n# Initialize the Random Forest classifier\nrf_classifier = RandomForestClassifier(n_estimators=100, random_state=42)\n\n# Train the classifier\nrf_classifier.fit(x_train, y_train)\n\n# Predict on the validation set\ny_pred = rf_classifier.predict(x_val)\n\n# Calculate accuracy\naccuracy = accuracy_score(y_val, y_pred)\nprint(\"Validation Accuracy:\", accuracy)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T16:35:20.798753Z","iopub.execute_input":"2024-02-15T16:35:20.79918Z","iopub.status.idle":"2024-02-15T16:35:20.847118Z","shell.execute_reply.started":"2024-02-15T16:35:20.799139Z","shell.execute_reply":"2024-02-15T16:35:20.845667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the test data\n#metadata = pd.read_csv('/kaggle/input/skin-cancer-mnist-ham10000')\n\n# Make predictions on the test data\ntest_predictions = rf_classifier.predict(metadata)\n\n# Assuming you have ground truth labels for evaluation\ntest_labels = pd.read_csv(\"skin-cancer-mnist-ham10000/test_labels.csv\")\ntest_accuracy = accuracy_score(test_labels, test_predictions)\nprint(\"Test Accuracy:\", test_accuracy)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T16:27:44.076776Z","iopub.execute_input":"2024-02-15T16:27:44.077194Z","iopub.status.idle":"2024-02-15T16:27:44.116087Z","shell.execute_reply.started":"2024-02-15T16:27:44.077162Z","shell.execute_reply":"2024-02-15T16:27:44.114026Z"},"trusted":true},"execution_count":null,"outputs":[]}]}