{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-27T15:30:02.942701Z","iopub.execute_input":"2024-10-27T15:30:02.943025Z","iopub.status.idle":"2024-10-27T15:30:03.905122Z","shell.execute_reply.started":"2024-10-27T15:30:02.942990Z","shell.execute_reply":"2024-10-27T15:30:03.904334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install xgboost","metadata":{"execution":{"iopub.status.busy":"2024-10-27T15:30:23.268427Z","iopub.execute_input":"2024-10-27T15:30:23.269260Z","iopub.status.idle":"2024-10-27T15:30:23.273070Z","shell.execute_reply.started":"2024-10-27T15:30:23.269221Z","shell.execute_reply":"2024-10-27T15:30:23.272114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport cv2\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score, roc_curve\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.preprocessing import LabelEncoder, OneHotEncoder\nfrom sklearn.impute import SimpleImputer\nimport xgboost as xgb\nimport matplotlib.pyplot as plt\n\n# Load the train and test datasets\ntrain_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/train.csv')\ntest_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/test.csv')\n\n# Path to images\ntrain_image_dir = '/kaggle/input/siim-isic-melanoma-classification/jpeg/train/'\ntest_image_dir = '/kaggle/input/siim-isic-melanoma-classification/jpeg/test/'\n\n# Set constants\nIMG_SIZE = 196  # Reduced image size for faster computation\nBATCH_SIZE = 16\n\n# Image preprocessing function\ndef preprocess_image(image_path, img_size=IMG_SIZE):\n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = cv2.resize(img, (img_size, img_size))\n    img = img / 255.0  # Normalize\n    return img\n\ndef load_and_preprocess_image(image_name, label=None, is_train=True):\n    image_path = tf.strings.join([train_image_dir, image_name, '.jpg'], separator='')\n    image = tf.io.read_file(image_path)\n    image = tf.image.decode_jpeg(image, channels=3)\n    image = tf.image.resize(image, [IMG_SIZE, IMG_SIZE])\n    image = image / 255.0  # Normalize\n\n    if is_train:\n        image = tf.image.random_flip_left_right(image)\n        image = tf.image.random_brightness(image, max_delta=0.1)\n        image = tf.image.random_contrast(image, lower=0.8, upper=1.2)\n        image = tf.image.random_saturation(image, lower=0.8, upper=1.2)\n        image = tf.image.random_hue(image, max_delta=0.1)\n\n    if label is not None:\n        return image, label\n    return image\n\n\ndef create_dataset(df, is_train=True):\n    dataset = tf.data.Dataset.from_tensor_slices((df['image_name'], df['target']))\n    dataset = dataset.map(lambda x, y: (load_and_preprocess_image(x, y)), \n                          num_parallel_calls=tf.data.AUTOTUNE)\n    if is_train:\n        dataset = dataset.shuffle(buffer_size=len(df)).batch(BATCH_SIZE).prefetch(tf.data.experimental.AUTOTUNE)\n    else:\n        dataset = dataset.batch(BATCH_SIZE).prefetch(tf.data.experimental.AUTOTUNE)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2024-10-27T16:09:19.229035Z","iopub.execute_input":"2024-10-27T16:09:19.229429Z","iopub.status.idle":"2024-10-27T16:09:29.694953Z","shell.execute_reply.started":"2024-10-27T16:09:19.229378Z","shell.execute_reply":"2024-10-27T16:09:29.694085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for path in train_df['image_name']:\n#     if not os.path.exists(path):\n#         print(f\"File not found: {path}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-27T16:09:32.548878Z","iopub.execute_input":"2024-10-27T16:09:32.549795Z","iopub.status.idle":"2024-10-27T16:09:32.553708Z","shell.execute_reply.started":"2024-10-27T16:09:32.549753Z","shell.execute_reply":"2024-10-27T16:09:32.552515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-27T16:09:33.493934Z","iopub.execute_input":"2024-10-27T16:09:33.494948Z","iopub.status.idle":"2024-10-27T16:09:33.503227Z","shell.execute_reply.started":"2024-10-27T16:09:33.494888Z","shell.execute_reply":"2024-10-27T16:09:33.502049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"target\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T16:09:34.154334Z","iopub.execute_input":"2024-10-27T16:09:34.155386Z","iopub.status.idle":"2024-10-27T16:09:34.174267Z","shell.execute_reply.started":"2024-10-27T16:09:34.155329Z","shell.execute_reply":"2024-10-27T16:09:34.173345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"anatom_site_general_challenge\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T16:09:34.683506Z","iopub.execute_input":"2024-10-27T16:09:34.684472Z","iopub.status.idle":"2024-10-27T16:09:34.696258Z","shell.execute_reply.started":"2024-10-27T16:09:34.684427Z","shell.execute_reply":"2024-10-27T16:09:34.695255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.dropna(subset=['sex'])","metadata":{"execution":{"iopub.status.busy":"2024-10-27T16:09:35.273761Z","iopub.execute_input":"2024-10-27T16:09:35.274688Z","iopub.status.idle":"2024-10-27T16:09:35.296003Z","shell.execute_reply.started":"2024-10-27T16:09:35.274638Z","shell.execute_reply":"2024-10-27T16:09:35.295144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"sex\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T16:09:35.868822Z","iopub.execute_input":"2024-10-27T16:09:35.869651Z","iopub.status.idle":"2024-10-27T16:09:35.882881Z","shell.execute_reply.started":"2024-10-27T16:09:35.869603Z","shell.execute_reply":"2024-10-27T16:09:35.881639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T16:09:36.584098Z","iopub.execute_input":"2024-10-27T16:09:36.584538Z","iopub.status.idle":"2024-10-27T16:09:36.603555Z","shell.execute_reply.started":"2024-10-27T16:09:36.584497Z","shell.execute_reply":"2024-10-27T16:09:36.602612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fill missing values for age\ntrain_df['age_approx'].fillna(train_df['age_approx'].mean(), inplace=True)\n\n# Fill missing categorical features with 'unknown'\ntrain_df['sex'].fillna('unknown', inplace=True)\ntrain_df['anatom_site_general_challenge'].fillna(train_df['anatom_site_general_challenge'].mode()[0], inplace=True)\n# train_df['diagnosis'].fillna('unknown', inplace=True)\n# train_df['benign_malignant'].fillna('unknown', inplace=True)\n\n# Encode categorical features using one-hot encoding\none_hot_columns = ['sex', 'anatom_site_general_challenge']\ntrain_df = pd.get_dummies(train_df, columns=one_hot_columns)\n\ny = train_df['target']\n\n# Train-Validation Split\n# Set the random seed for reproducibility\nnp.random.seed(0)\n\n# Generate one random number between 0 and 100\nrandom_number = np.random.randint(0, 50) \ntrain_df, val_df = train_test_split(train_df, test_size=0.2, random_state=random_number)\n\ntrain_dataset = create_dataset(train_df)\nval_dataset = create_dataset(val_df, is_train=False)\n\ntrain_dataset = train_dataset.prefetch(tf.data.experimental.AUTOTUNE)\nval_dataset = val_dataset.prefetch(tf.data.experimental.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T16:09:37.259598Z","iopub.execute_input":"2024-10-27T16:09:37.260191Z","iopub.status.idle":"2024-10-27T16:09:38.297421Z","shell.execute_reply.started":"2024-10-27T16:09:37.260146Z","shell.execute_reply":"2024-10-27T16:09:38.296354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T16:09:38.299093Z","iopub.execute_input":"2024-10-27T16:09:38.299452Z","iopub.status.idle":"2024-10-27T16:09:38.316434Z","shell.execute_reply.started":"2024-10-27T16:09:38.299416Z","shell.execute_reply":"2024-10-27T16:09:38.315432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils import class_weight\nfrom sklearn.metrics import classification_report\n# import tensorflow_addons as tfa\n# from tensorflow_addons.optimizers import CyclicalLearningRate\n\nlr_schedule = tf.keras.optimizers.schedules.ExponentialDecay(\n    initial_learning_rate=1e-4, decay_steps=10000, decay_rate=0.9)\n\nsteps_per_epoch = len(train_df) // BATCH_SIZE\nvalidation_steps = len(val_df) // BATCH_SIZE\n\n# early_stopping = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=3, restore_best_weights=True)\nearly_stopping = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=3, restore_best_weights=True)\n\n# Calculate class weights\nclass_weights = class_weight.compute_class_weight(\n    class_weight='balanced',\n    classes=np.unique(y),\n    y=y\n)\n\n# Convert class weights to a dictionary\nclass_weights = {0: class_weights[0], 1: class_weights[1]}\n","metadata":{"execution":{"iopub.status.busy":"2024-10-27T16:09:38.838145Z","iopub.execute_input":"2024-10-27T16:09:38.839139Z","iopub.status.idle":"2024-10-27T16:09:38.877393Z","shell.execute_reply.started":"2024-10-27T16:09:38.839087Z","shell.execute_reply":"2024-10-27T16:09:38.876432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# try:\n#     tpu = tf.distribute.cluster_resolver.TPUClusterResolver()  # Detect TPU\n#     tf.config.experimental_connect_to_cluster(tpu)\n#     tf.tpu.experimental.initialize_tpu_system(tpu)\n#     strategy = tf.distribute.TPUStrategy(tpu)\n# except ValueError:\n#     strategy = tf.distribute.get_strategy()  # Fallback for no TPU","metadata":{"execution":{"iopub.status.busy":"2024-10-27T16:09:39.545277Z","iopub.execute_input":"2024-10-27T16:09:39.546219Z","iopub.status.idle":"2024-10-27T16:09:39.550579Z","shell.execute_reply.started":"2024-10-27T16:09:39.546177Z","shell.execute_reply":"2024-10-27T16:09:39.549343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_and_preprocess_image_test(image_name, label=None):\n    image_path = tf.strings.join([test_image_dir, image_name, '.jpg'], separator='')\n    image = tf.io.read_file(image_path)\n    image = tf.image.decode_jpeg(image, channels=3)\n    image = tf.image.resize(image, [IMG_SIZE, IMG_SIZE])\n    image = image / 255.0  # Normalize\n    \n    # Move data augmentation to GPU\n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_brightness(image, max_delta=0.1)\n    \n    if label is not None:\n        return image, label\n    return image\n\n# Prepare the test dataset\ndef create_test_dataset(df):\n    dataset = tf.data.Dataset.from_tensor_slices(df['image_name'])\n    dataset = dataset.map(lambda x: load_and_preprocess_image_test(x), num_parallel_calls=tf.data.AUTOTUNE)\n    dataset = dataset.batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2024-10-27T16:09:40.548773Z","iopub.execute_input":"2024-10-27T16:09:40.549176Z","iopub.status.idle":"2024-10-27T16:09:40.558206Z","shell.execute_reply.started":"2024-10-27T16:09:40.549138Z","shell.execute_reply":"2024-10-27T16:09:40.557208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.drop(columns=['image_name', 'patient_id', 'diagnosis', 'benign_malignant'], axis=1)\nval_df = val_df.drop(columns=['image_name', 'patient_id', 'diagnosis', 'benign_malignant'], axis=1)\n\n# Extract tabular features for RandomForest and XGBoost\nX_train_tabular = train_df.drop(columns=['target'], axis=1)\ny_train = train_df['target']\n\nX_val_tabular = val_df.drop(columns=['target'], axis=1)\ny_val = val_df['target']\n\n# Train RandomForest\n\nrf_model = RandomForestClassifier(n_estimators=100, random_state=42)\nrf_model.fit(X_train_tabular, y_train)\n\n# Train XGBoost\nxgb_model = xgb.XGBClassifier(use_label_encoder=False, eval_metric='logloss')\nxgb_model.fit(X_train_tabular, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T16:09:43.112889Z","iopub.execute_input":"2024-10-27T16:09:43.113320Z","iopub.status.idle":"2024-10-27T16:09:44.484692Z","shell.execute_reply.started":"2024-10-27T16:09:43.113261Z","shell.execute_reply":"2024-10-27T16:09:44.483793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fill missing values for age\ntest_df['age_approx'].fillna(test_df['age_approx'].mean(), inplace=True)\n\n# Fill missing categorical features with 'unknown'\ntest_df['sex'].fillna('unknown', inplace=True)\ntest_df['anatom_site_general_challenge'].fillna(test_df['anatom_site_general_challenge'].mode()[0], inplace=True)\n# train_df['diagnosis'].fillna('unknown', inplace=True)\n# train_df['benign_malignant'].fillna('unknown', inplace=True)\n\n# Encode categorical features using one-hot encoding\none_hot_columns = ['sex', 'anatom_site_general_challenge']\ntest_df = pd.get_dummies(test_df, columns=one_hot_columns)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T16:09:45.894242Z","iopub.execute_input":"2024-10-27T16:09:45.895055Z","iopub.status.idle":"2024-10-27T16:09:45.914346Z","shell.execute_reply.started":"2024-10-27T16:09:45.895010Z","shell.execute_reply":"2024-10-27T16:09:45.913341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = create_test_dataset(test_df)\ntest_dataset = test_dataset.prefetch(tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T16:09:49.543834Z","iopub.execute_input":"2024-10-27T16:09:49.544691Z","iopub.status.idle":"2024-10-27T16:09:49.898785Z","shell.execute_reply.started":"2024-10-27T16:09:49.544640Z","shell.execute_reply":"2024-10-27T16:09:49.897951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# with strategy.scope():\n# Define your model here (using EfficientNetB4)\nbase_model = tf.keras.applications.EfficientNetB3(\n    include_top=False,\n    weights='imagenet',\n    input_shape=(IMG_SIZE, IMG_SIZE, 3)\n)\n\nbase_model.trainable = False\n\n# Custom classification head\nmodel = tf.keras.Sequential([\n    base_model,\n    tf.keras.layers.GlobalAveragePooling2D(),\n    tf.keras.layers.Dense(128, activation='relu'),\n    tf.keras.layers.Dropout(0.5),\n    tf.keras.layers.Dense(1, activation='sigmoid')  # Adjust for binary classification\n])\n\nbase_model.trainable = True\n\n\n# Compile the model\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.0001),\n    loss='binary_crossentropy',\n    metrics=['accuracy']\n)\n\n# Train the model\nhistory = model.fit(\n    train_dataset,\n    epochs=4,\n    validation_data=val_dataset,\n    class_weight=class_weights,\n    callbacks=[early_stopping],\n    verbose=1\n)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T16:10:08.786339Z","iopub.execute_input":"2024-10-27T16:10:08.786744Z","iopub.status.idle":"2024-10-27T17:32:51.389599Z","shell.execute_reply.started":"2024-10-27T16:10:08.786710Z","shell.execute_reply":"2024-10-27T17:32:51.388479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_proba_cnn = model.predict(val_dataset)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T17:39:22.887907Z","iopub.execute_input":"2024-10-27T17:39:22.888346Z","iopub.status.idle":"2024-10-27T17:42:29.842121Z","shell.execute_reply.started":"2024-10-27T17:39:22.888277Z","shell.execute_reply":"2024-10-27T17:42:29.841086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(type(test_features))","metadata":{"execution":{"iopub.status.busy":"2024-10-27T15:19:46.901726Z","iopub.execute_input":"2024-10-27T15:19:46.902125Z","iopub.status.idle":"2024-10-27T15:19:46.906784Z","shell.execute_reply.started":"2024-10-27T15:19:46.902092Z","shell.execute_reply":"2024-10-27T15:19:46.905947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(f\"Model input shape: {model.input_shape}\")\n# print(f\"Test features shape: {test_features.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-27T17:42:58.912956Z","iopub.execute_input":"2024-10-27T17:42:58.913798Z","iopub.status.idle":"2024-10-27T17:42:58.918032Z","shell.execute_reply.started":"2024-10-27T17:42:58.913755Z","shell.execute_reply":"2024-10-27T17:42:58.916938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_predictions = model.predict(test_features)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T17:43:13.448601Z","iopub.execute_input":"2024-10-27T17:43:13.449394Z","iopub.status.idle":"2024-10-27T17:43:13.453311Z","shell.execute_reply.started":"2024-10-27T17:43:13.449343Z","shell.execute_reply":"2024-10-27T17:43:13.452351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# with strategy.scope():\n# test_predictions = model.predict(test_dataset)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T17:43:33.917686Z","iopub.execute_input":"2024-10-27T17:43:33.918459Z","iopub.status.idle":"2024-10-27T17:43:33.922451Z","shell.execute_reply.started":"2024-10-27T17:43:33.918415Z","shell.execute_reply":"2024-10-27T17:43:33.921404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract features from the test_dataset\n# valid_features = np.concatenate([x for x, _ in val_dataset], axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T17:43:49.833279Z","iopub.execute_input":"2024-10-27T17:43:49.833745Z","iopub.status.idle":"2024-10-27T17:43:49.838080Z","shell.execute_reply.started":"2024-10-27T17:43:49.833704Z","shell.execute_reply":"2024-10-27T17:43:49.837115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CNN Model AUC-ROC\n# y_pred_proba_cnn = model.predict(val)\nauc_cnn = roc_auc_score(y_val, y_pred_proba_cnn)\nprint(f'CNN Model AUC: {auc_cnn}')\n\n# RandomForest AUC-ROC\ny_pred_proba_rf = rf_model.predict_proba(X_val_tabular)[:, 1]\nauc_rf = roc_auc_score(y_val, y_pred_proba_rf)\nprint(f'RandomForest Model AUC: {auc_rf}')\n\n# XGBoost AUC-ROC\ny_pred_proba_xgb = xgb_model.predict_proba(X_val_tabular)[:, 1]\nauc_xgb = roc_auc_score(y_val, y_pred_proba_xgb)\nprint(f'XGBoost Model AUC: {auc_xgb}')","metadata":{"execution":{"iopub.status.busy":"2024-10-27T17:44:09.224970Z","iopub.execute_input":"2024-10-27T17:44:09.225506Z","iopub.status.idle":"2024-10-27T17:44:09.351992Z","shell.execute_reply.started":"2024-10-27T17:44:09.225453Z","shell.execute_reply":"2024-10-27T17:44:09.351233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Average the predicted probabilities (Ensemble)\nensemble_proba = (y_pred_proba_cnn.flatten() + y_pred_proba_rf + y_pred_proba_xgb) / 3\nauc_ensemble = roc_auc_score(y_val, ensemble_proba)\nprint(f'Ensemble Model AUC: {auc_ensemble}')","metadata":{"execution":{"iopub.status.busy":"2024-10-27T17:45:07.949061Z","iopub.execute_input":"2024-10-27T17:45:07.949784Z","iopub.status.idle":"2024-10-27T17:45:07.961192Z","shell.execute_reply.started":"2024-10-27T17:45:07.949742Z","shell.execute_reply":"2024-10-27T17:45:07.959798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predictions = model.predict(test_dataset)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T17:46:23.675361Z","iopub.execute_input":"2024-10-27T17:46:23.676098Z","iopub.status.idle":"2024-10-27T17:50:47.443520Z","shell.execute_reply.started":"2024-10-27T17:46:23.676057Z","shell.execute_reply":"2024-10-27T17:50:47.442598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = test_df.drop(columns=['image_name', 'patient_id'], axis=1)\n\n# Extract tabular features for RandomForest and XGBoost\nX_test_tabular = test\n\ny_pred_proba_rf = rf_model.predict_proba(X_test_tabular)[:, 1]\n\n# XGBoost AUC-ROC\ny_pred_proba_xgb = xgb_model.predict_proba(X_test_tabular)[:, 1]\n\nensemble_proba = (test_predictions.flatten() + y_pred_proba_rf + y_pred_proba_xgb) / 3\n\n# Create submission file\nsubmission_df = pd.DataFrame({'image_name': test_df['image_name'], 'target': ensemble_proba})\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T17:51:20.894814Z","iopub.execute_input":"2024-10-27T17:51:20.895913Z","iopub.status.idle":"2024-10-27T17:51:21.139812Z","shell.execute_reply.started":"2024-10-27T17:51:20.895851Z","shell.execute_reply":"2024-10-27T17:51:21.138767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}