{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"}],"dockerImageVersionId":30716,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.impute import KNNImputer\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Concatenate, Input, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, Callback\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.regularizers import l2\nimport psutil\nimport os\nimport time\nimport gc\nfrom sklearn.utils import class_weight\nimport math\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-17T01:20:00.942901Z","iopub.execute_input":"2024-06-17T01:20:00.943564Z","iopub.status.idle":"2024-06-17T01:20:00.951005Z","shell.execute_reply.started":"2024-06-17T01:20:00.943531Z","shell.execute_reply":"2024-06-17T01:20:00.949888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/train.csv')\ntest_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/test.csv')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:20:00.952375Z","iopub.execute_input":"2024-06-17T01:20:00.952704Z","iopub.status.idle":"2024-06-17T01:20:01.054820Z","shell.execute_reply.started":"2024-06-17T01:20:00.952679Z","shell.execute_reply":"2024-06-17T01:20:01.053895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Map 'sex' to binary values: male -> 1, female -> 0\ntrain_df['sex'] = train_df['sex'].map({'male': 1, 'female': 0})\ntest_df['sex'] = test_df['sex'].map({'male': 1, 'female': 0})\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:20:01.056778Z","iopub.execute_input":"2024-06-17T01:20:01.057533Z","iopub.status.idle":"2024-06-17T01:20:01.068267Z","shell.execute_reply.started":"2024-06-17T01:20:01.057497Z","shell.execute_reply":"2024-06-17T01:20:01.067369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fill missing values with 'unknown'\ntrain_df['anatom_site_general_challenge'] = train_df['anatom_site_general_challenge'].fillna('unknown')\ntest_df['anatom_site_general_challenge'] = test_df['anatom_site_general_challenge'].fillna('unknown')\n\n# Convert all values in 'anatom_site_general_challenge' and 'sex' to strings\ntrain_df['anatom_site_general_challenge'] = train_df['anatom_site_general_challenge'].astype(str)\ntest_df['anatom_site_general_challenge'] = test_df['anatom_site_general_challenge'].astype(str)\n\n# Encode 'anatom_site_general_challenge' column\nle_anatom_site = LabelEncoder()\ntrain_df['anatom_site_general_challenge'] = le_anatom_site.fit_transform(train_df['anatom_site_general_challenge'])\ntest_df['anatom_site_general_challenge'] = le_anatom_site.transform(test_df['anatom_site_general_challenge'])","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:20:01.069529Z","iopub.execute_input":"2024-06-17T01:20:01.069811Z","iopub.status.idle":"2024-06-17T01:20:01.095573Z","shell.execute_reply.started":"2024-06-17T01:20:01.069784Z","shell.execute_reply":"2024-06-17T01:20:01.094746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert 'target' to float\ntrain_df['target'] = train_df['target'].astype(float)","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:20:01.097639Z","iopub.execute_input":"2024-06-17T01:20:01.098424Z","iopub.status.idle":"2024-06-17T01:20:01.102733Z","shell.execute_reply.started":"2024-06-17T01:20:01.098398Z","shell.execute_reply":"2024-06-17T01:20:01.101740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_tf_data_generator(df, img_dir, batch_size=32, target_size=(128, 128), is_train=True):\n    def load_data(row):\n        img_path = tf.strings.join([img_dir, '/', row['image_name'], '.jpg'])\n        img = tf.io.read_file(img_path)\n        img = tf.image.decode_jpeg(img, channels=3)\n        img = tf.image.resize(img, target_size)\n        img = img / 255.0\n        patient_data = tf.stack([\n            tf.cast(row['sex'], tf.float32),\n            tf.cast(row['age_approx'], tf.float32),\n            tf.cast(row['anatom_site_general_challenge'], tf.float32)\n        ], axis=-1)\n        return img, patient_data\n\n    def load_data_with_labels(row):\n        img, patient_data = load_data(row)\n        label = tf.cast(row['target'], tf.float32)\n        return (img, patient_data), label\n\n    dataset = tf.data.Dataset.from_tensor_slices(dict(df))\n    if is_train:\n        dataset = dataset.shuffle(buffer_size=len(df))\n    dataset = dataset.map(load_data_with_labels, num_parallel_calls=tf.data.experimental.AUTOTUNE)\n    dataset = dataset.batch(batch_size)\n    dataset = dataset.prefetch(buffer_size=tf.data.experimental.AUTOTUNE)\n    return dataset\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:20:01.103987Z","iopub.execute_input":"2024-06-17T01:20:01.104319Z","iopub.status.idle":"2024-06-17T01:20:01.115182Z","shell.execute_reply.started":"2024-06-17T01:20:01.104288Z","shell.execute_reply":"2024-06-17T01:20:01.114385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create data generators\ntrain_generator = create_tf_data_generator(train_df, img_dir='/kaggle/input/siim-isic-melanoma-classification/jpeg/train', is_train=True)\nval_generator = create_tf_data_generator(train_df.sample(frac=0.2, random_state=42), img_dir='/kaggle/input/siim-isic-melanoma-classification/jpeg/train', is_train=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:20:01.116319Z","iopub.execute_input":"2024-06-17T01:20:01.116863Z","iopub.status.idle":"2024-06-17T01:20:01.212672Z","shell.execute_reply.started":"2024-06-17T01:20:01.116838Z","shell.execute_reply":"2024-06-17T01:20:01.211943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the directory containing the test images\ntest_img_dir = '/kaggle/input/siim-isic-melanoma-classification/jpeg/test'\n\n# Filter test_df to include only the image files that exist in the directory\nexisting_files = os.listdir(test_img_dir)\ntest_df = test_df[test_df['image_name'].apply(lambda x: f'{x}.jpg' in existing_files)]\n\ndef create_tf_test_data_generator(df, img_dir, batch_size=32, target_size=(128, 128)):\n    def load_data(row):\n        img_path = tf.strings.join([img_dir, '/', row['image_name'], '.jpg'])\n        img = tf.io.read_file(img_path)\n        img = tf.image.decode_jpeg(img, channels=3)\n        img = tf.image.resize(img, target_size)\n        img = img / 255.0\n        patient_data = tf.stack([\n            tf.cast(row['sex'], tf.float32),\n            tf.cast(row['age_approx'], tf.float32),\n            tf.cast(row['anatom_site_general_challenge'], tf.float32)\n        ], axis=-1)\n        return img, patient_data\n\n    def load_data_no_labels(row):\n        img, patient_data = load_data(row)\n        return img, patient_data\n\n    dataset = tf.data.Dataset.from_tensor_slices(dict(df))\n    dataset = dataset.map(load_data_no_labels, num_parallel_calls=tf.data.experimental.AUTOTUNE)\n    dataset = dataset.batch(batch_size)\n    dataset = dataset.prefetch(buffer_size=tf.data.experimental.AUTOTUNE)\n    return dataset\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:20:01.214857Z","iopub.execute_input":"2024-06-17T01:20:01.215129Z","iopub.status.idle":"2024-06-17T01:20:02.724406Z","shell.execute_reply.started":"2024-06-17T01:20:01.215105Z","shell.execute_reply":"2024-06-17T01:20:02.723346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for NaN values in the input data\nprint(test_df.isna().sum())\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:20:02.725709Z","iopub.execute_input":"2024-06-17T01:20:02.726072Z","iopub.status.idle":"2024-06-17T01:20:02.734237Z","shell.execute_reply.started":"2024-06-17T01:20:02.726037Z","shell.execute_reply":"2024-06-17T01:20:02.733368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Create test data generator\ntest_generator = create_tf_test_data_generator(test_df, img_dir=test_img_dir)\n\n# Generate predictions\ntest_images, test_patient_data = [], []\nfor batch in test_generator:\n    images, patient_data = batch\n    test_images.append(images)\n    test_patient_data.append(patient_data)\n\ntest_images = tf.concat(test_images, axis=0)\ntest_patient_data = tf.concat(test_patient_data, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:20:02.735523Z","iopub.execute_input":"2024-06-17T01:20:02.735822Z","iopub.status.idle":"2024-06-17T01:24:02.766153Z","shell.execute_reply.started":"2024-06-17T01:20:02.735797Z","shell.execute_reply":"2024-06-17T01:24:02.765250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ensure shapes are as expected\nprint(f\"Test images shape: {test_images.shape}\")\nprint(f\"Test patient data shape: {test_patient_data.shape}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:24:02.767414Z","iopub.execute_input":"2024-06-17T01:24:02.767750Z","iopub.status.idle":"2024-06-17T01:24:02.772938Z","shell.execute_reply.started":"2024-06-17T01:24:02.767725Z","shell.execute_reply":"2024-06-17T01:24:02.771813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:24:02.774249Z","iopub.execute_input":"2024-06-17T01:24:02.774631Z","iopub.status.idle":"2024-06-17T01:24:02.796034Z","shell.execute_reply.started":"2024-06-17T01:24:02.774598Z","shell.execute_reply":"2024-06-17T01:24:02.795089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build the model\nimage_input = Input(shape=(128, 128, 3))\nbase_model = EfficientNetB0(weights='imagenet', include_top=False, input_tensor=image_input)\nx = GlobalAveragePooling2D()(base_model.output)\nx = Dropout(0.5)(x)\nimage_features = Model(inputs=image_input, outputs=x)\n\n# Define patient data input\npatient_input = Input(shape=(train_df[['sex', 'age_approx', 'anatom_site_general_challenge']].shape[1],))\ny = Dense(128, activation='relu', kernel_regularizer=l2(0.01))(patient_input)\ny = Dropout(0.5)(y)\ny = Dense(64, activation='relu', kernel_regularizer=l2(0.01))(y)\n\n# Combine image and patient data features\ncombined = Concatenate()([image_features.output, y])\nz = Dense(64, activation='relu', kernel_regularizer=l2(0.01))(combined)\nz = Dropout(0.5)(z)\noutput = Dense(1, activation='sigmoid', dtype=tf.float32)(z)  # Ensure correct output dtype for mixed precision\n\n# Define and compile the model\noptimizer = Adam(learning_rate=1e-5, clipvalue=1.0)\nmodel = Model(inputs=[image_features.input, patient_input], outputs=output)\nmodel.compile(optimizer=optimizer, loss='binary_crossentropy', metrics=['accuracy'])\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:24:02.797239Z","iopub.execute_input":"2024-06-17T01:24:02.797573Z","iopub.status.idle":"2024-06-17T01:24:04.219777Z","shell.execute_reply.started":"2024-06-17T01:24:02.797544Z","shell.execute_reply":"2024-06-17T01:24:04.218816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to monitor memory usage\ndef print_memory_usage():\n    process = psutil.Process(os.getpid())\n    print(f\"Memory Usage: {process.memory_info().rss / 1024 ** 2:.2f} MB\")\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:24:04.221082Z","iopub.execute_input":"2024-06-17T01:24:04.221375Z","iopub.status.idle":"2024-06-17T01:24:04.226389Z","shell.execute_reply.started":"2024-06-17T01:24:04.221349Z","shell.execute_reply":"2024-06-17T01:24:04.225505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Custom callback to log epoch duration and monitor memory\nclass MemoryCallback(Callback):\n    def on_epoch_begin(self, epoch, logs=None):\n        self.epoch_time_start = time.time()\n\n    def on_epoch_end(self, epoch, logs=None):\n        print(f\"Epoch {epoch+1} took {time.time() - self.epoch_time_start:.2f} seconds\")\n        print_memory_usage()\n        gc.collect()  # Trigger garbage collection to free up memory","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:24:04.227576Z","iopub.execute_input":"2024-06-17T01:24:04.227903Z","iopub.status.idle":"2024-06-17T01:24:04.234186Z","shell.execute_reply.started":"2024-06-17T01:24:04.227869Z","shell.execute_reply":"2024-06-17T01:24:04.233390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Callbacks\n\nearly_stopping = EarlyStopping(monitor='val_loss', patience=5, min_delta=0.001, mode='min', restore_best_weights=True, verbose=1)\nreduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=5, min_lr=1e-6, verbose=1)\nmemory_callback = MemoryCallback()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:24:04.238036Z","iopub.execute_input":"2024-06-17T01:24:04.238386Z","iopub.status.idle":"2024-06-17T01:24:04.243555Z","shell.execute_reply.started":"2024-06-17T01:24:04.238344Z","shell.execute_reply":"2024-06-17T01:24:04.242733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate class weights\nclass_weights = class_weight.compute_class_weight(\n    class_weight='balanced',\n    classes=np.unique(train_df['target'].values),\n    y=train_df['target'].values\n)\n\nclass_weights_dict = dict(enumerate(class_weights))\nprint(f\"Class weights: {class_weights_dict}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:24:04.244635Z","iopub.execute_input":"2024-06-17T01:24:04.244880Z","iopub.status.idle":"2024-06-17T01:24:04.265678Z","shell.execute_reply.started":"2024-06-17T01:24:04.244858Z","shell.execute_reply":"2024-06-17T01:24:04.264851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model\nhistory = model.fit(\n    train_generator,\n    validation_data=val_generator,\n    epochs=10,\n    steps_per_epoch=50,\n    validation_steps=100,\n    class_weight=class_weights_dict,\n    callbacks=[early_stopping, reduce_lr, memory_callback]\n)","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:30:39.171737Z","iopub.execute_input":"2024-06-17T01:30:39.172447Z","iopub.status.idle":"2024-06-17T01:31:19.529550Z","shell.execute_reply.started":"2024-06-17T01:30:39.172410Z","shell.execute_reply":"2024-06-17T01:31:19.518258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions on test data\ntest_predictions = model.predict([test_images, test_patient_data], verbose=1)","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:29:05.042552Z","iopub.execute_input":"2024-06-17T01:29:05.042838Z","iopub.status.idle":"2024-06-17T01:29:20.109082Z","shell.execute_reply.started":"2024-06-17T01:29:05.042813Z","shell.execute_reply":"2024-06-17T01:29:20.108304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.isnan(test_predictions).sum()","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:29:20.110419Z","iopub.execute_input":"2024-06-17T01:29:20.110746Z","iopub.status.idle":"2024-06-17T01:29:20.117157Z","shell.execute_reply.started":"2024-06-17T01:29:20.110720Z","shell.execute_reply":"2024-06-17T01:29:20.116239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare the submission\nsubmission = pd.DataFrame({'image_name': test_df['image_name'],'target': test_predictions.squeeze()})\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:29:20.118673Z","iopub.execute_input":"2024-06-17T01:29:20.119018Z","iopub.status.idle":"2024-06-17T01:29:20.142679Z","shell.execute_reply.started":"2024-06-17T01:29:20.118969Z","shell.execute_reply":"2024-06-17T01:29:20.142012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2024-06-17T01:29:20.143593Z","iopub.execute_input":"2024-06-17T01:29:20.143848Z","iopub.status.idle":"2024-06-17T01:29:20.153657Z","shell.execute_reply.started":"2024-06-17T01:29:20.143825Z","shell.execute_reply":"2024-06-17T01:29:20.152676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}