{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:39:43.520222Z","iopub.execute_input":"2024-12-07T11:39:43.521007Z","iopub.status.idle":"2024-12-07T11:41:51.432877Z","shell.execute_reply.started":"2024-12-07T11:39:43.520965Z","shell.execute_reply":"2024-12-07T11:41:51.431802Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Generated Model Configuration**","metadata":{}},{"cell_type":"code","source":"# Importing necessary libraries\nimport os\nimport tensorflow as tf\n\n# Validate inputs\ndef validate_inputs(image_shape, competition_data):\n    valid_shapes = [256, 384, 512, 768]\n    valid_competition_data = [\"2019\", \"2020\", \"2019-2020\"]\n\n    if image_shape not in valid_shapes:\n        raise ValueError(f\"Invalid image shape: {image_shape}. Valid options are {valid_shapes}.\")\n    if competition_data not in valid_competition_data:\n        raise ValueError(f\"Invalid competition data: {competition_data}. Valid options are {valid_competition_data}.\")\n\n# Configuration for crop sizes\nCROP_SIZES = {\n    256: 250,\n    384: 370,\n    512: 500,\n    768: 750\n}\n\n# Resized dimensions based on competition data\nNET_SIZES = {\n    \"2019\": {256: 248, 384: 370, 512: 500, 768: 750},\n    \"2020\": {256: 248, 384: 370, 512: 500, 768: 750},\n    \"2019-2020\": {256: 248, 384: 370, 512: 500, 768: 750}\n}\n\n# Epochs based on competition data\nEPOCHS = {\n    \"2019\": {256: 12, 384: 12, 512: 13, 768: 15},\n    \"2020\": {256: 12, 384: 12, 512: 13, 768: 15},\n    \"2019-2020\": {256: 12, 384: 12, 512: 13, 768: 15}\n}\n\n# Hair augmentation settings\nHAIR_AUGMENTATION = {\n    \"2019\": {256: False, 384: True, 512: False, 768: True},\n    \"2020\": {256: False, 384: True, 512: False, 768: True},\n    \"2019-2020\": {256: False, 384: True, 512: False, 768: True}\n}\n\n# Define device type\nDEVICE = \"TPU\"\n\n# Function to dynamically create configuration\ndef create_config(image_shape, competition_data):\n    validate_inputs(image_shape, competition_data)\n    \n    return {\n        \"batch_size\": 16,\n        \"read_size\": image_shape,\n        \"crop_size\": CROP_SIZES[image_shape],\n        \"net_size\": NET_SIZES[competition_data][image_shape],\n        \"LR_START\": 3e-6,\n        \"LR_MAX\": 2e-5,\n        \"LR_MIN\": 1e-6,\n        \"LR_RAMPUP_EPOCHS\": 5,\n        \"LR_SUSTAIN_EPOCHS\": 0,\n        \"LR_EXP_DECAY\": 0.8,\n        \"epochs\": EPOCHS[competition_data][image_shape],\n        \"rot\": 180.0,\n        \"shr\": 1.5,\n        \"hzoom\": 6.0,\n        \"wzoom\": 6.0,\n        \"hshift\": 6.0,\n        \"wshift\": 6.0,\n        \"DROP_FREQ\": 0,\n        \"DROP_CT\": 0,\n        \"DROP_SIZE\": 0,\n        \"hair_augm\": HAIR_AUGMENTATION[competition_data][image_shape],\n        \"optimizer\": \"adam\",\n        \"label_smooth_fac\": 0.05,\n        \"tta_steps\": 25,\n        \"device\": DEVICE,\n        \"model_weights\": \"imagenet\"\n    }\n\n# Example Usage\nimage_shape = 256\ncompetition_data = \"2020\"\n\n# Generate configuration\nconfig = create_config(image_shape, competition_data)\n\n# Display the configuration\nprint(\"Generated Model Configuration:\")\nfor key, value in config.items():\n    print(f\"{key}: {value}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:41:51.434751Z","iopub.execute_input":"2024-12-07T11:41:51.435229Z","iopub.status.idle":"2024-12-07T11:42:05.301232Z","shell.execute_reply.started":"2024-12-07T11:41:51.435194Z","shell.execute_reply":"2024-12-07T11:42:05.300019Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Setup and Initialization","metadata":{}},{"cell_type":"code","source":"# Import necessary libraries\nimport os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPooling2D\nfrom sklearn.model_selection import train_test_split\n\n# Set random seed for reproducibility\nSEED = 42\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:42:05.302309Z","iopub.execute_input":"2024-12-07T11:42:05.302907Z","iopub.status.idle":"2024-12-07T11:42:06.735473Z","shell.execute_reply.started":"2024-12-07T11:42:05.302872Z","shell.execute_reply":"2024-12-07T11:42:06.734257Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Loading and Preprocessing","metadata":{}},{"cell_type":"code","source":"# Load dataset (example assumes CSV metadata)\ndef load_dataset(csv_path, image_dir):\n    data = pd.read_csv(csv_path)\n    data['image_path'] = data['image_name'].apply(lambda x: os.path.join(image_dir, f\"{x}.jpg\"))\n    return data\n\n# Image preprocessing function\ndef preprocess_image(image_path, target_size):\n    image = tf.io.read_file(image_path)\n    image = tf.image.decode_jpeg(image, channels=3)\n    image = tf.image.resize(image, target_size)\n    image = tf.cast(image, tf.float32) / 255.0  # Normalize to [0, 1]\n    return image\n\n# Create TensorFlow dataset\ndef create_tf_dataset(dataframe, target_size, batch_size, shuffle=True):\n    image_paths = dataframe['image_path'].values\n    labels = dataframe['target'].values\n    dataset = tf.data.Dataset.from_tensor_slices((image_paths, labels))\n    dataset = dataset.map(lambda x, y: (preprocess_image(x, target_size), tf.cast(y, tf.float32)))\n    if shuffle:\n        dataset = dataset.shuffle(buffer_size=1024)\n    dataset = dataset.batch(batch_size).prefetch(buffer_size=tf.data.AUTOTUNE)\n    return dataset\n\n# Example usage\ncsv_path = \"../input/siim-isic-melanoma-classification/train.csv\"\nimage_dir = \"../input/siim-isic-melanoma-classification/jpeg/train/\"\ndata = load_dataset(csv_path, image_dir)\n\n# Split dataset\ntrain_data, val_data = train_test_split(data, test_size=0.2, random_state=SEED)\n\n# Generate TensorFlow datasets\ntarget_size = (256, 256)  # Replace with `config['read_size']` if dynamic\nbatch_size = 16\ntrain_dataset = create_tf_dataset(train_data, target_size, batch_size)\nval_dataset = create_tf_dataset(val_data, target_size, batch_size, shuffle=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:42:06.738120Z","iopub.execute_input":"2024-12-07T11:42:06.739322Z","iopub.status.idle":"2024-12-07T11:42:07.114565Z","shell.execute_reply.started":"2024-12-07T11:42:06.739254Z","shell.execute_reply":"2024-12-07T11:42:07.113304Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Architecture","metadata":{}},{"cell_type":"code","source":"# Define a simple CNN model\ndef build_model(input_shape, num_classes=1):\n    model = Sequential([\n        Conv2D(32, kernel_size=(3, 3), activation='relu', input_shape=input_shape),\n        MaxPooling2D(pool_size=(2, 2)),\n        Dropout(0.25),\n        Conv2D(64, kernel_size=(3, 3), activation='relu'),\n        MaxPooling2D(pool_size=(2, 2)),\n        Dropout(0.25),\n        Flatten(),\n        Dense(128, activation='relu'),\n        Dropout(0.5),\n        Dense(num_classes, activation='sigmoid')\n    ])\n    return model\n\n# Initialize model\ninput_shape = (256, 256, 3)  # Replace with `config['read_size']` dynamically if needed\nmodel = build_model(input_shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:42:07.115971Z","iopub.execute_input":"2024-12-07T11:42:07.116600Z","iopub.status.idle":"2024-12-07T11:42:07.528059Z","shell.execute_reply.started":"2024-12-07T11:42:07.116544Z","shell.execute_reply":"2024-12-07T11:42:07.526810Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Compile the Model","metadata":{}},{"cell_type":"code","source":"# Compile model with optimizer and loss\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=3e-4),  # Replace with `config['optimizer']`\n    loss='binary_crossentropy',\n    metrics=['accuracy']\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:42:07.529434Z","iopub.execute_input":"2024-12-07T11:42:07.529758Z","iopub.status.idle":"2024-12-07T11:42:07.544989Z","shell.execute_reply.started":"2024-12-07T11:42:07.529718Z","shell.execute_reply":"2024-12-07T11:42:07.543896Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Define Callbacks","metadata":{}},{"cell_type":"code","source":"# Define callbacks for training\ncheckpoint_path = \"model_checkpoint.weights.h5\"  # Updated file extension\ncallbacks = [\n    tf.keras.callbacks.ModelCheckpoint(\n        filepath=checkpoint_path,\n        monitor='val_loss',\n        save_best_only=True,\n        save_weights_only=True,\n        verbose=1\n    ),\n    tf.keras.callbacks.ReduceLROnPlateau(\n        monitor='val_loss',\n        factor=0.1,\n        patience=3,\n        verbose=1\n    ),\n    tf.keras.callbacks.EarlyStopping(\n        monitor='val_loss',\n        patience=5,\n        restore_best_weights=True,\n        verbose=1\n    )\n]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:42:07.546483Z","iopub.execute_input":"2024-12-07T11:42:07.547044Z","iopub.status.idle":"2024-12-07T11:42:07.554095Z","shell.execute_reply.started":"2024-12-07T11:42:07.546998Z","shell.execute_reply":"2024-12-07T11:42:07.552934Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train the Model","metadata":{}},{"cell_type":"code","source":"# Train the model\nhistory = model.fit(\n    train_dataset,\n    validation_data=val_dataset,\n    epochs=15, \n    callbacks=callbacks\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T11:42:07.555547Z","iopub.execute_input":"2024-12-07T11:42:07.555979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}}]}