{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"},{"sourceId":8061946,"sourceType":"datasetVersion","datasetId":4755693},{"sourceId":8131867,"sourceType":"datasetVersion","datasetId":4753915}],"dockerImageVersionId":30673,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport json\nimport gc\nimport cv2\nimport joblib\nimport sys\nimport random\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torchvision.models as models\nfrom torch.utils.data import DataLoader, Dataset\nimport tensorflow as tf\nimport tensorflow_hub as hub\nfrom tensorflow import keras\nfrom tensorflow.keras.preprocessing.image import img_to_array, load_img\nfrom sklearn.metrics import accuracy_score\nfrom tqdm import tqdm\nfrom PIL import Image\nfrom albumentations import (Compose, OneOf, Normalize, Resize, RandomResizedCrop, RandomCrop, CenterCrop, \n                            HorizontalFlip, VerticalFlip, Rotate, ShiftScaleRotate, Transpose, ImageOnlyTransform)\nfrom albumentations.pytorch import ToTensorV2\nimport timm\nimport shutil\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-16T03:24:33.798594Z","iopub.execute_input":"2024-04-16T03:24:33.799924Z","iopub.status.idle":"2024-04-16T03:24:33.809426Z","shell.execute_reply.started":"2024-04-16T03:24:33.799872Z","shell.execute_reply":"2024-04-16T03:24:33.808282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission for Crop Net","metadata":{}},{"cell_type":"code","source":"path = \"/kaggle/input/cassava-leaf-disease-classification/\"\nprint(f\"Contents of the data folder: {os.listdir(path)}\")\nimage_path = path+\"test_images/\"\nimage_folder = os.path.join(path, 'test_images')\n\n\nIMAGE_SIZE = (512,512)\nsubmission_df = pd.DataFrame(columns=[\"image_id\", \"label\"])\nsubmission_df[\"image_id\"] = os.listdir(image_path)\nsubmission_df[\"label\"] = 0","metadata":{"execution":{"iopub.status.busy":"2024-04-16T03:24:33.811268Z","iopub.execute_input":"2024-04-16T03:24:33.811673Z","iopub.status.idle":"2024-04-16T03:24:33.825080Z","shell.execute_reply.started":"2024-04-16T03:24:33.811641Z","shell.execute_reply":"2024-04-16T03:24:33.823998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_hub as hub\n\ndef load_model_cropnet():\n    model_path = '/kaggle/input/cropnet-model/cropnet_mobilenetv3/cropnet_mobilenetv3'\n    # Load the model from the local directory\n    classifier = hub.KerasLayer(model_path, trainable=False)\n    return classifier","metadata":{"execution":{"iopub.status.busy":"2024-04-16T03:24:33.826652Z","iopub.execute_input":"2024-04-16T03:24:33.827357Z","iopub.status.idle":"2024-04-16T03:24:33.832755Z","shell.execute_reply.started":"2024-04-16T03:24:33.827317Z","shell.execute_reply":"2024-04-16T03:24:33.831829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def center_crop_and_resize(image, target_size=(224, 224)):\n    \"\"\"Crop the center of the image and resize it to the target size.\"\"\"\n    # Assuming image is a NumPy array of shape (height, width, channels)\n    original_height, original_width, _ = image.shape\n    target_height, target_width = target_size\n    \n    # Calculate the coordinates for the center crop\n    crop_height_start = (original_height - min(original_height, original_width)) // 2\n    crop_width_start = (original_width - min(original_height, original_width)) // 2\n    crop_height_end = crop_height_start + min(original_height, original_width)\n    crop_width_end = crop_width_start + min(original_height, original_width)\n    \n    # Perform the center crop\n    image_cropped = image[crop_height_start:crop_height_end, crop_width_start:crop_width_end]\n    \n    # Resize the cropped image to the target size\n    image_resized = tf.image.resize(image_cropped, target_size)\n    return image_resized.numpy()\n\ndef preprocess_image_cropnet(image_path, image_size):\n    \"\"\"\n    Load the image from the given path and resize it to the target size.\n    \"\"\"\n    image = load_img(image_path) # Load and resize to expected size\n    image = img_to_array(image) / 255.0 # Convert to array and normalize\n    image = center_crop_and_resize(image,image_size)\n    image = np.expand_dims(image, axis=0)\n    return image\n\ndef remap_probabilities(probabilities):\n    # Convert probabilities to a numpy array if it's a TensorFlow tensor\n    if isinstance(probabilities, tf.Tensor):\n        probabilities = probabilities.numpy()\n    num_known_classes = 5\n    # Check if the probabilities include the sixth 'Unknown' class. This comes with cropnet\n    if probabilities.shape[1] > num_known_classes:\n        # Extract the probability of the 'Unknown' class\n        unknown_prob = probabilities[:, -1] / num_known_classes\n        # Remove the 'Unknown' class probability\n        known_class_probs = probabilities[:, :num_known_classes]\n        # Evenly distribute the 'Unknown' class probability among the known classes\n        adjusted_probs = known_class_probs + unknown_prob[:, None]\n    else:\n        adjusted_probs = probabilities\n\n    return adjusted_probs\n","metadata":{"execution":{"iopub.status.busy":"2024-04-16T03:24:33.835556Z","iopub.execute_input":"2024-04-16T03:24:33.835893Z","iopub.status.idle":"2024-04-16T03:24:33.849376Z","shell.execute_reply.started":"2024-04-16T03:24:33.835864Z","shell.execute_reply":"2024-04-16T03:24:33.848450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_image_cropnet(model, image_name, image_folder):\n    image_path_full = os.path.join(image_folder, image_name)\n    image = preprocess_image_cropnet(image_path_full, (224,224))\n    probabilities = model(image)\n    probabilities = remap_probabilities(probabilities)\n    #         predict_label = tf.argmax(probabilities, axis=-1).numpy()[0]\n    return probabilities","metadata":{"execution":{"iopub.status.busy":"2024-04-16T03:24:33.850704Z","iopub.execute_input":"2024-04-16T03:24:33.851666Z","iopub.status.idle":"2024-04-16T03:24:33.860647Z","shell.execute_reply.started":"2024-04-16T03:24:33.851626Z","shell.execute_reply":"2024-04-16T03:24:33.859810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EffecientNet","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\n\ndef extract_patches(image_path, patch_size=(224, 224)):\n    # Load the image from path\n    img = load_img(image_path)\n    img = img_to_array(img)  # Convert it to array\n    height, width, _ = img.shape\n\n    # Example of extracting a center patch\n    center_x, center_y = height // 2, width // 2\n    center_patch = img[center_x - patch_size[0]//2:center_x + patch_size[0]//2, \n                       center_y - patch_size[1]//2:center_y + patch_size[1]//2, :]\n\n    # Resize patch to the target size if necessary\n    center_patch = Image.fromarray(np.uint8(center_patch))  # Convert array to PIL Image\n    center_patch = center_patch.resize(patch_size, Image.ANTIALIAS)  # Resize image\n    center_patch = img_to_array(center_patch)  # Convert back to array if necessary for further processing\n\n    # Assuming you are handling multiple patches, append to a list\n    patches = [center_patch]  # Start with the center patch as an example\n\n    return np.array(patches)\n\ndef augment_patches(patches):\n    \"\"\"Apply augmentations to each patch.\"\"\"\n    # Assuming patches is a batch of images of shape (N, H, W, C)\n    flips = np.flip(patches, axis=2)  # Horizontal flip\n    rotates = np.rot90(patches, k=1, axes=(1, 2))  # 90 degree rotation\n    transposes = np.transpose(patches, (0, 2, 1, 3))  # Transpose\n    \n    return np.concatenate([flips, rotates, transposes], axis=0)\n\ndef preprocess_input_efficientnet(patches):\n    \"\"\"Normalize patches to the range expected by EfficientNet.\"\"\"\n    return patches / 255.0  # Adjust if your model expects a different range\n","metadata":{"execution":{"iopub.status.busy":"2024-04-16T03:24:33.861730Z","iopub.execute_input":"2024-04-16T03:24:33.862394Z","iopub.status.idle":"2024-04-16T03:24:33.874491Z","shell.execute_reply.started":"2024-04-16T03:24:33.862317Z","shell.execute_reply":"2024-04-16T03:24:33.873461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_model_efficientnet():\n    # this is done to bypass some read only OS issues when loading model in kaggle \n    source_path = '/kaggle/input/cropnet-model/EffecientNet.keras'\n    target_path = '/kaggle/working/EffecientNet.keras'\n    shutil.copy(source_path, target_path)\n    efficient_net = tf.keras.models.load_model(target_path)\n    return efficient_net\n\n\ndef predict_image_efficientnet(model, image_name, image_folder):\n    image_path = os.path.join(image_folder, image_name)\n    patches = extract_patches(image_path)\n    patches_augmented = augment_patches(patches[:4])  # Apply augmentations to the first 4 patches\n    patches_combined = np.concatenate([patches, patches_augmented], axis=0)  # Combine all patches\n    patches_preprocessed = preprocess_input_efficientnet(patches_combined)\n    predictions = model.predict(patches_preprocessed, verbose=0)\n    overall_prediction = np.mean(predictions, axis=0)\n    return overall_prediction\n\n# image_folder = os.path.join(path, 'train_images')\n# model = load_model_efficientnet()\n# predict_image_efficientnet (model, '1001320321.jpg', image_folder)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T03:24:33.955938Z","iopub.execute_input":"2024-04-16T03:24:33.956399Z","iopub.status.idle":"2024-04-16T03:24:33.965630Z","shell.execute_reply.started":"2024-04-16T03:24:33.956358Z","shell.execute_reply":"2024-04-16T03:24:33.964601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# VGG","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing import image\nfrom keras.applications.vgg16 import preprocess_input\n\ndef preprocess_image_vgg16(img_path):\n    \"\"\"Load and preprocess an image for VGG16 model.\"\"\"\n    img = image.load_img(img_path, target_size=(300, 300))\n    \n    # Convert the image to a numpy array\n    img_array = image.img_to_array(img)\n    \n    # Add a dimension to the image to account for the batch size\n    img_array_expanded = np.expand_dims(img_array, axis=0)\n    \n    # Use the preprocess_input function from keras.applications.vgg16\n    img_preprocessed = preprocess_input(img_array_expanded)\n    \n    return img_preprocessed\n","metadata":{"execution":{"iopub.status.busy":"2024-04-16T03:24:33.968156Z","iopub.execute_input":"2024-04-16T03:24:33.968564Z","iopub.status.idle":"2024-04-16T03:24:33.977722Z","shell.execute_reply.started":"2024-04-16T03:24:33.968530Z","shell.execute_reply":"2024-04-16T03:24:33.976796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n\ndef load_model_vgg():\n    # this is done to bypass some read only OS issues when loading model in kaggle \n    source_path = '/kaggle/input/cropnet-model/model_vgg16_v2.keras'\n    target_path = '/kaggle/working/model_vgg16_v2.keras'\n    shutil.copy(source_path, target_path)\n    vgg_model = tf.keras.models.load_model(target_path)\n    return vgg_model","metadata":{"execution":{"iopub.status.busy":"2024-04-16T03:24:33.979336Z","iopub.execute_input":"2024-04-16T03:24:33.979644Z","iopub.status.idle":"2024-04-16T03:24:33.986828Z","shell.execute_reply.started":"2024-04-16T03:24:33.979619Z","shell.execute_reply":"2024-04-16T03:24:33.985929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_image_vgg(model, image_name, image_folder):\n    image_path_full = os.path.join(image_folder, image_name)\n    image = preprocess_image_vgg16(image_path_full)\n    prediction = model.predict(image, verbose=0)\n    return prediction","metadata":{"execution":{"iopub.status.busy":"2024-04-16T03:24:33.988150Z","iopub.execute_input":"2024-04-16T03:24:33.990048Z","iopub.status.idle":"2024-04-16T03:24:33.997689Z","shell.execute_reply.started":"2024-04-16T03:24:33.990021Z","shell.execute_reply":"2024-04-16T03:24:33.996651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ViT","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torchvision.models as models\n\ndef load_model_vit():\n    model_path = '/kaggle/input/cropnet-model/model_vit_8179acc.pth'\n    pretrain_model_path = '/kaggle/input/cropnet-model/vit_b_16-c867db91.pth'\n\n    # Initialize the ViT model with default pre-trained weights\n    model = models.vit_b_16(weights=None)\n    state_dict = torch.load(pretrain_model_path)\n    model.load_state_dict(state_dict, strict=False)  # Use strict=False if the keys do not exactly match\n\n    # Freeze all the parameters in the model\n    for param in model.parameters():\n        param.requires_grad = False\n    \n    # Replace the classifier head\n    model.heads = nn.Sequential(\n        nn.Linear(in_features=768, out_features=5, bias=True)\n    )\n    \n    # Load the custom weights from a saved state\n    model.load_state_dict(torch.load(model_path))\n    \n    # Switch the model to evaluation mode\n    model.eval()\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2024-04-16T03:24:34.001121Z","iopub.execute_input":"2024-04-16T03:24:34.001476Z","iopub.status.idle":"2024-04-16T03:24:34.008968Z","shell.execute_reply.started":"2024-04-16T03:24:34.001450Z","shell.execute_reply":"2024-04-16T03:24:34.007850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torchvision import transforms\nfrom PIL import Image\n\ndef preprocess_image_vit(image_path):\n    # Define the standard ViT preprocessing\n    transform = transforms.Compose([\n        transforms.Resize((224, 224)),  # Resize to the input size expected by ViT\n        transforms.ToTensor(),  # Convert the image to a tensor\n        transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])  # Normalize\n    ])\n    \n    # Load the image\n    image = Image.open(image_path).convert('RGB')\n    image_tensor = transform(image).unsqueeze(0)  # Add batch dimension\n    return image_tensor\n\ndef predict_image_vit(model, image_name, image_folder):\n    image_path = os.path.join(image_folder, image_name)\n    image_tensor = preprocess_image_vit(image_path)\n    # Perform prediction\n    with torch.no_grad():\n        outputs = model(image_tensor)\n    \n    probabilities = torch.nn.functional.softmax(outputs, dim=1)\n#     predicted_class = probabilities.argmax(dim=1).item()\n    probabilities = probabilities.cpu().detach().numpy()  # Detach from the graph, move to CPU and convert to numpy\n\n    return probabilities\n","metadata":{"execution":{"iopub.status.busy":"2024-04-16T03:24:34.010553Z","iopub.execute_input":"2024-04-16T03:24:34.011153Z","iopub.status.idle":"2024-04-16T03:24:34.020789Z","shell.execute_reply.started":"2024-04-16T03:24:34.011119Z","shell.execute_reply":"2024-04-16T03:24:34.019824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ensemble Learning","metadata":{}},{"cell_type":"code","source":"from tqdm import tqdm\n\ndef get_predictions_vgg(model, images, image_folder):\n    predictions = []\n    # Wrap the loop with tqdm for a progress bar\n    for image in tqdm(images, desc=\"Predicting with VGG\"):\n        predictions.append(predict_image_vgg(model, image, image_folder))\n    return predictions\n\ndef get_predictions_cropnet(model, images, image_folder, n_jobs=-1):\n    predictions = []\n    # Wrap the loop with tqdm for a progress bar\n    for image in tqdm(images, desc=\"Predicting with CropNet\"):\n        predictions.append(predict_image_cropnet(model, image, image_folder))\n    return predictions\n\ndef get_predictions_vit(model, images, image_folder, n_jobs=-1):\n    predictions = []\n    # Wrap the loop with tqdm for a progress bar\n    for image in tqdm(images, desc=\"Predicting with ViT\"):\n        predictions.append(predict_image_vit(model, image, image_folder))\n    return predictions\n\ndef get_predictions_efficient(model, images, image_folder, n_jobs=-1):\n    predictions = []\n    # Wrap the loop with tqdm for a progress bar\n    for image in tqdm(images, desc=\"Predicting with EfficientNet\"):\n        predictions.append(predict_image_efficientnet(model, image, image_folder))\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2024-04-16T03:24:34.022186Z","iopub.execute_input":"2024-04-16T03:24:34.022857Z","iopub.status.idle":"2024-04-16T03:24:34.033154Z","shell.execute_reply.started":"2024-04-16T03:24:34.022822Z","shell.execute_reply":"2024-04-16T03:24:34.032296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# First collect all the results\n\npath = \"/kaggle/input/cassava-leaf-disease-classification/\"\nimage_folder = os.path.join(path, 'train_images')\n\nIMAGE_SIZE = (512,512)\ntrain_df = pd.read_csv(path+\"train.csv\")\ntrain_df = train_df.head(5)\n\nimages = train_df['image_id']\n\ncropnet_model = load_model_cropnet()\nvgg_model = load_model_vgg()\nvit_model = load_model_vit()\neff_model = load_model_efficientnet()\n\n\ntrain_df['Cropnet'] = get_predictions_cropnet(cropnet_model, images, image_folder)\ntrain_df['VGG'] = get_predictions_vgg(vgg_model, images, image_folder)\ntrain_df['ViT'] = get_predictions_vit(vit_model, images, image_folder)\ntrain_df['EfficientNet'] = get_predictions_efficient(eff_model, images, image_folder)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T03:24:34.034426Z","iopub.execute_input":"2024-04-16T03:24:34.035013Z","iopub.status.idle":"2024-04-16T03:24:36.508747Z","shell.execute_reply.started":"2024-04-16T03:24:34.034979Z","shell.execute_reply":"2024-04-16T03:24:36.506721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-16T03:24:36.509982Z","iopub.status.idle":"2024-04-16T03:24:36.510512Z","shell.execute_reply.started":"2024-04-16T03:24:36.510246Z","shell.execute_reply":"2024-04-16T03:24:36.510267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Meta-Model","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold, KFold\nimport pandas as pd\n\ndata = pd.read_csv('/kaggle/input/cassava-leaf-disease-classification/train.csv')\n\n# Number of splits\nn_splits = 5 \n\n\n# Assuming 'target' is your class label column\nskf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n\n# Add fold numbers\nfor fold, (train_idx, val_idx) in enumerate(skf.split(X=data, y=data['label'])):\n    data.loc[val_idx, 'fold'] = fold","metadata":{"execution":{"iopub.status.busy":"2024-04-16T03:34:00.039793Z","iopub.execute_input":"2024-04-16T03:34:00.040218Z","iopub.status.idle":"2024-04-16T03:34:00.079533Z","shell.execute_reply.started":"2024-04-16T03:34:00.040186Z","shell.execute_reply":"2024-04-16T03:34:00.078344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\npath = \"/kaggle/input/cassava-leaf-disease-classification/\"\nimage_folder = os.path.join(path, 'train_images')\nIMAGE_SIZE = (512,512)\ncropnet_model = load_model_cropnet()\nvgg_model = load_model_vgg()\nvit_model = load_model_vit()\neff_model = load_model_efficientnet()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-16T03:24:36.514487Z","iopub.status.idle":"2024-04-16T03:24:36.514989Z","shell.execute_reply.started":"2024-04-16T03:24:36.514724Z","shell.execute_reply":"2024-04-16T03:24:36.514743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm  # Import tqdm\n\ndef run_fold_predictions(data, fold_number, vit_model, cropnet_model, vgg_model, eff_model, image_folder):\n    # Split data based on the fold\n    train_data = data[data['fold'] != fold_number]\n    val_data = data[data['fold'] == fold_number]\n\n    # Retrieve image names from validation set\n    images = val_data['image_id'].tolist()\n\n    vit_predictions = get_predictions_vit(vit_model, images, image_folder)\n    cropnet_predictions = get_predictions_cropnet(cropnet_model, images, image_folder)\n    vgg_predictions = get_predictions_vgg(vgg_model, images, image_folder)\n    eff_predictions = get_predictions_efficient(eff_model, images, image_folder)\n\n    # Combine predictions into a dataframe\n    predictions_df = pd.DataFrame({\n        'image_id': images,\n        'fold': [fold_number] * len(images),  # Include fold number in the dataframe\n        'vit_predictions': vit_predictions,\n        'cropnet_predictions': cropnet_predictions,\n        'vgg_predictions': vgg_predictions,\n        'eff_predictions': eff_predictions\n    })\n    return predictions_df\n\nK = 5\nall_predictions = pd.DataFrame()\n\nfor fold in range(K):\n    print(f\"{fold+1}/{K}\")\n    fold_predictions = run_fold_predictions(data, fold, vit_model, cropnet_model, vgg_model, eff_model, image_folder)\n    all_predictions = pd.concat([all_predictions, fold_predictions], ignore_index=True)\n\nall_predictions.to_csv('/kaggle/working/all_predictions.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-16T03:24:36.516032Z","iopub.status.idle":"2024-04-16T03:24:36.516369Z","shell.execute_reply.started":"2024-04-16T03:24:36.516200Z","shell.execute_reply":"2024-04-16T03:24:36.516214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nimport ast \nimport re \n\n# Load the combined predictions and actual labels if they're not already included\nall_predictions = pd.read_csv('/kaggle/input/cropnet-model/all_predictions.csv')\n# data = pd.read_csv('/kaggle/input/cropnet-model/train-fold.csv') \n\n# Ensure the predictions and original data are aligned\nall_predictions = all_predictions.merge(data[['image_id', 'label']], on='image_id')\nall_predictions = all_predictions.drop(\"label_y\", axis=1)\nall_predictions = all_predictions.rename(columns={'label_x': 'label'})\n\ndef string_to_list(string):\n    #replace spaces between numbers with ,\n    formatted_str = re.sub(r'(?<=\\d)\\s+(?=\\d)', ', ', string)\n    return ast.literal_eval(formatted_str)\n\n# # Predictions are stored as strings of lists; convert them to actual lists\n# import ast\nall_predictions['vit_predictions'] = all_predictions['vit_predictions'].apply(string_to_list)\nall_predictions['cropnet_predictions'] = all_predictions['cropnet_predictions'].apply(string_to_list)\nall_predictions['vgg_predictions'] = all_predictions['vgg_predictions'].apply(string_to_list)\nall_predictions['eff_predictions'] = all_predictions['eff_predictions'].apply(string_to_list)\n\n# # # Prepare features for the meta-model by combining predictions\n# # Here we simply concatenate the prediction lists\nX = np.vstack(all_predictions.apply(lambda row: np.concatenate([\n    row['vit_predictions'][0],\n    row['cropnet_predictions'][0],\n    row['vgg_predictions'][0],\n    row['eff_predictions'][0]\n]), axis=1))\ny = all_predictions['label'].values  # Labels\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\nmeta_model = LogisticRegression(max_iter=1000)\nmeta_model.fit(X_train, y_train)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-16T04:01:04.448277Z","iopub.execute_input":"2024-04-16T04:01:04.448946Z","iopub.status.idle":"2024-04-16T04:01:09.676548Z","shell.execute_reply.started":"2024-04-16T04:01:04.448913Z","shell.execute_reply":"2024-04-16T04:01:09.674824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, classification_report\n\ny_pred = meta_model.predict(X_test)\nprint(\"Accuracy:\", accuracy_score(y_test, y_pred))\nprint(\"Classification Report:\\n\", classification_report(y_test, y_pred))\n","metadata":{"execution":{"iopub.status.busy":"2024-04-16T04:01:25.703340Z","iopub.execute_input":"2024-04-16T04:01:25.703701Z","iopub.status.idle":"2024-04-16T04:01:25.743979Z","shell.execute_reply.started":"2024-04-16T04:01:25.703672Z","shell.execute_reply":"2024-04-16T04:01:25.742936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\n# save the model to disk\nfilename = 'logistic_regressor.sav'\npickle.dump(meta_model, open(filename, 'wb')) ","metadata":{"execution":{"iopub.status.busy":"2024-04-16T04:24:38.441076Z","iopub.execute_input":"2024-04-16T04:24:38.441464Z","iopub.status.idle":"2024-04-16T04:24:38.446958Z","shell.execute_reply.started":"2024-04-16T04:24:38.441435Z","shell.execute_reply":"2024-04-16T04:24:38.445843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Running the model","metadata":{}},{"cell_type":"code","source":"loaded_model = pickle.load(open(filename, 'rb'))","metadata":{"execution":{"iopub.status.busy":"2024-04-16T04:25:01.845137Z","iopub.execute_input":"2024-04-16T04:25:01.845541Z","iopub.status.idle":"2024-04-16T04:25:01.850638Z","shell.execute_reply.started":"2024-04-16T04:25:01.845511Z","shell.execute_reply":"2024-04-16T04:25:01.849661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loaded_model.predict()","metadata":{},"execution_count":null,"outputs":[]}]}