{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":29653,"databundleVersionId":2420395,"sourceType":"competition"},{"sourceId":9404481,"sourceType":"datasetVersion","datasetId":3380728}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# <a href=\"https://www.kaggle.com/code/yannicksteph/rsna-miccai-brain-tumor-classification\" target=\"_blank\"><img align=\"left\" alt=\"Kaggle\" title=\"Open in Kaggle\" src=\"https://kaggle.com/static/images/open-in-kaggle.svg\"></a>","metadata":{"_uuid":"db314319-d6c3-4adb-8c2e-45f4523acceb","_cell_guid":"41cf6070-bb6e-4ef4-93d1-bb3ca5e070b0","_kg_hide-input":false,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2024-10-03T05:04:11.388427Z","iopub.execute_input":"2024-10-03T05:04:11.389168Z","iopub.status.idle":"2024-10-03T05:04:11.393294Z","shell.execute_reply.started":"2024-10-03T05:04:11.389128Z","shell.execute_reply":"2024-10-03T05:04:11.392583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Operating System and File System\nimport os\n\n# Scikit-learn\nfrom sklearn.feature_selection import RFECV  # Recursive Feature Elimination with Cross-Validation\nfrom sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor, AdaBoostRegressor\nfrom sklearn.linear_model import LinearRegression, Lasso, ElasticNet, Ridge\nfrom sklearn.svm import SVR\nfrom sklearn.datasets import make_regression  # Generate synthetic regression datasets\nfrom sklearn.decomposition import PCA  # Principal Component Analysis\nfrom sklearn.model_selection import KFold  # K-Fold cross-validation\n\n# XGBoost and LightGBM\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\n\n# Basic\nimport math  # Mathematical functions\nfrom enum import Enum  # Enumeration (enum) types\nfrom itertools import chain, permutations, combinations  # Iteration tools\n\n# Data Manipulation and Analysis\nimport numpy as np\nimport pandas as pd\n\n# Data Visualization\nimport matplotlib.pyplot as plt\nimport matplotlib.animation as anim  # Animation in Matplotlib\nimport seaborn as sns  # Statistical data visualization\nfrom mpl_toolkits.mplot3d import Axes3D  # 3D plotting\n\n# Warnings\nimport warnings  # For suppressing warnings\n\n# JSON Handling\nimport json  # For working with JSON data\n\n# Encoding and Decoding Binary Data\nimport base64  # For encoding and decoding binary data\n\n# Interactive Widgets and Display\nimport ipywidgets as widgets  # For creating interactive widgets in Jupyter Notebook\nfrom IPython.display import HTML, display  # For displaying HTML content\nfrom IPython.display import Image as show_gif  # GIF display\n\n# Deep Learning Framework\nimport torch  # For working with PyTorch deep learning framework\n\n# DICOM File Handling\nimport pydicom  # For reading DICOM files\nfrom pydicom import dcmread  # For reading DICOM files\n\n# Image Processing and Filtering\nimport SimpleITK as sitk  # For image filtering\nfrom PIL import Image  # For image processing using the Python Imaging Library (PIL)\n\n# Machine Learning and Data Splitting\nfrom sklearn.model_selection import train_test_split  # For splitting data into training and testing sets\nfrom sklearn.preprocessing import RobustScaler, StandardScaler  # Data scaling for machine learning models\n\n# scipy\nimport scipy.cluster.hierarchy as sch\nfrom scipy.cluster.hierarchy import fcluster, linkage  # Hierarchical clustering\nfrom scipy import stats\n\n# statsmodels\nfrom statsmodels.stats.diagnostic import lilliefors\nimport statsmodels.api as sm\n\n# Additional Libraries\nimport glob  # File pattern matching\nimport random  # Random number generation\n\n!pip install pyradiomics > /dev/null  # Installing the pyradiomics library for radiomics feature extraction\nimport radiomics  # For extracting radiomics features from medical images","metadata":{"_uuid":"7ee05952-0cc5-464f-ad60-40778f96252e","_cell_guid":"6fe6a9d6-9a01-4516-bd63-de3387ea5362","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-10-03T05:04:11.397059Z","iopub.execute_input":"2024-10-03T05:04:11.39731Z","iopub.status.idle":"2024-10-03T05:04:59.218472Z","shell.execute_reply.started":"2024-10-03T05:04:11.397282Z","shell.execute_reply":"2024-10-03T05:04:59.215518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Paths","metadata":{}},{"cell_type":"code","source":"# ------------------------------------#    \n# Definition Version\n# ------------------------------------#    \n\nclass DatasetVersion(Enum):\n    V_2D = '2D'\n    V_3D = '3D'\n    \n# ------------------------------------#  \n# Paths\n# ------------------------------------#  \n    \n# ------------#\n# Folders\n# ------------# \nbase_folder_path = \"../input/rsna-miccai-brain-tumor-radiogenomic-classification/\"\ntrain_folder_path = f\"{base_folder_path}train/\"\ndataset_folder_path = \"../input/githut-rsna-miccai-brain-tumor-classification-ai/dataset/\"\n\n# ------------#\n# Train / files\n# ------------# \ntrain_labels_path = f\"{base_folder_path}train_labels.csv\"\nsubmission_labels_path = f\"{base_folder_path}sample_submission.csv\"\n\ndataset_3d_path = f\"{dataset_folder_path}3d_rsna_miccai_brain_tumor_brain_segmentation_pytorch_unet.csv\"\ndataset_2d_path = f\"{dataset_folder_path}2d_rsna_miccai_brain_tumor_brain_segmentation_pytorch_unet.csv\"","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:04:59.221235Z","iopub.execute_input":"2024-10-03T05:04:59.222512Z","iopub.status.idle":"2024-10-03T05:04:59.230318Z","shell.execute_reply.started":"2024-10-03T05:04:59.222463Z","shell.execute_reply":"2024-10-03T05:04:59.22926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ------------------------------------#   \n# Configuration\n# ------------------------------------#    \n    \n# ℹ️ Flag to skip brain segmentation with PyTorch UNet\n# If set to True, we will import the dataset that has already been generated\nskip_brain_segmentation_pytorch_unet = True\n\n# ℹ️ Set the dataset version to use when <skip_brain_segmentation_pytorch_unet = True>\ndataset_version =  DatasetVersion.V_3D\n\n#Color\ndefault_palette_color = \"deep\"","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:04:59.23182Z","iopub.execute_input":"2024-10-03T05:04:59.232194Z","iopub.status.idle":"2024-10-03T05:04:59.247326Z","shell.execute_reply.started":"2024-10-03T05:04:59.23215Z","shell.execute_reply":"2024-10-03T05:04:59.246463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ------------------------------------#\n# Configuration options\n# ------------------------------------#\n\n# Pandas configuration\npd.set_option('display.max_columns', None)\npd.set_option('display.expand_frame_repr', False)\npd.set_option('max_colwidth', None)\npd.set_option('display.max_rows', None)\n\n# Suppressing Warnings\nwarnings.filterwarnings('ignore')\n\n# Define \"reader\"\n# Read serie of image files into a SimpleTK image\nsitk_reader = sitk.ImageSeriesReader()\nsitk_reader.LoadPrivateTagsOn()\n\n# Suppressing warnings SimpleITK\nsitk.ProcessObject.SetGlobalWarningDisplay(False)\n\n# Default color\nsns.set_theme(palette=default_palette_color)\n\n# ------------\n# Segmentation\n# ------------\n\n# Load mateuszbuda/brain-segmentation-pytorch, U-Net with batch normalization for biomedical image segmentation with pretrained weights for abnormality segmentation in brain MRI\nsegmentation_model = torch.hub.load('mateuszbuda/brain-segmentation-pytorch', \n                                    'unet', \n                                    in_channels=3, \n                                    out_channels=1, \n                                    init_features=32, \n                                    pretrained=True, \n                                    trust_repo=True)\n\n# ------------\n# Dataset\n# ------------\n# Dataset of the project, explanation in next section.    \npreview_dataset_df = pd.read_csv(train_labels_path)\nsamp_subm = pd.read_csv(submission_labels_path)","metadata":{"_uuid":"7caa653c-b826-4e99-852e-c22b28757274","_cell_guid":"ddf3ae13-9bc9-47e5-a6be-7c343074b818","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-10-03T05:04:59.250311Z","iopub.execute_input":"2024-10-03T05:04:59.251005Z","iopub.status.idle":"2024-10-03T05:05:03.493897Z","shell.execute_reply.started":"2024-10-03T05:04:59.25092Z","shell.execute_reply":"2024-10-03T05:05:03.493008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" # ===============================================\n# Base methods\n# ===============================================\n\ndef is_even(nombre):\n    return nombre % 2 == 0\n\ndef map_dataframe_to_tuples(df, columns):\n    mapped_tuples = []\n    for row in df.itertuples(index=False):\n        mapped_tuple = tuple(getattr(row, col) for col in columns)\n        mapped_tuples.append(mapped_tuple)\n    return mapped_tuples\n\ndef extract_number_from_filename(filename):\n    # Extrayez les chiffres à la fin du nom de fichier\n    number_str = ''.join(filter(str.isdigit, filename))\n    if number_str:\n        return int(number_str)\n    else:\n        return -1  # Si aucun chiffre n'est trouvé, renvoyez une valeur par défaut\n    import os\n    \ndef remove_values(values_to_remove, source_array):\n    \"\"\"\n    Removes elements from the source_array that are present in the values_to_remove.\n    \n    Args:\n        values_to_remove (list): The array containing the values to be removed.\n        source_array (list): The source array containing the values.\n\n    \n    Returns:\n        list: The source array with the matching values removed.\n    \"\"\"\n    for value in values_to_remove:\n        if value in source_array:\n            source_array.remove(value)\n    return source_array\n\n# ===============================================\n# Images methods\n# ===============================================\n\ndef get_processed_image(patient_id, dataset_version):\n    \"\"\"\n    Retrieves and processes the images for a given patient, grouping them for segmentation.\n\n    Args:\n        patient_id (str): The ID of the patient (BraTS21ID).\n\n    Returns:\n        numpy.ndarray: A processed image composed of the different images of the patient.\n    \"\"\"\n    # SEGMENTATION MODEL LIMITED TO 3 LAYERS\n    # T2W SKIPPED\n    patient_id = int(patient_id)\n\n    # Paths for image sequences\n    t1w_path = f'{train_folder_path}/{str(patient_id).zfill(5)}/T1w'\n    #t2w_path = f'{train_folder_path}/{str(patient_id).zfill(5)}/T2w'\n\n    flair_path = f'{train_folder_path}/{str(patient_id).zfill(5)}/FLAIR'\n    t1wce_path = f'{train_folder_path}/{str(patient_id).zfill(5)}/T1wCE'\n\n    # Retrieve image sequences\n    if dataset_version == DatasetVersion.V_2D:\n        t1w_image = sequence_filenames(t1w_path)\n        flair_image = sequence_filenames(flair_path)\n        t1wce_image = sequence_filenames(t1wce_path)\n        t2w_image = None\n    else:\n        t1w_image = sequence_filenames(t1w_path)\n        flair_image = sequence_filenames(flair_path)\n        t1wce_image = sequence_filenames(t1wce_path)\n        t2w_image = None\n        \n    # Resampling\n    if dataset_version == DatasetVersion.V_2D:\n        re_sampled_flair = re_sample_image(flair_image, t1w_image)\n        re_sampled_t1wce = re_sample_image(t1wce_image, t1w_image)\n        t2w_array = None\n    else:\n        re_sampled_flair = re_sample_image(flair_image, t1w_image)\n        re_sampled_t1wce = re_sample_image(t1wce_image, t1w_image)\n        t2w_array = None\n\n    \n    # Normalization\n    if dataset_version == DatasetVersion.V_2D:\n        t1w_array = normalize(sitk.GetArrayFromImage(t1w_image))\n        re_sampled_flair_array = normalize(sitk.GetArrayFromImage(re_sampled_flair))\n        re_sampled_t1wce_array = normalize(sitk.GetArrayFromImage(re_sampled_t1wce))\n        re_sampled_t2w_array = None\n    else:\n        t1w_array = normalize(sitk.GetArrayFromImage(t1w_image))\n        re_sampled_flair_array = normalize(sitk.GetArrayFromImage(re_sampled_flair))\n        re_sampled_t1wce_array = normalize(sitk.GetArrayFromImage(re_sampled_t1wce))\n        re_sampled_t2w_array = None\n\n    # Stack sequences\n    if dataset_version == DatasetVersion.V_2D:\n        sequence_stacked = np.stack([t1w_array, re_sampled_flair_array, re_sampled_t1wce_array])\n        central_slice = t1w_array.shape[0] // 2\n    else:\n        sequence_stacked = np.stack([t1w_array, re_sampled_flair_array, re_sampled_t1wce_array])\n        central_slice = t1w_array.shape[0] // 2\n\n    rvb = sequence_stacked[:, central_slice, :, :].transpose(1, 2, 0)\n    image = Image.fromarray((rvb * 255).astype(np.uint8))\n    return np.array([np.moveaxis(np.array(image.resize((256, 256))), -1, 0)])\n\n\n    \ndef sequence_filenames(path) :\n    \"\"\"\n    Retrieves a sequence of images for a given directory.\n\n    Args:\n        path (str): The path to the directory containing the DICOM data set.\n\n    Returns:\n        SimpleITK.Image: A sequence of images corresponding to the DICOM files in the directory.\n\n    Raises:\n        FileNotFoundError: If the specified path does not exist.\n    \"\"\"\n    filenames = sitk_reader.GetGDCMSeriesFileNames(path)\n    sitk_reader.SetFileNames(filenames)\n    image = sitk_reader.Execute()\n    \n    return image    \n\ndef normalize(dataset) :\n    \"\"\"\n    Normalizes the data obtained from the images.\n\n    Args:\n        dataset (numpy.ndarray): The dataset to be normalized.\n\n    Returns:\n        numpy.ndarray: The normalized dataset.\n    \"\"\"\n    return (dataset - np.min(dataset)) / (np.max(dataset) - np.min(dataset))\n\n\ndef re_sample_image(image, ref_img):\n    \"\"\"\n    Resamples the image to match the dimensions and properties of the reference image.\n\n    Args:\n        image (SimpleITK.Image): The image to be resampled.\n        ref_img (SimpleITK.Image): The reference image used for resampling.\n\n    Returns:\n        SimpleITK.Image: The resampled image.\n    \"\"\"\n    re_sampler = sitk.ResampleImageFilter()\n    re_sampler.SetReferenceImage(ref_img)\n    re_sampler.SetDefaultPixelValue(image.GetPixelIDValue())\n    re_sampler.SetInterpolator(sitk.sitkLinear)\n    re_sampler.SetOutputSpacing(ref_img.GetSpacing())\n    re_sampler.SetOutputDirection(ref_img.GetDirection())\n    re_sampler.SetOutputOrigin(ref_img.GetOrigin())\n    re_sampler.SetSize(ref_img.GetSize())\n    re_sampler.SetTransform(sitk.AffineTransform(image.GetDimension()))\n    re_sampled_image = re_sampler.Execute(image)\n    \n    return re_sampled_image\n\ndef segmentation_process(image_resized):\n    \"\"\"\n    Obtains the segmented image.\n\n    Args:\n        image_resized (numpy.ndarray): The resized image.\n\n    Returns:\n        numpy.ndarray: The segmented image.\n    \"\"\"\n    segmentation = segmentation_model(torch.Tensor(image_resized))\n    return segmentation\n    \n# ===============================================\n# Dataset creation methods\n# ===============================================\n\ndef init_dataset_radiomics() :\n    \"\"\"\n    Initializes the DataFrame structures for radiomics data acquisition.\n\n    Returns:\n        None\n    \"\"\"\n    global df_shapes\n    global df_textures\n    global df_first_orders_features\n    \n    if dataset_version == DatasetVersion.V_3D:\n        df_shapes_columns = ['ID','BraTS21ID','MeshVolume','VoxelVolume','SurfaceArea','SurfaceVolumeRatio','Sphericity',\n                             'Compactness1','Compactness2','SphericalDisproportion','Maximum3DDiameter','Maximum2DDiameterRow',\n                             'Maximum2DDiameterColumn','MajorAxisLength','MinorAxisLenth','LeastAxisLength','Elongation','Flatness']\n\n    else :\n        df_shapes_columns = ['ID','BraTS21ID','MeshSurface','PixelSurface','Perimeter','PerimeterSurfaceRatio','Sphericity',\n                             'SphericalDisproportion','MaximumDiameter','MajorAxisLength','MinorAxisLenth','Elongation']\n\n    \n    df_shapes = pd.DataFrame(columns=df_shapes_columns) \n\n    # GLCM features\n    df_textures_columns = ['ID','Autocorrelation', 'ClusterProminence', 'ClusterShade', 'ClusterTendency', 'Contrast', \n                           'Correlation', 'DifferenceAverage', 'DifferenceEntropy', 'DifferenceVariance', 'Id', 'Idm', \n                           'Idmn', 'Idn', 'Imc1', 'Imc2', 'InverseVariance', 'JointAverage', 'JointEnergy', 'JointEntropy', \n                           'MCC', 'MaximumProbability', 'SumAverage', 'SumEntropy', 'SumSquares']\n    df_textures = pd.DataFrame(columns=df_textures_columns) \n\n\n    # FIRST ORDERS features\n    df_first_orders_features_columns=['ID','10Percentile', '90Percentile', 'Energy', 'Entropy', 'InterquartileRange', 'Kurtosis', \n                                      'Maximum', 'MeanAbsoluteDeviation', 'Mean', 'Median', 'Minimum', 'Range', 'RobustMeanAbsoluteDeviation', \n                                      'RootMeanSquared', 'Skewness', 'TotalEnergy', 'Uniformity', 'Variance']\n    df_first_orders_features = pd.DataFrame(columns=df_first_orders_features_columns) \n\n    # gzlm features\n    df_gzlm_features_columns = ['ID','SmallAreaEmphasis','LargeAreaEmphasis','GrayLevelNonUniformity','GrayLevelNonUniformityNormalized',\n                                'SizeZoneNonUniformity','SizeZoneNonUniformityNormalized','ZonePercentage','GrayLevelVariance','ZoneVariance',\n                                'ZoneEntropy','LowGrayLevelZoneEmphasis','HighGrayLevelZoneEmphasis','SmallAreaLowGrayLevelEmphasis',\n                                'SmallAreaHighGrayLevelEmphasis','LargeAreaLowGrayLevelEmphasis','LargeAreaHighGrayLevelEmphasis']\n    df_gzlm_features = pd.DataFrame(columns=df_gzlm_features_columns)\n    \n    #glrlm_features\n    df_glrlm_features_columns = ['ID','ShortRunEmphasis','LongRunEmphasis','GrayLevelNonUniformity','RunLengthNonUniformity','RunLengthNonUniformityNormalized'\n                                 ,'RunPercentage','GrayLevelVariance','RunVariance','RunEntropy','LowGrayLevelRunEmphasis','HighGrayLevelRunEmphasis'\n                                 ,'ShortRunLowGrayLevelEmphasis','ShortRunHighGrayLevelEmphasis','LongRunLowGrayLevelEmphasis','LongRunHighGrayLevelEmphasis']\n    df_glrlm_features = pd.DataFrame(columns=df_glrlm_features_columns)\n    # ngtdm features\n    df_ngtdm_features_columns = ['ID','Coarseness','Contrast','Busyness','Complexity','Strength']\n    df_ngtdm_features = pd.DataFrame(columns=df_ngtdm_features_columns)\n    \n    # gldm features\n    df_gldm_features_columns = ['ID','SmallDependenceEmphasis','LargeDependenceEmphasis','GrayLevelNonUniformity','GrayLevelNonUniformityNormalized','DependenceNonUniformity'\n                               , 'DependenceNonUniformityNormalized', 'GrayLevelVariance', 'DependenceVariance', 'DependenceEntropy', 'DependencePercentage'\n                                , 'LowGrayLevelEmphasis', 'HighGrayLevelEmphasis', 'SmallDependenceLowGrayLevelEmphasis', 'SmallDependenceHighGrayLevelEmphasis'\n                               , 'LargeDependenceLowGrayLevelEmphasis', 'LargeDependenceHighGrayLevelEmphasis']\n    df_gldm_features = pd.DataFrame(columns=df_gldm_features_columns)\n    \n\ndef add_patient_data2D(ID,img_resized,segmentation) :\n    \"\"\"\n    Adds data from the specified patient's images to the analysis dataset.\n\n    Args:\n        ID (int): The ID of the patient.\n        img_resized (numpy.ndarray): Resized image of the patient.\n        segmentation (torch.Tensor): Segmentation of the patient's image.\n\n    Returns:\n        None\n    \"\"\"\n    global df_shapes\n    global df_textures\n    global df_first_orders_features\n    \n    # shape\n    results = radiomics.shape2D.RadiomicsShape2D(\n        sitk.GetImageFromArray(img_resized), \n        sitk.GetImageFromArray(np.array([\n            segmentation[0][0].detach().cpu().numpy() > 0.5\n        ]).astype(np.uint8)),\n        force2D=True\n    )\n    \n    shape2D = {}\n    shape2D['ID'] = int(ID)\n    shape2D['BraTS21ID'] = int(ID)\n    shape2D['MeshSurface'] = results.getMeshSurfaceFeatureValue()\n    shape2D['PixelSurface'] = results.getPixelSurfaceFeatureValue()\n    shape2D['Perimeter'] = results.getPerimeterFeatureValue()\n    shape2D['PerimeterSurfaceRatio'] = results.getPerimeterSurfaceRatioFeatureValue()\n    shape2D['Sphericity'] = results.getSphericityFeatureValue()\n    shape2D['SphericalDisproportion'] = results.getSphericalDisproportionFeatureValue()\n    shape2D['MaximumDiameter'] = results.getMaximumDiameterFeatureValue()\n    shape2D['MajorAxisLength'] = results.getMajorAxisLengthFeatureValue()\n    shape2D['MinorAxisLenth'] = results.getMinorAxisLengthFeatureValue()\n    shape2D['Elongation'] = results.getElongationFeatureValue()\n    \n    df_shapes=df_shapes.append(shape2D,ignore_index=True)\n    \n    # GLCM\n    results=radiomics.glcm.RadiomicsGLCM(\n        sitk.GetImageFromArray(img_resized[0,0,:,:].reshape(1, 256, 256)), \n        sitk.GetImageFromArray(np.array([\n            segmentation[0][0].detach().cpu().numpy() > 0.5\n        ]).astype(np.uint8)),\n        force2D=True\n    )\n\n    results.enableAllFeatures()\n    res = results.execute()\n    res['ID']=int(ID)\n\n    df_textures=df_textures.append(res,ignore_index=True)\n    \n    # First-orders features\n    results =  radiomics.firstorder.RadiomicsFirstOrder(\n        sitk.GetImageFromArray(img_resized[0,0,:,:].reshape(1, 256, 256)), \n        sitk.GetImageFromArray(np.array([\n            segmentation[0][0].detach().cpu().numpy() > 0.5\n        ]).astype(np.uint8)),\n        force2D=True\n    )\n\n    results.enableAllFeatures()\n    res = results.execute()\n    res['ID']=int(ID)\n\n    df_first_orders_features=df_first_orders_features.append(res,ignore_index=True)\n    \n\ndef add_patient_data3D(ID,img_resized,segmentation) :\n    \"\"\"\n    Adds data from the specified patient's images to the analysis dataset.\n\n    Args:\n        ID (int): The ID of the patient.\n        img_resized (numpy.ndarray): Resized image of the patient.\n        segmentation (torch.Tensor): Segmentation of the patient's image.\n\n    Returns:\n        None\n    \"\"\"\n    global df_shapes\n    global df_textures\n    global df_first_orders_features\n    global df_gzlm_features\n    global df_glrlm_features\n    global df_ngtdm_features\n    global df_gldm_features\n\n    # shape\n    results = radiomics.shape.RadiomicsShape(\n        sitk.GetImageFromArray(img_resized), \n        sitk.GetImageFromArray(np.array([\n            segmentation[0][0].detach().cpu().numpy() > 0.5\n        ]).astype(np.uint8))#,\n        #force2D=True\n    )\n    \n    shape3D = {}\n    shape3D['ID'] = int(ID)\n    shape3D['BraTS21ID'] = int(ID)\n    shape3D['MeshVolume'] = results.getMeshVolumeFeatureValue()\n    shape3D['VoxelVolume'] = results.getVoxelVolumeFeatureValue()\n    shape3D['SurfaceArea'] = results.getSurfaceAreaFeatureValue()\n    shape3D['SurfaceVolumeRatio']=results.getSurfaceVolumeRatioFeatureValue()\n    shape3D['Sphericity'] = results.getSphericityFeatureValue()\n    shape3D['Compactness1']=results.getCompactness1FeatureValue()\n    shape3D['Compactness2']=results.getCompactness2FeatureValue()\n    shape3D['SphericalDisproportion']=results.getSphericalDisproportionFeatureValue()\n    shape3D['Maximum3DDiameter']=results.getMaximum3DDiameterFeatureValue()\n    shape3D['Maximum2DDiameterRow']=results.getMaximum2DDiameterRowFeatureValue()\n    shape3D['Maximum2DDiameterColumn']=results.getMaximum2DDiameterColumnFeatureValue()\n    shape3D['MajorAxisLength'] = results.getMajorAxisLengthFeatureValue()\n    shape3D['MinorAxisLenth'] = results.getMinorAxisLengthFeatureValue()\n    shape3D['LeastAxisLength'] = results.getLeastAxisLengthFeatureValue()\n    shape3D['Elongation'] = results.getElongationFeatureValue()\n    shape3D['Flatness'] = results.getFlatnessFeatureValue()\n  \n    df_shapes=df_shapes.append(shape3D,ignore_index=True)\n    \n    # GLCM\n    results=radiomics.glcm.RadiomicsGLCM(\n        sitk.GetImageFromArray(img_resized[0,0,:,:].reshape(1, 256, 256)), \n        sitk.GetImageFromArray(np.array([\n            segmentation[0][0].detach().cpu().numpy() > 0.5\n        ]).astype(np.uint8))#,\n        #force2D=True\n    )\n\n    results.enableAllFeatures()\n    res = results.execute()\n    res['ID']=int(ID)\n    \n    df_first_orders_features=df_first_orders_features.append(res,ignore_index=True)  \n    \n    # GLSZM features\n    results = radiomics.glszm.RadiomicsGLSZM(\n    \n        sitk.GetImageFromArray(img_resized[0,0,:,:].reshape(1, 256, 256)), \n        sitk.GetImageFromArray(np.array([\n            segmentation[0][0].detach().cpu().numpy() > 0.5\n        ]).astype(np.uint8))#,\n        #force2D=True\n    )\n    results.enableAllFeatures()\n    res = results.execute()\n    res['ID']=int(ID)\n    df_gzlm_features = df_gzlm_features.append(res,ignore_index=True)\n    \n    # GLRLM features\n    results = radiomics.glrlm.RadiomicsGLRLM(\n         sitk.GetImageFromArray(img_resized[0,0,:,:].reshape(1, 256, 256)), \n        sitk.GetImageFromArray(np.array([\n            segmentation[0][0].detach().cpu().numpy() > 0.5\n        ]).astype(np.uint8))#,\n        #force2D=True\n    )\n    results.enableAllFeatures()\n    res = results.execute()\n    res['ID']=int(ID)\n    df_glrlm_features = df_glrlm_features.append(res,ignore_index=True)\n    \n    # NGTDM features\n    results = radiomics.ngtdm.RadiomicsNGTDM(\n        sitk.GetImageFromArray(img_resized[0,0,:,:].reshape(1, 256, 256)), \n        sitk.GetImageFromArray(np.array([\n            segmentation[0][0].detach().cpu().numpy() > 0.5\n        ]).astype(np.uint8))#,\n        #force2D=True\n    )\n    results.enableAllFeatures()\n    res = results.execute()\n    res['ID']=int(ID)\n    df_ngtdm_features = df_ngtdm_features.append(res,ignore_index=True)\n    \n    # GLDM features\n    results=radiomics.gldm.RadiomicsGLDM(\n        sitk.GetImageFromArray(img_resized[0,0,:,:].reshape(1, 256, 256)), \n        sitk.GetImageFromArray(np.array([\n            segmentation[0][0].detach().cpu().numpy() > 0.5\n        ]).astype(np.uint8))#,\n        #force2D=True\n    )\n    results.enableAllFeatures()\n    res = results.execute()\n    res['ID']=int(ID)\n    df_gldm_features = df_gldm_features.append(res,ignore_index=True) \n\n    \n# ===============================================\n# Show methods\n# ===============================================\n    \ndef show_brains(patient_id_mgmt_tuple_list, figsize, dataset_version, scans_cmap=['red', 'yellow'], show_segmentation=True, segmentation_cmap=\"Reds\"):\n    \"\"\"\n    Display a preview of patients from the given dataset.\n\n    Args:\n        patient_id_mgmt_tuples (list): List of tuples containing patient IDs and MGMT values. (Id, MGMT)\n        dataset_version (str): Version of the dataset (2D or 3D).\n    \"\"\"\n    # Définir le thème seaborn avec la palette personnalisée\n    custom_palette = sns.color_palette(scans_cmap)\n    sns.set_palette(custom_palette)\n    \n    # Loading\n    loading_max = len(patient_id_mgmt_tuple_list)\n    progress_bar = widgets.IntProgress(min=0, max=loading_max, description='Processing patients:', bar_style='info')\n    display(progress_bar)\n\n    # Process and display images for each patient\n    for i, patient_id_mgmt_value in enumerate(patient_id_mgmt_tuple_list, start=1):\n        # Process images for patient with MGMT value\n        patient_id = patient_id_mgmt_value[0]\n        patient_mgmt = patient_id_mgmt_value[1]\n        img_resized = get_processed_image(patient_id, dataset_version)\n        \n        if dataset_version == DatasetVersion.V_3D:\n            titles = ['T1w', 'FLAIR', 'T1wce', 'Segmentation'] # T2w\n        else:\n            titles = ['T1w', 'FLAIR', 'T1wce', 'Segmentation'] # T2w\n        \n        # Create the main figure\n        fig = plt.figure(figsize = figsize)\n\n        # Adjust top margin for the main figure\n        fig.subplots_adjust(top=0.85)\n\n        # Set the global title\n        fig.suptitle(f\"Patient {patient_id} and MGMT = {patient_mgmt}\", y=0.75)\n\n        x_size_subplot = 4 if show_segmentation else 3  \n        # Iterate over the source images\n        for i in range(3):\n            ax = fig.add_subplot(1, x_size_subplot, 1+i, )\n            \n            ax.imshow(img_resized[0, i])\n            ax.set_title(titles[i])\n            ax.set_xticks([])\n            ax.set_yticks([])\n            \n        if show_segmentation:\n            segmentation = segmentation_process(img_resized)\n            # Add the segmentation image\n            ax_segmentation = fig.add_subplot(1, 4, 4)\n            ax_segmentation.imshow(segmentation.detach().numpy()[0, 0], cmap=segmentation_cmap)\n            ax_segmentation.set_title(titles[3])\n            ax_segmentation.set_xticks([])\n            ax_segmentation.set_yticks([])\n   \n        plt.tight_layout()\n        plt.show()\n        \n        progress_bar.value = i + 1\n\n    sns.set_theme(palette=default_palette_color)\n    progress_bar.close()\n\n    \ndef show_segmentation(title, figsize, img_src, dataset_version, segmentation):\n    \"\"\"\n    Displays the resized source images and the segmentation image in a single line.\n\n    Args:\n        title (str): Global title for the plot.\n        img_src (numpy.ndarray): Resized source images.\n        segmentation (torch.Tensor): Segmentation image.\n\n    Returns:\n        None\n    \"\"\"\n    \ndef show_download_link(df, title = \"Download CSV file\", filename = \"data.csv\"):\n    \"\"\"\n    Displays a download link for a DataFrame as a CSV file.\n\n    Args:\n        df (pandas.DataFrame): The DataFrame to be downloaded.\n        title (str): The title of the download link (default: \"Download CSV file\").\n        filename (str): The name of the downloaded file (default: \"data.csv\").\n\n    Returns:\n        None\n    \"\"\"\n    csv = df.to_csv()\n    b64 = base64.b64encode(csv.encode())\n    payload = b64.decode()\n    html = '<a download=\"{filename}\" href=\"data:text/csv;base64,{payload}\" target=\"_blank\">{title}</a>'\n    html = html.format(payload=payload, title=title, filename=filename)\n    display(HTML(html))\n    \ndef fonc_test_normality(df, quantitative_var) : \n    fig = plt.figure(figsize=(12, 4))\n    plt.subplot(1,2,1)\n    sns.histplot(data=df[df['MGMT_value'] == 1][['MGMT_value', quantitative_var]], x=quantitative_var, kde=True, stat='density', color='orange', alpha=0.4)\n    sns.histplot(data=df[df['MGMT_value'] == 0][['MGMT_value', quantitative_var]], x=quantitative_var, kde=True, stat='density', color='blue', alpha=0.4)\n    plt.title('histogram of '+quantitative_var,fontsize=8)\n    plt.xlabel('Value',fontsize=7)\n    plt.ylabel('Frequence',fontsize=7)\n    plt.subplot(1,2,2)\n    sns.boxplot(data=df, y=quantitative_var, x='MGMT_value')\n    plt.title('Boxplot of '+quantitative_var,fontsize=8)\n    #plt.xlabel('Valeur')\n    plt.show();\n    \ndef create_kde_mgmt_pos_neg(df, x, y) : \n    \"\"\"\n    Displays 3 KDE plots that show the bivariate density using the columns x and y from a dataframe.\n\n    Args:\n        df: DataFrame, the dataframe whose 2 variables are going to be used for the KDE plot.\n        x, y: String, the two columns\n        \n    \"\"\"\n    fig = plt.figure(figsize=(15, 3))\n    fig.suptitle(\"Bivariate density of \"+y+\" considering \"+x, fontsize=10)\n    \n    plt.subplot(1,3,1)\n    sns.kdeplot(data=df, x=x, y=y, hue=\"MGMT_value\")\n\n    plt.subplot(1,3,2)\n    plt.title(\"Bivariate density for MGMT positive\", fontsize = 6)\n    sns.kdeplot(data=df[df[\"MGMT_value\"] == 1], x=x, y=y, color=\"orange\")\n\n    plt.subplot(1,3,3)\n    plt.title(\"Bivariate density for MGMT negative\", fontsize = 6)\n    sns.kdeplot(data=df[df[\"MGMT_value\"] == 0], x=x, y=y)\n    plt.show();\n\n\ndef show_explore_distribution_normality(df, quantitative_var, figsize=(20, 5)):\n    fig, axes = plt.subplots(nrows=1, ncols=6, figsize=figsize)\n\n    # Histogram with density estimation\n    histplot = sns.histplot(data=df, x=quantitative_var, kde=True, hue='MGMT_value', ax=axes[0])\n    axes[0].set_title('Distribution of Data')\n    axes[0].set_xlabel(quantitative_var)\n    axes[0].set_ylabel('Frequency')\n\n    legend = histplot.get_legend()\n    legend.set_title('MGMT_value')\n    legend.set_bbox_to_anchor((1, 1))\n\n    # Density plot with normal distribution\n    sns.kdeplot(data=df, x=quantitative_var, ax=axes[1], label='Density')\n    x = np.linspace(np.min(df[quantitative_var]), np.max(df[quantitative_var]), 100)\n    y = stats.norm.pdf(x, np.mean(df[quantitative_var]), np.std(df[quantitative_var]))\n    axes[1].plot(x, y, linewidth=2, label='Normal Distribution')\n    axes[1].set(title='Density Plot with Normal Distribution', xlabel=quantitative_var, ylabel='Density')\n    axes[1].legend(loc='upper right')\n\n    \n    # Boxplot\n    sns.boxplot(data=df, y=quantitative_var, x='MGMT_value', hue='MGMT_value', ax=axes[2])\n    axes[2].set_title('Comparison of Distributions')\n    axes[2].set_xlabel('MGMT_value')\n    axes[2].set_ylabel(quantitative_var)\n    axes[2].legend(title='MGMT_value', loc='upper center')\n\n    # Violin plot\n    sns.violinplot(data=df, y=quantitative_var, x='MGMT_value', hue='MGMT_value', ax=axes[3])\n    axes[3].set_title('Distribution and Density')\n    axes[3].set_xlabel('MGMT_value')\n    axes[3].set_ylabel(quantitative_var)\n    axes[3].legend(title='MGMT_value', loc='upper center')\n\n    # Q-Q plot\n    stats.probplot(df[quantitative_var], plot=axes[4])\n    axes[4].set_title('Quantile-Quantile (Q-Q) Comparison')\n    axes[4].set_xlabel('Theoretical Quantiles')\n    axes[4].set_ylabel('Ordered Values')\n    \n    # Scatter plot of values vs. mean\n    grouped = df.groupby('MGMT_value')\n    for group_name, group_data in grouped:\n        axes[5].scatter(range(len(group_data)), group_data[quantitative_var], label=group_name, alpha=0.5)\n    mean_value = np.mean(df[quantitative_var])\n    axes[5].axhline(y=mean_value, color='r', linestyle='--')\n    axes[5].set_title('Values vs. Mean')\n    axes[5].set_xlabel('Index')\n    axes[5].set_ylabel(quantitative_var)\n    axes[5].legend(title='MGMT_value', loc='upper right')\n\n    plt.suptitle(f'Exploratory Analysis of Data Distribution and Normality Assessment of {quantitative_var}', fontsize=14)\n    plt.tight_layout()\n    plt.show()\n\n\ndef show_test_normality(df):\n    new_df = pd.DataFrame(index=['skewness', 'kurtosis', 'excess_kurtosis', 'shapiro_test', 'normalite'])\n\n    for col in df.columns:\n        skewness = stats.skew(df[col])\n        kurtosis = stats.kurtosis(df[col], fisher=False)\n        excess_kurtosis = stats.kurtosis(df[col])\n        shapiro_test = stats.shapiro(df[col])[1]\n        normalite = 'Yes' if shapiro_test > 0.05 else 'No'\n\n\n        new_df[col] = [0.000001 if skewness == 0 else skewness,\n                       0.000001 if kurtosis == 0 else kurtosis,\n                       0.000001 if excess_kurtosis == 0 else excess_kurtosis,\n                       shapiro_test,\n                       normalite]\n\n    return new_df\n\n\n\n\ndef show_3D_scatter_plots(\n    combinations_party,\n    dataset, \n    title=None,\n    figsize=(14, 25),\n    elev_angle=30, \n    azimuth_angle=15, \n    show_legend=False,\n    legend_loc='lower center',\n    cmap='coolwarm',\n    alpha=1.0\n):\n    \"\"\"\n    Display 3D scatter plots for each combination of variables in combinations_party using the given dataset.\n\n    Args:\n        combinations_party (list): List of combinations of variables to plot (x, y, z).\n        dataset (DataFrame): The dataset containing the variables.\n        figsize (tuple, optional): Figure size in inches. Defaults to (14, 25).\n        elev_angle (float, optional): Elevation angle in degrees. Defaults to 30.\n        azimuth_angle (float, optional): Azimuth angle in degrees. Defaults to 15.\n        title (str, optional): Title for the plot. Defaults to None.\n        show_legend (bool, optional): Whether to show the legend. Defaults to True.\n        cmap (str, optional): Color map for the MGMT values. Defaults to 'coolwarm'.\n        alpha (float, optional): Transparency of the data points. Defaults to 1.0.\n    \"\"\"\n    num_combinations = len(combinations_party)\n    num_rows = int(math.ceil(num_combinations / 2))\n    num_cols = min(2, num_combinations)\n\n    fig, axes = plt.subplots(num_rows, num_cols, subplot_kw={'projection': '3d'}, figsize=figsize)\n\n    # Loop to create scatter plots for each combination\n    for i, combo in enumerate(combinations_party):\n        if num_combinations == 1:\n            ax = axes\n        elif num_rows == 1:\n            ax = axes[i]\n        else:\n            row = i // num_cols\n            col = i % num_cols\n            ax = axes[row, col]\n\n        value_x = combo[0]\n        value_y = combo[1]\n        value_z = combo[2]\n        value_c = value_z\n        \n        x = dataset[value_x]\n        y = dataset[value_y]\n        z = dataset[value_z]\n        c = dataset[value_c]\n            \n\n        # Display 3D scatter plot\n        scatter = ax.scatter(x, y, z, c=c, cmap=cmap, alpha=alpha)\n\n        # Add labels to axes\n        ax.set_xlabel(value_x)\n        ax.set_ylabel(value_y)\n        ax.set_zlabel(value_z)\n        ax.view_init(elev=elev_angle, azim=azimuth_angle)\n        ax.set_facecolor('white')\n        scatter.set_label(title)\n\n        if show_legend:\n            unique_values = np.unique(c)\n            colors = plt.get_cmap(cmap)(np.linspace(0, 1, len(unique_values)))\n            labels = [f\"{value_c} = {value}\" for value in unique_values]\n                    \n            # Ajouter une légende personnalisée\n            custom_legend = [plt.Line2D([0], [0], marker='o', color='w', markerfacecolor=color, markersize=10) for color in colors]\n            ax.legend(custom_legend, labels, loc=legend_loc, ncol=1)\n\n\n\n    # Hide the last subplot if the number of combinations is odd\n    if num_combinations % 2 == 1 and num_combinations > 1:\n        axes[num_rows-1, num_cols-1].axis('off')\n\n    fig.suptitle(title)  # Set the overall title for the plot\n\n    fig.tight_layout()\n    plt.show()\n    plt.close()\n\n\ndef show_triangle_correlation_matrix(data, title=\"Correlation Matrix\", filter_threshold=0, figsize=(50, 20)):\n    corr_df = data.corr()\n    if filter_threshold > 0:\n        corr_df = filter_correlation_matrix(corr_df, filter_threshold)\n\n    plt.figure(figsize=figsize)\n    \n    mask = np.triu(np.ones_like(corr_df))\n    ax = sns.heatmap(corr_df, annot=True, mask=mask, cmap=\"coolwarm\")  # Ajout de linewidths\n    ax.set_facecolor('white')\n    plt.title(title)\n    plt.setp(ax.get_xticklabels(), rotation=45, ha=\"right\", rotation_mode=\"anchor\")\n    \n    plt.show()\n    \ndef show_brain_slices(segmentation_path_folder, split_by = None, figsize=(50, 10)):\n    file_list = sorted(os.listdir(segmentation_path_folder), key=extract_number_from_filename)\n        \n    if split_by is not None:\n        file_list = file_list[::split_by]\n    \n    if len(file_list) == 0:\n        print(\"Aucune image à afficher.\")\n        return\n    \n    slice_images = []\n    \n    for file_name in file_list:\n        if file_name.endswith('.dcm'):\n            segmentation_path = os.path.join(segmentation_path_folder, file_name)\n            segmented_image = sitk.ReadImage(segmentation_path)\n            segmented_array = sitk.GetArrayFromImage(segmented_image)\n            slice_image = segmented_array[0]\n            slice_images.append(slice_image)\n    \n    \n    num_rows = len(slice_images) // 10 \n    num_columns =  len(slice_images) // num_rows  \n    fig, axes = plt.subplots(num_rows, num_columns, figsize=figsize)\n    \n    for ax, image in zip(axes.ravel(), slice_images):\n        ax.imshow(image, cmap='gray')\n        ax.axis('off')\n\n    plt.show()\n# ===============================================\n# Best Features\n# ===============================================\nfrom sklearn.feature_selection import RFECV\nfrom sklearn.feature_selection import RFECV\n\ndef select_best_features_RFE(df, target, models, features_scalers, reduction_step=0.001, cv=5, scoring='r2'):\n    \"\"\"\n    This function selects features using Recursive Feature Elimination (RFE) method.\n\n    Parameters:\n    df (DataFrame): The input dataframe.\n    features_and_scaler_tuple_list (list): List of tuples containing feature names and corresponding scaler.\n    target (str): The name of the target variable.\n    model (estimator): The estimator supports either fit_transform method or feature_importances_ attribute.\n    n_features_to_select (int, optional): The number of features to select. If None, half of the features are selected. Defaults to None.\n    reduction_step (float or int, optional): If greater than or equal to 1, then step corresponds to the (integer) number of features to remove at each iteration. If within (0.0, 1.0), then step corresponds to the percentage (rounded down) of features to remove at each iteration. Defaults to 0.001.\n\n    Returns:\n    list: The list of selected feature names ordered by importance.\n    \"\"\"\n    all_features = [feature for features, scaler in features_scalers for feature in features]\n\n    # Get the feature matrix and the target variable\n    X = df[all_features]\n    X_scaled = X.copy()\n    y = df[target]\n\n    # Iterate over the feature-scaler tuples\n    for features, scaler in features_scalers:\n        # Apply the scaler on the features\n        for feature in features:\n            X_scaled[feature] = scaler.fit_transform(X_scaled[feature].values.reshape(-1, 1))\n\n    results = []\n    for name, model, apply_scalers in models:\n        rfe = RFECV(estimator=model, step=reduction_step, cv=cv, scoring=scoring)\n        X_fit = X_scaled if apply_scalers else X\n        rfe.fit(X_fit, y)\n        selected_features = X_fit.iloc[:, rfe.support_]\n        score = rfe.score(X_fit, y)\n        results.append((name, score, selected_features))\n\n    return sorted(results, key=lambda x: x[1], reverse=True)\n\n\nfrom boruta import BorutaPy\nfrom sklearn.model_selection import cross_val_score\n\ndef select_best_features_Boruta(df, target, models, features_scalers, max_iter=100, n_estimators='auto', cv=5, scoring='r2', random_state=0):\n    \"\"\"\n    This function selects features using BorutaPy method.\n\n    Parameters:\n    df (DataFrame): The input dataframe.\n    target (str): The name of the target variable.\n    models (list): List of tuples containing model names and corresponding model instances.\n    features_scalers (list): List of tuples containing feature names and corresponding scaler instances.\n    max_iter (int, optional): Maximum number of iterations for Boruta. Defaults to 100.\n    cv (int or cross-validation generator, optional): Determines the cross-validation splitting strategy. Defaults to 5.\n    random_state (int, optional): Random seed for reproducibility. Defaults to 0.\n\n    Returns:\n    list: The list of selected feature names ordered by importance.\n    \"\"\"\n    all_features = [feature for features, scaler in features_scalers for feature in features]\n    \n    # Get the feature matrix and the target variable\n    X = df[all_features]\n    y = df[target]\n\n    # Iterate over the feature-scaler tuples\n    for features, scaler in features_scalers:\n        # Apply the scaler on the features\n        for feature in features:\n            X[feature] = scaler.fit_transform(X[feature].values.reshape(-1, 1))\n\n    results = []\n    for name, model in models:\n        boruta = BorutaPy(estimator=model, n_estimators=n_estimators, max_iter=max_iter, random_state=random_state)\n        # Perform cross-validation with BorutaPy\n        scores = cross_val_score(boruta, X.values, y.values, cv=cv, scoring=scoring)\n        boruta.fit(X.values, y.values)\n        selected_features = X.iloc[:, boruta.support_]\n        results.append((name, scores.mean(), selected_features.columns.tolist()))\n\n    return results\n\n# ===============================================\n# OutlierDetector\n# ===============================================\n\nclass OutlierDetector:\n    def __init__(self, threshold=1.5):\n        self.threshold = threshold\n\n    def detect_outliers(self, dataset_df, target_column, filter_column=None, filter_value=None):\n        # Calculate quartiles for each column\n        quartiles = dataset_df.quantile([0.25, 0.75])\n\n        # Calculate the interquartile range (IQR) for each column\n        iqr = quartiles.loc[0.75] - quartiles.loc[0.25]\n\n        # Filter the DataFrame based on the filter_column and filter_value if provided\n        if filter_column and filter_value:\n            dataset_filtered_df = dataset_df[dataset_df[filter_column] == filter_value]\n        else:\n            dataset_filtered_df = dataset_df\n\n        # Find columns with outliers\n        outlier_columns = dataset_filtered_df.drop(target_column, axis=1).apply(\n            lambda x: any((x > quartiles.loc[0.75, x.name] + self.threshold * iqr[x.name]) | \n                          (x < quartiles.loc[0.25, x.name] - self.threshold * iqr[x.name])),\n            axis=0\n        )\n\n        # Create a DataFrame with columns containing outliers\n        df_outliers = pd.DataFrame({\n            'Features': outlier_columns.index,\n            'Outliers present': outlier_columns.values\n        })\n\n        # Set the index of the DataFrame\n        df_outliers.set_index(\"Features\", inplace=True)\n\n        # Count the occurrences of outliers\n        count = df_outliers['Outliers present'].value_counts()\n\n        return df_outliers, count\n    \n# ===============================================\n# Correlation\n# ===============================================\n\ndef generate_correlated_features_table(dataset, correlation_thresholds):\n    \"\"\"\n    Generate a table of correlated features based on the given correlation thresholds.\n\n    Args:\n        dataset (pd.DataFrame): The dataset to analyze.\n        correlation_thresholds (list): List of correlation thresholds to consider.\n\n    Returns:\n        pd.DataFrame: The table of correlated features.\n    \"\"\"\n\n    groups_correlated = {}\n\n    for correlation_threshold in correlation_thresholds:\n        groups_correlated[correlation_threshold] = find_highly_correlated_groups(dataset.corr(), correlation_threshold)\n\n    # Préparer les données pour le tableau\n    correlated_features = {}\n\n    for correlation_threshold, groups_correlated_threshold in groups_correlated.items():\n        correlated_features[correlation_threshold] = [\", \".join(group_correlated) for group_correlated in groups_correlated_threshold if len(group_correlated) > 1]\n\n    # Vérifier et ajuster les longueurs des listes\n    max_length = max(len(features) for features in correlated_features.values())\n\n    for features in correlated_features.values():\n        features.extend([\"\"] * (max_length - len(features)))\n\n    # Créer un DataFrame avec les données\n    df_correlated = pd.DataFrame(correlated_features)\n    # Ajouter un titre au DataFrame\n    df_correlated.columns.name = \"Correlation Thresholds\"\n\n    return df_correlated\n\ndef extract_correlated_features(dataset, correlation_thresholds):\n    \"\"\"\n    Extracts correlated features from a dataset based on given correlation thresholds.\n    \n    Args:\n        dataset (pandas.DataFrame): The input dataset.\n        correlation_thresholds (list): A list of correlation thresholds.\n    \n    Returns:\n        numpy.ndarray: An array of correlated features.\n    \"\"\"\n    correlation_features = generate_correlated_features_table(dataset.copy(), [correlation_thresholds])\n    correlation_features = correlation_features.T.to_numpy()\n    correlation_features = np.concatenate(correlation_features).ravel()\n    correlation_features = [x for x in correlation_features if x != \"\"]\n    correlation_features = [string.split(\", \") for string in correlation_features]\n    correlation_features = np.concatenate(correlation_features).ravel()\n    \n    return correlation_features\n\n\ndef filter_correlation_matrix(correlation_matrix, correlation_threshold):\n    \"\"\"\n    Filters a correlation matrix by keeping only the absolute values greater than or equal to the correlation threshold.\n\n    Args:\n        correlation_matrix (pd.DataFrame): The correlation matrix.\n        correlation_threshold (float): The correlation threshold for filtering the matrix.\n\n    Returns:\n        pd.DataFrame: The filtered correlation matrix.\n\n    \"\"\"\n    filtered_correlation_matrix = correlation_matrix.copy()\n    filtered_correlation_matrix[abs(filtered_correlation_matrix) < correlation_threshold] = np.nan\n\n    return filtered_correlation_matrix\n\n\ndef find_highly_correlated_groups(correlation_matrix, correlation_threshold = 0.8, filter_duplicated_group = True, convert_indices_to_column_names = True):\n    \"\"\"\n    Finds groups of highly correlated variables from a correlation matrix.\n\n    Args:\n        correlation_matrix (pd.DataFrame): The correlation matrix.\n        correlation_threshold (float): The correlation threshold to consider as highly correlated.\n        filter_duplicated_group (bool): Indicates whether to filter out duplicated values in the correlated groups.\n        convert_indices_to_column_names (bool): Indicates whether to convert indices to column names.\n\n    Returns:\n        List[List[str]]: A list of groups, where each group contains the names of variables that are highly correlated.\n\n    \"\"\"\n    n = correlation_matrix.shape[0]  # Number of variables in the correlation matrix\n    groups_correlated = []  # List to store the correlated groups\n    \n    # Retrieve column names\n    column_names = correlation_matrix.columns\n    \n    # Traverse each variable\n    for i in range(n):\n        if column_names[i] not in [column_names[v] for g in groups_correlated for v in g]:  # Check if the variable has already been added to a group\n            group = [i]  # Create a new group containing the current variable (i)\n            for j in range(i+1, n):\n                if column_names[j] not in [column_names[v] for g in groups_correlated for v in g]:  # Check if the variable has already been added to a group\n                    correlation = correlation_matrix.iloc[i, j]  # Retrieve the correlation between variables i and j\n                    if abs(correlation) >= correlation_threshold:  # Strong correlation condition (adjust as needed)\n                        group.append(j)  # Add variable j to the group\n            \n            groups_correlated.append(group)  # Add the group to the list of correlated groups\n    \n    # Filter out duplicated values in the correlated groups\n    if filter_duplicated_group:\n        filtered_groups_correlated = []\n        for group in groups_correlated:\n            filtered_group = list(set(group))  # Convert to a set to eliminate duplicates, then convert back to a list\n            filtered_groups_correlated.append(filtered_group)\n        groups_correlated = filtered_groups_correlated # Reset by new one\n    \n    # Convert indices to column names\n    if convert_indices_to_column_names:\n        groups_correlated = [[column_names[i] for i in group] for group in groups_correlated]\n    \n    return groups_correlated","metadata":{"_uuid":"5a760e1d-74f2-4c96-8949-7e6daf4bcff8","_cell_guid":"afc69598-2d6f-42f4-904d-7e836dc83f2f","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-10-03T05:05:03.495444Z","iopub.execute_input":"2024-10-03T05:05:03.495818Z","iopub.status.idle":"2024-10-03T05:05:03.641363Z","shell.execute_reply.started":"2024-10-03T05:05:03.495784Z","shell.execute_reply":"2024-10-03T05:05:03.640513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total = len(preview_dataset_df)  # Nombre total d'échantillons\ntitle = \"Total number of files\"\ndata = {title: [total]}\n\ndf = pd.DataFrame(data)\ndf.set_index(title, inplace=True)\ndf.head()","metadata":{"_uuid":"9e68198c-5bce-4c61-a33b-5ee3d6451821","_cell_guid":"b5bfcec0-ef1f-430c-829b-5524f221a045","_kg_hide-output":false,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:05:03.642725Z","iopub.execute_input":"2024-10-03T05:05:03.643116Z","iopub.status.idle":"2024-10-03T05:05:03.664533Z","shell.execute_reply.started":"2024-10-03T05:05:03.643067Z","shell.execute_reply":"2024-10-03T05:05:03.663507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preview_dataset_df.head(10)","metadata":{"_uuid":"dfabdc6d-078c-4d65-a86c-3cd1546c0127","_cell_guid":"7788dfe2-782c-4786-99e4-2dbc5a2083a2","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:05:03.665662Z","iopub.execute_input":"2024-10-03T05:05:03.665969Z","iopub.status.idle":"2024-10-03T05:05:03.675539Z","shell.execute_reply.started":"2024-10-03T05:05:03.665935Z","shell.execute_reply":"2024-10-03T05:05:03.674669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(6, 6))\nsns.countplot(data=preview_dataset_df, x=\"MGMT_value\")\nplt.title(\"Distribution of MGMT values\")\nplt.xlabel(\"MGMT Value\")\nplt.xticks([0, 1], [\"Not Present\", \"Present\"])\nplt.show()","metadata":{"_uuid":"cf02cf8f-6ceb-4e45-a61e-6ed420d78166","_cell_guid":"f0249ce2-c619-4dde-a7fb-5439377e9ec1","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:05:03.677015Z","iopub.execute_input":"2024-10-03T05:05:03.67737Z","iopub.status.idle":"2024-10-03T05:05:03.974124Z","shell.execute_reply.started":"2024-10-03T05:05:03.67733Z","shell.execute_reply":"2024-10-03T05:05:03.973247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samp_subm.head(1)","metadata":{"_uuid":"e5d32f7a-462f-4b99-bee3-5997ca8fe212","_cell_guid":"9abe5753-0094-4f8a-b46b-c27c0fd39c15","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:05:03.975141Z","iopub.execute_input":"2024-10-03T05:05:03.975399Z","iopub.status.idle":"2024-10-03T05:05:03.984393Z","shell.execute_reply.started":"2024-10-03T05:05:03.97537Z","shell.execute_reply":"2024-10-03T05:05:03.983402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"first_folder = str(preview_dataset_df.loc[0, 'BraTS21ID']).zfill(5) + \"/\"\ntitle = \"Folders content for all patients\"\n# Folders content\nfolder_path = train_folder_path + first_folder  # Replace train_folder_path with the actual path\nfolder_content = os.listdir(folder_path)\n\ndf = pd.DataFrame({\n        title: folder_content\n    })\ndf.set_index(title, inplace=True)\ndf.head()","metadata":{"_uuid":"16b496d4-fea4-4af0-84c5-5fd26747fbd1","_cell_guid":"25a41fda-4818-4491-b61e-b6841540cf70","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:05:03.987793Z","iopub.execute_input":"2024-10-03T05:05:03.988328Z","iopub.status.idle":"2024-10-03T05:05:04.009131Z","shell.execute_reply.started":"2024-10-03T05:05:03.988297Z","shell.execute_reply":"2024-10-03T05:05:04.008369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"first_folder = str(preview_dataset_df.loc[0, 'BraTS21ID']).zfill(5) + \"/\"\nfolder_path = train_folder_path + first_folder  # Replace train_folder_path with the actual path\n\ntitle = 'Image Type'\nflair_count = len(os.listdir(folder_path + 'FLAIR'))\nt1w_count = len(os.listdir(folder_path + 'T1w'))\nt1wce_count = len(os.listdir(folder_path + 'T1wCE'))\nt2w_count = len(os.listdir(folder_path + 'T2w'))\n\ndata = {\n    title: ['FLAIR', 'T1w', 'T1wCE', 'T2w'],\n    'Count': [flair_count, t1w_count, t1wce_count, t2w_count]\n}\n\n\ndf = pd.DataFrame(data)\ndf.set_index(title, inplace=True)\n\n\ndf.head()","metadata":{"_uuid":"b4582e6f-229c-4e4b-8118-ea10eaef37db","_cell_guid":"76951d5c-e505-4e36-af00-f70a897a26fb","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:05:04.0101Z","iopub.execute_input":"2024-10-03T05:05:04.010527Z","iopub.status.idle":"2024-10-03T05:05:04.401344Z","shell.execute_reply.started":"2024-10-03T05:05:04.010493Z","shell.execute_reply":"2024-10-03T05:05:04.400487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_brain_slices(train_folder_path + \"00000/FLAIR\", split_by = 13)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:05:04.402301Z","iopub.execute_input":"2024-10-03T05:05:04.402595Z","iopub.status.idle":"2024-10-03T05:05:08.142383Z","shell.execute_reply.started":"2024-10-03T05:05:04.402564Z","shell.execute_reply":"2024-10-03T05:05:08.141331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_brain_slices(train_folder_path + \"00000/T1w\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:05:08.143817Z","iopub.execute_input":"2024-10-03T05:05:08.144212Z","iopub.status.idle":"2024-10-03T05:05:12.416351Z","shell.execute_reply.started":"2024-10-03T05:05:08.144168Z","shell.execute_reply":"2024-10-03T05:05:12.41548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_brain_slices(train_folder_path + \"00000/T1wCE\", split_by = 4) ","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:05:12.417529Z","iopub.execute_input":"2024-10-03T05:05:12.417869Z","iopub.status.idle":"2024-10-03T05:05:16.357859Z","shell.execute_reply.started":"2024-10-03T05:05:12.417835Z","shell.execute_reply":"2024-10-03T05:05:16.356902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_brain_slices(train_folder_path + \"00000/T2w\", split_by = 13)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:05:16.359075Z","iopub.execute_input":"2024-10-03T05:05:16.359401Z","iopub.status.idle":"2024-10-03T05:05:20.193164Z","shell.execute_reply.started":"2024-10-03T05:05:16.359366Z","shell.execute_reply":"2024-10-03T05:05:20.192214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_path = \"https://github.com/YanSteph/RSNA-MICCAI-Brain-Tumor-Classification-AI/blob/main/img/scan1.png?raw=true\"\nhtml_code = f'<img src=\"{image_path}\" style=\"width: 700px;\" />'\ndisplay(HTML(html_code))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:05:20.194663Z","iopub.execute_input":"2024-10-03T05:05:20.195012Z","iopub.status.idle":"2024-10-03T05:05:20.201793Z","shell.execute_reply.started":"2024-10-03T05:05:20.194972Z","shell.execute_reply":"2024-10-03T05:05:20.200833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"number_of_extraction = 3\n\n\n# Filtre df mght 1 and 0\npatient_without_mgmt = preview_dataset_df.loc[preview_dataset_df[\"MGMT_value\"] == 0].sort_values('BraTS21ID')\npatient_with_mgmt = preview_dataset_df.loc[preview_dataset_df[\"MGMT_value\"] == 1].sort_values('BraTS21ID')\n\n# Collonne selected\npatient_with_mgmt_folder_ids = patient_with_mgmt[[\"BraTS21ID\", \"MGMT_value\"]].iloc[:number_of_extraction]\npatient_without_mgmt_folder_ids = patient_without_mgmt[[\"BraTS21ID\", \"MGMT_value\"]].iloc[:number_of_extraction]\n\n# map df into tupple\npatient_with_mgmt_mapped_tuples = map_dataframe_to_tuples(patient_with_mgmt_folder_ids, [\"BraTS21ID\", \"MGMT_value\"])\npatient_without_mgmt_mapped_tuples = map_dataframe_to_tuples(patient_without_mgmt_folder_ids, [\"BraTS21ID\", \"MGMT_value\"])\n\n# Merge\npatient_combined_with_and_without_mgmt_ids = list(chain.from_iterable(zip(patient_with_mgmt_mapped_tuples, patient_without_mgmt_mapped_tuples)))\n\n# Show\nshow_brains(patient_combined_with_and_without_mgmt_ids, figsize = (8, 5), dataset_version = dataset_version)","metadata":{"_uuid":"6fcd178a-3d53-43b7-bac0-03d2c70a3730","_cell_guid":"f90ac8c9-9fc4-4b35-bc2b-20e67bed7ef5","_kg_hide-input":true,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2024-10-03T05:05:20.203133Z","iopub.execute_input":"2024-10-03T05:05:20.203803Z","iopub.status.idle":"2024-10-03T05:06:05.064014Z","shell.execute_reply.started":"2024-10-03T05:05:20.203741Z","shell.execute_reply":"2024-10-03T05:06:05.063058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"excluded_patient_ids = [109, 123, 709] \n\npatient_with_wrong_mri_ids = preview_dataset_df.loc[(preview_dataset_df[\"BraTS21ID\"].isin(excluded_patient_ids))].sort_values('BraTS21ID')\npatient_with_wrong_mri_ids_tuples = map_dataframe_to_tuples(patient_with_wrong_mri_ids, [\"BraTS21ID\", \"MGMT_value\"])\n\nshow_brains(patient_with_wrong_mri_ids_tuples, figsize = (8, 5), dataset_version = dataset_version)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:06:05.065438Z","iopub.execute_input":"2024-10-03T05:06:05.065932Z","iopub.status.idle":"2024-10-03T05:06:23.742411Z","shell.execute_reply.started":"2024-10-03T05:06:05.065886Z","shell.execute_reply":"2024-10-03T05:06:23.741462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if skip_brain_segmentation_pytorch_unet:\n    dataset_df = pd.read_csv(dataset_3d_path if dataset_version == DatasetVersion.V_3D else dataset_2d_path)\n    \nelse:\n    loader = widgets.IntProgress(min=0, max=len(preview_dataset_df), description='Loading:')\n    display(loader)\n    \n    # Empty creation of datasets \n    init_dataset_radiomics()\n\n    for i in preview_dataset_df.BraTS21ID :\n        loader.value += 1\n        img_resized = get_processed_image(i, dataset_version)\n        segmentation = segmentation_process(img_resized)\n\n        if dataset_version == DatasetVersion.V_3D:\n            add_patient_data3D(i, img_resized, segmentation)\n        else : \n            add_patient_data2D(i, img_resized, segmentation)\n\n    # Join the 7 datasets\n    df_shapes = df_shapes.set_index('ID')\n    df_textures = df_textures.set_index('ID')\n    df_first_orders_features = df_first_orders_features.set_index('ID')\n\n    df_gzlm_features = df_gzlm_features.set_index('ID').add_prefix('gzlm_')\n    df_glrlm_features = df_glrlm_features.set_index('ID').add_prefix('glrlm_')\n    df = df_shapes.join(df_textures).join(df_first_orders_features)\n    df_ngtdm_features = df_ngtdm_features.set_index('ID').add_prefix('ngtdm_')\n    df_gldm_features = df_gldm_features.set_index('ID').add_prefix('gldm_')\n    \n    df = df_shapes.join(df_textures).join(df_first_orders_features).join(df_gzlm_features).join(df_glrlm_features).join(df_ngtdm_features).join(df_gldm_features)\n\n    # Define 'BraTS21ID' column as integer IDs\n    df['BraTS21ID'] = df['BraTS21ID'].astype(int)\n    \n    # Merge the old dataset with the new one\n    dataset_df = pd.merge(preview_dataset_df, df, left_on='BraTS21ID', right_on='BraTS21ID')\n    dataset_df.rename(columns={'BraTS21ID': 'ID'}, inplace=True)\n\n# Patient BraTS21ID now is ID, and ID of Dataset\ndataset_df = dataset_df.set_index('ID')\n# Drop patient\ndataset_df = dataset_df.drop(index=excluded_patient_ids)\n\nshow_download_link(dataset_df)","metadata":{"_uuid":"3c7d4169-1b44-4576-bdbd-3482098642c6","_cell_guid":"98df2c38-fc59-4868-9f32-dda82317efca","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:06:23.743821Z","iopub.execute_input":"2024-10-03T05:06:23.744217Z","iopub.status.idle":"2024-10-03T05:06:23.922432Z","shell.execute_reply.started":"2024-10-03T05:06:23.744171Z","shell.execute_reply":"2024-10-03T05:06:23.921346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df.head()","metadata":{"_uuid":"5ccde63a-9288-44ea-995f-3d04a83c3fea","_cell_guid":"b32ba107-fb49-45a4-a3bc-6dc4c7ea990d","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:06:23.923504Z","iopub.execute_input":"2024-10-03T05:06:23.923873Z","iopub.status.idle":"2024-10-03T05:06:24.010982Z","shell.execute_reply.started":"2024-10-03T05:06:23.923837Z","shell.execute_reply":"2024-10-03T05:06:24.010102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df.info()","metadata":{"_uuid":"0ad26915-cc1f-4626-a0e7-65d4adac4646","_cell_guid":"7b82d05f-9b77-46f1-bb7c-1318037cf75b","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:06:24.012168Z","iopub.execute_input":"2024-10-03T05:06:24.012483Z","iopub.status.idle":"2024-10-03T05:06:24.030075Z","shell.execute_reply.started":"2024-10-03T05:06:24.012451Z","shell.execute_reply":"2024-10-03T05:06:24.029078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df.describe()","metadata":{"_uuid":"c2528cda-f551-4897-8130-28f51a447494","_cell_guid":"644a88a8-43b1-4faf-88b9-5180fd04fd4d","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:06:24.03119Z","iopub.execute_input":"2024-10-03T05:06:24.031515Z","iopub.status.idle":"2024-10-03T05:06:24.300874Z","shell.execute_reply.started":"2024-10-03T05:06:24.031482Z","shell.execute_reply":"2024-10-03T05:06:24.299995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# First Orders features\nfirst_orders_features_columns_names =['10Percentile', '90Percentile', 'Energy', 'Entropy', 'InterquartileRange', 'Kurtosis', \n                                      'Maximum', 'MeanAbsoluteDeviation', 'Mean', 'Median', 'Minimum', 'Range', 'RobustMeanAbsoluteDeviation', \n                                      'RootMeanSquared', 'Skewness', 'TotalEnergy', 'Uniformity', 'Variance']\n\n# Shape 3D  \nshapes_features_columns_names = ['MeshVolume','VoxelVolume','SurfaceArea','SurfaceVolumeRatio','Sphericity',\n                             'Compactness1','Compactness2','SphericalDisproportion','Maximum3DDiameter','Maximum2DDiameterRow',\n                             'Maximum2DDiameterColumn','MajorAxisLength','MinorAxisLenth','Elongation']\n\n# GLCM features\ntextures_features_columns_names = ['Autocorrelation', 'ClusterProminence', 'ClusterShade', 'ClusterTendency', 'Contrast', \n                           'Correlation', 'DifferenceAverage', 'DifferenceEntropy', 'DifferenceVariance', 'Id', 'Idm', \n                           'Idmn', 'Idn', 'Imc1', 'Imc2', 'InverseVariance', 'JointAverage', 'JointEnergy', 'JointEntropy', \n                           'MCC', 'MaximumProbability', 'SumAverage', 'SumEntropy', 'SumSquares']\n \n# GZLM features\ngzlm_features_columns_names = ['gzlm_SmallAreaEmphasis','gzlm_LargeAreaEmphasis','gzlm_GrayLevelNonUniformity','gzlm_GrayLevelNonUniformityNormalized',\n                                'gzlm_SizeZoneNonUniformity','gzlm_SizeZoneNonUniformityNormalized','gzlm_ZonePercentage','gzlm_GrayLevelVariance','gzlm_ZoneVariance',\n                                'gzlm_ZoneEntropy','gzlm_LowGrayLevelZoneEmphasis','gzlm_HighGrayLevelZoneEmphasis','gzlm_SmallAreaLowGrayLevelEmphasis',\n                                'gzlm_SmallAreaHighGrayLevelEmphasis','gzlm_LargeAreaLowGrayLevelEmphasis','gzlm_LargeAreaHighGrayLevelEmphasis']\n\n# GLRLM\nglrlm_features_columns_names = ['glrlm_ShortRunEmphasis','glrlm_LongRunEmphasis','glrlm_GrayLevelNonUniformity','glrlm_RunLengthNonUniformity','glrlm_RunLengthNonUniformityNormalized'\n                                 ,'glrlm_RunPercentage','glrlm_GrayLevelVariance','glrlm_RunVariance','glrlm_RunEntropy','glrlm_LowGrayLevelRunEmphasis','glrlm_HighGrayLevelRunEmphasis'\n                                 ,'glrlm_ShortRunLowGrayLevelEmphasis','glrlm_ShortRunHighGrayLevelEmphasis','glrlm_LongRunLowGrayLevelEmphasis','glrlm_LongRunHighGrayLevelEmphasis']\n\n# NGTM features\nngtdm_features_columns_names = ['ngtdm_Coarseness','ngtdm_Contrast','ngtdm_Busyness','ngtdm_Complexity','ngtdm_Strength']\n\n    \n# GLDM features\ngldm_features_columns_names = ['gldm_SmallDependenceEmphasis','gldm_LargeDependenceEmphasis','gldm_GrayLevelNonUniformity','gldm_DependenceNonUniformity'\n                               , 'gldm_DependenceNonUniformityNormalized', 'gldm_GrayLevelVariance', 'gldm_DependenceVariance', 'gldm_DependenceEntropy'\n                                , 'gldm_LowGrayLevelEmphasis', 'gldm_HighGrayLevelEmphasis', 'gldm_SmallDependenceLowGrayLevelEmphasis', 'gldm_SmallDependenceHighGrayLevelEmphasis'\n                               , 'gldm_LargeDependenceLowGrayLevelEmphasis', 'gldm_LargeDependenceHighGrayLevelEmphasis']\n\nall_features_column_names = {\n    'first': first_orders_features_columns_names,\n    'shapes': shapes_features_columns_names,\n    'textures': textures_features_columns_names,\n    'gzlm': gzlm_features_columns_names,\n    'glrlm': glrlm_features_columns_names,\n    'ngtdm': ngtdm_features_columns_names,\n    'gldm': gldm_features_columns_names,\n}\n\n# Drop\nexcluded_features_column_names = [\"LeastAxisLength\", \"Flatness\", \"gldm_GrayLevelNonUniformityNormalized\", \"gldm_DependencePercentage\"]\ndataset_df.drop(excluded_features_column_names, axis=1, inplace=True)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:06:24.302344Z","iopub.execute_input":"2024-10-03T05:06:24.302698Z","iopub.status.idle":"2024-10-03T05:06:24.315413Z","shell.execute_reply.started":"2024-10-03T05:06:24.302666Z","shell.execute_reply":"2024-10-03T05:06:24.31446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract features from the DataFrame\nfeatures = dataset_df.drop(\"MGMT_value\", axis=1)\n\n# Calculate the distances between features\ndistances = sch.distance.pdist(features)\n\n# Perform hierarchical clustering and obtain the linkage matrix\nlinkage_matrix = sch.linkage(distances, method='ward')\n\n# Plot the dendrogram\nplt.figure(figsize=(30, 8))\ndendrogram = sch.dendrogram(linkage_matrix, no_labels=True)\n# Adjust the spacing of y-axis tick labels\n\nplt.xlabel('Scans ID')\nplt.ylabel('Distance')\n\nplt.title('Hierarchical Clustering Dendrogram of Scans ID')\n\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:06:24.316719Z","iopub.execute_input":"2024-10-03T05:06:24.317023Z","iopub.status.idle":"2024-10-03T05:06:24.709737Z","shell.execute_reply.started":"2024-10-03T05:06:24.316975Z","shell.execute_reply":"2024-10-03T05:06:24.708841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_clusters = 3\nlabels = sch.fcluster(linkage_matrix, num_clusters, criterion='maxclust')\n\n# Create a dictionary to group features by label\ncluster_three_groups = {}\nfor i, label in enumerate(labels):\n    if label not in cluster_three_groups:\n        cluster_three_groups[label] = []\n    cluster_three_groups[label].append(features.index[i])","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:06:24.71102Z","iopub.execute_input":"2024-10-03T05:06:24.711325Z","iopub.status.idle":"2024-10-03T05:06:24.720941Z","shell.execute_reply.started":"2024-10-03T05:06:24.711292Z","shell.execute_reply":"2024-10-03T05:06:24.719818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def check_and_display_group(key, patients, anomaly_type):\n    if np.isin(patients, cluster_three_groups[key]).any():\n        display(HTML(f\"<h3>Group {key} ({anomaly_type})</h3>\"))\n        patient_with_wrong_mri_ids = preview_dataset_df.loc[preview_dataset_df[\"BraTS21ID\"].isin(patients)]\n        patient_with_wrong_mri_ids_tuples = map_dataframe_to_tuples(patient_with_wrong_mri_ids, [\"BraTS21ID\", \"MGMT_value\"])\n        show_brains(patient_with_wrong_mri_ids_tuples, figsize=(8, 5), dataset_version=dataset_version)\n\nkey = 1\npatients = [111, 456]\ncheck_and_display_group(key, patients, \"ROI with compact shape\")\n\nkey = 2\npatients = [35, 380, 305]\n\ncheck_and_display_group(key, patients, \"ROI with compact and dispersed shape with noisy\")\n\npatients = [351, 442, 589]\ncheck_and_display_group(key, patients, \"Imperfect, anomaly\")\n\nkey = 3\npatients = [810, 483]\ncheck_and_display_group(key, patients, \"ROI with dispersed shape\")\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:06:24.722089Z","iopub.execute_input":"2024-10-03T05:06:24.722406Z","iopub.status.idle":"2024-10-03T05:07:18.192624Z","shell.execute_reply.started":"2024-10-03T05:06:24.722372Z","shell.execute_reply":"2024-10-03T05:07:18.191721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patients_with_wrong_scans_ids = [11, 351, 353, 442, 445, 554, 556, 568, 570, 572, 578, 581, 584, 587, 589, 593, 594, 596, 610]\ndataset_df.drop(index=patients_with_wrong_scans_ids, inplace=True)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:07:18.19384Z","iopub.execute_input":"2024-10-03T05:07:18.194152Z","iopub.status.idle":"2024-10-03T05:07:18.201104Z","shell.execute_reply.started":"2024-10-03T05:07:18.19412Z","shell.execute_reply":"2024-10-03T05:07:18.200326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fun_significant(x):\n    if float(x[:-1]) >= 1:\n        return True\n    elif float(x[:-1]) <= 1:\n        return True\n    else:\n        return False\n    \ndef fun_significant_symbol(x):\n    value = float(x[:-1])\n    if value > 1:\n        return \"⬆️\"\n    elif value < -1:\n        return \"⬇️\"\n    else:\n        return \"-\"\n\n# Select the first row for patients with MGMT_value = 1\npatient_with_mgmt = dataset_df.loc[dataset_df[\"MGMT_value\"] == 1].mean()\n\n# Select the first row for patients with MGMT_value = 0\npatient_without_mgmt = dataset_df.loc[dataset_df[\"MGMT_value\"] == 0].mean()\n\n# Transpose the dataframes\npatient_with_mgmt = patient_with_mgmt.T\npatient_without_mgmt = patient_without_mgmt.T\n\n# Concatenate the transposed dataframes horizontally\nsignificant_values_df = pd.concat([patient_without_mgmt, patient_with_mgmt, patient_without_mgmt - patient_with_mgmt], axis=1)\n\n# Rename the columns\nsignificant_values_df.columns = ['Patient MGMT 0 (Mean)', 'Patient MGMT 1 (Mean)', 'Difference']\n\n# Calculate the percentage difference\nsignificant_values_df['Difference (%)'] = (significant_values_df['Patient MGMT 1 (Mean)'] - significant_values_df['Patient MGMT 0 (Mean)']) / significant_values_df['Patient MGMT 0 (Mean)'] * 100\n\nsignificant_values_df = significant_values_df.sort_values(by=['Difference (%)'], ascending=False)\n\n# Set the format of the 'Difference (%)' column\nsignificant_values_df['Difference (%)'] = np.where((np.isinf(significant_values_df['Difference (%)'])) | (significant_values_df['Difference (%)'].isna()), '0%', significant_values_df['Difference (%)'].apply(lambda x: f\"{x:.2f}%\"))\n\n#Drop MGMT\nsignificant_values_df = significant_values_df.drop(significant_values_df.index[0])\n\n# Add the 'Significant' and 'No Significant' columns\nsignificant_values_df['Significant'] = significant_values_df['Difference (%)'].apply(fun_significant_symbol)\n\n\nsignificant_values_df","metadata":{"_uuid":"c94296ca-9434-4f8a-9db9-c5807654a5fa","_cell_guid":"0d6d20ca-3061-436a-8b21-195682b1e9f4","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:07:18.20919Z","iopub.execute_input":"2024-10-03T05:07:18.209519Z","iopub.status.idle":"2024-10-03T05:07:18.264522Z","shell.execute_reply.started":"2024-10-03T05:07:18.209487Z","shell.execute_reply":"2024-10-03T05:07:18.263598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(HTML(\"<h3>3 first relevant ⬆️</h3>\"))\nfirst_5 = significant_values_df.head(3)\nfor column_name in first_5.index:\n    show_explore_distribution_normality(dataset_df, column_name)    \n\n\ndisplay(HTML(\"<h3>3 last relevant ⬇️</h3>\"))\nlast_5 = significant_values_df.tail(3)\nfor column_name in last_5.index:\n    show_explore_distribution_normality(dataset_df, column_name)  \n    \ndisplay(HTML(\"<h3>Sample Follow normaly law (Correct skewness) ⏺</h3>\"))    \nshow_explore_distribution_normality(dataset_df, \"MeshVolume\")\nshow_explore_distribution_normality(dataset_df, \"SurfaceArea\")\n\ndisplay(HTML(\"<h3>Sample Follow normaly law (Good kurtosis) ⏺</h3>\"))\nshow_explore_distribution_normality(dataset_df, \"Sphericity\")\nshow_explore_distribution_normality(dataset_df, \"Compactness1\")\n\n\ndisplay(HTML(\"<h3>Sample Follow normaly law (Good Q-Q) ⏺</h3>\"))\nshow_explore_distribution_normality(dataset_df, \"Elongation\")\nshow_explore_distribution_normality(dataset_df, \"DifferenceAverage\")\n\n","metadata":{"_uuid":"8355d298-9bce-470c-9e09-6267a5bad61a","_cell_guid":"94d3fc65-ecb3-423a-932b-9a630d304bda","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:07:18.26546Z","iopub.execute_input":"2024-10-03T05:07:18.265719Z","iopub.status.idle":"2024-10-03T05:07:50.922512Z","shell.execute_reply.started":"2024-10-03T05:07:18.265691Z","shell.execute_reply":"2024-10-03T05:07:50.921615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"describe = show_test_normality(dataset_df.drop('MGMT_value',axis=1))\ndisplay(describe)\n\ndescribe_transposed = describe.loc[['skewness','kurtosis','excess_kurtosis']].transpose()\ndescribe_transposed['absolute_value'] = np.abs(describe_transposed['excess_kurtosis']).astype('float')\ntop_50_closest = describe_transposed.nsmallest(50, 'absolute_value').sort_values('absolute_value')\n\ndescribe_transposed = describe.loc[['skewness', 'kurtosis', 'excess_kurtosis']].transpose()\ndescribe_transposed['absolute_value'] = np.abs(describe_transposed['excess_kurtosis']).astype(float)\n\n# Calculer la différence entre la colonne 'skewness' et la valeur idéale de 1\ndescribe_transposed['skewness_diff'] = np.abs(describe_transposed['skewness'] - 1).astype(float)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:07:50.923958Z","iopub.execute_input":"2024-10-03T05:07:50.924365Z","iopub.status.idle":"2024-10-03T05:07:51.328124Z","shell.execute_reply.started":"2024-10-03T05:07:50.924315Z","shell.execute_reply":"2024-10-03T05:07:51.327338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"describe_MGMT_1=show_test_normality(dataset_df[dataset_df.MGMT_value == 1].drop('MGMT_value',axis=1))\ndescribe_MGMT_0=show_test_normality(dataset_df[dataset_df.MGMT_value == 0].drop('MGMT_value',axis=1))\n\nT_describe = describe.drop(['Kurtosis'], axis=1).T\nT_MGMT_1 = describe_MGMT_1.drop(['Kurtosis', 'Maximum'], axis=1).T\nT_MGMT_0 = describe_MGMT_0.drop(['Kurtosis', 'Maximum'], axis=1).T\n\nplt.figure(figsize=(20, 5))\nT_describe[T_describe['kurtosis'] < 10]['kurtosis'].plot(label='Complet', color='blue')\nT_MGMT_1[T_MGMT_1['kurtosis'] < 10]['kurtosis'].plot(label='MGMT_1', color='red', linestyle='-.')\nT_MGMT_0[T_MGMT_0['kurtosis'] < 10]['kurtosis'].plot(label='MGMT_0', color='tab:orange', linestyle='--')\n\nplt.title('Comparison of Kurtosis Values for Different Datasets')\nplt.ylabel('Kurtosis')\nplt.xticks(range(len(T_describe.index)), T_describe.index, rotation=90)  # Afficher toutes les lignes sur l'axe X avec une rotation de 90 degrés\nplt.legend()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:07:51.329278Z","iopub.execute_input":"2024-10-03T05:07:51.329629Z","iopub.status.idle":"2024-10-03T05:07:53.567114Z","shell.execute_reply.started":"2024-10-03T05:07:51.329586Z","shell.execute_reply":"2024-10-03T05:07:53.566195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Instantiate the OutlierDetector\ndetector = OutlierDetector(threshold=2)\n# Call the detect_outliers method with an additional filter\noutliers_mgmt_zeros_df, count = detector.detect_outliers(dataset_df, \"MGMT_value\", filter_column=\"MGMT_value\", filter_value=1)\n\n\nwith_outliers_desc = \"With outliers:\"\nwithout_outliers_desc = \"Without outliers:\"\n\n# Print the count of outliers present\ndisplay(HTML(\"<h3>Outliers with MGMT 1</h3>\"))\n# Print the count of outliers present with descriptions\ncount_str = count.to_string()\noutput_str = f\"{with_outliers_desc} {count[True]}\\n{without_outliers_desc} {count[False]}\"\nprint(output_str)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:07:53.568489Z","iopub.execute_input":"2024-10-03T05:07:53.568885Z","iopub.status.idle":"2024-10-03T05:07:53.628534Z","shell.execute_reply.started":"2024-10-03T05:07:53.568844Z","shell.execute_reply":"2024-10-03T05:07:53.627612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Call the detect_outliers method with an additional filter\noutliers_mgmt_one_df, count = detector.detect_outliers(dataset_df, \"MGMT_value\", filter_column=\"MGMT_value\", filter_value=0)\n\n\ndisplay(HTML(\"<h3>Outliers with MGMT 0</h3>\"))\n# Print the count of outliers present with descriptions\ncount_str = count.to_string()\noutput_str = f\"{with_outliers_desc} {count[True]}\\n{without_outliers_desc} {count[False]}\"\nprint(output_str)","metadata":{"_uuid":"8a44ad53-ee36-4edf-947f-05ce076d7a16","_cell_guid":"7a047191-7bf8-4cf2-893a-694f98bb57e7","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:07:53.629654Z","iopub.execute_input":"2024-10-03T05:07:53.629948Z","iopub.status.idle":"2024-10-03T05:07:53.682205Z","shell.execute_reply.started":"2024-10-03T05:07:53.629916Z","shell.execute_reply":"2024-10-03T05:07:53.681281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Call the detect_outliers method with an additional filter\noutliers_df, count = detector.detect_outliers(dataset_df, \"MGMT_value\")\n\n\ndisplay(HTML(\"<h3>Outliers with MGMT 0 and 1</h3>\"))\n# Print the count of outliers present with descriptions\ncount_str = count.to_string()\noutput_str = f\"{with_outliers_desc} {count[True]}\\n{without_outliers_desc} {count[False]}\"\nprint(output_str)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:07:53.683126Z","iopub.execute_input":"2024-10-03T05:07:53.683406Z","iopub.status.idle":"2024-10-03T05:07:53.735146Z","shell.execute_reply.started":"2024-10-03T05:07:53.683368Z","shell.execute_reply":"2024-10-03T05:07:53.734215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_state=42\nfeature_importance_with_outliers = {}\n\noutlier_feature_names = outliers_df.loc[outliers_df[\"Outliers present\"] == True].index.values\nnon_outlier_feature_names = outliers_df.loc[outliers_df[\"Outliers present\"] == False].index.values\n\nfeatures_and_scaler_tuple_list = [\n    (outlier_feature_names, RobustScaler()),\n    (non_outlier_feature_names, StandardScaler()),\n]\n\n\nmodels = [\n    (\"Random Forest Regressor\", RandomForestRegressor(random_state=random_state), False),\n    (\"Gradient Boosting Regressor\", GradientBoostingRegressor(random_state=random_state), True),\n    (\"AdaBoost Regressor\", AdaBoostRegressor(random_state=random_state), True),\n    (\"XGBoost Regressor\", XGBRegressor(random_state=random_state), True),\n    (\"LightGBM Regressor\", LGBMRegressor(random_state=random_state), True),\n]\n\ncounter_df = pd.DataFrame(outliers_df.index.values, columns=[\"features\"])\n\ncounter_df[\"count\"] = 0\n\ndef check_feature_importance(feature):\n     return feature in selected_features\n\ncv = KFold(n_splits=5)    \n\nresults = select_best_features_RFE(\n        dataset_df, \n        \"MGMT_value\",\n        models, \n        features_and_scaler_tuple_list,\n        reduction_step=0.01,\n        cv=cv\n    )\n    \nfor name, score, selected_features in results:\n    model_name = f\"{name} - {score}\" \n    counter_df[model_name] = counter_df[\"features\"].apply(check_feature_importance)\n    counter_df[\"count\"] += counter_df[model_name].astype(int) \n    \nbest_selected_features_df = counter_df.sort_values(by=\"count\", ascending=False)\ndisplay(HTML(\"<h3>Best features:</h3>\"))\ndisplay(best_selected_features_df)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:07:53.736454Z","iopub.execute_input":"2024-10-03T05:07:53.736844Z","iopub.status.idle":"2024-10-03T05:32:47.742152Z","shell.execute_reply.started":"2024-10-03T05:07:53.736802Z","shell.execute_reply":"2024-10-03T05:32:47.741124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_triangle_correlation_matrix(dataset_df, figsize=(100, 50))","metadata":{"_uuid":"69352ff6-51b4-44a2-81ff-fe3cf02ff276","_cell_guid":"1204c4cd-dcc5-4c94-8574-a3818ac72c38","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:32:47.743768Z","iopub.execute_input":"2024-10-03T05:32:47.744112Z","iopub.status.idle":"2024-10-03T05:33:03.774364Z","shell.execute_reply.started":"2024-10-03T05:32:47.744076Z","shell.execute_reply":"2024-10-03T05:33:03.772851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filtrer by correlation threshold\nshow_triangle_correlation_matrix(dataset_df,filter_threshold = 0.7, figsize=(100, 50))","metadata":{"_uuid":"b22be0c4-edcf-46d5-abd8-503c6fa95531","_cell_guid":"63d56950-1566-4c45-99cb-0fdaae2b4982","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:33:03.775761Z","iopub.execute_input":"2024-10-03T05:33:03.776536Z","iopub.status.idle":"2024-10-03T05:33:10.013343Z","shell.execute_reply.started":"2024-10-03T05:33:03.776498Z","shell.execute_reply":"2024-10-03T05:33:10.012044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_triangle_correlation_matrix(dataset_df,filter_threshold = 0.9, figsize=(100, 50))","metadata":{"_uuid":"4cccb58e-3d39-4662-8ca4-a444931be6f3","_cell_guid":"d55e4a9d-6e91-444c-be13-80d879c324b3","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:33:10.014533Z","iopub.execute_input":"2024-10-03T05:33:10.014856Z","iopub.status.idle":"2024-10-03T05:33:15.550138Z","shell.execute_reply.started":"2024-10-03T05:33:10.014821Z","shell.execute_reply":"2024-10-03T05:33:15.549213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hight_correlated_features_tupples = [\n    ('JointAverage', 'SumAverage', 'MGMT_value'),\n    ('TotalEnergy', 'Energy', 'MGMT_value'),\n    ('MeshVolume', 'VoxelVolume', 'MGMT_value'),\n    ('MajorAxisLength', 'MinorAxisLenth', 'MGMT_value')\n]\n\n\nshow_3D_scatter_plots(\n    hight_correlated_features_tupples, \n    dataset_df, \n    title=\"Samples Correlation Thresholds >0.99 without angles\",\n    figsize=(10, 10), \n    elev_angle=90,\n    azimuth_angle=90, \n    show_legend=True,\n    alpha=0.2\n)\n\nshow_3D_scatter_plots(\n    hight_correlated_features_tupples, \n    dataset_df, \n    figsize=(10, 10), \n    elev_angle=45, \n    azimuth_angle=90, \n    title=\"Samples Correlation Thresholds >0.99\", \n    show_legend=True,\n    legend_loc=\"upper right\",\n    alpha=0.2\n)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:33:15.551493Z","iopub.execute_input":"2024-10-03T05:33:15.551837Z","iopub.status.idle":"2024-10-03T05:33:18.336728Z","shell.execute_reply.started":"2024-10-03T05:33:15.551802Z","shell.execute_reply":"2024-10-03T05:33:18.335793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correlation_thresholds = [0.5, 0.6, 0.7, 0.8, 0.9, 0.99, 1]\ncorrelated_features_table = generate_correlated_features_table(dataset_df, correlation_thresholds)\n\ncorrelated_features_table","metadata":{"_uuid":"5d3fa7f9-e02e-468e-ad16-17359cc55d3e","_cell_guid":"5e774ae9-f7ce-4790-891f-f87907f2552d","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:33:18.338029Z","iopub.execute_input":"2024-10-03T05:33:18.338667Z","iopub.status.idle":"2024-10-03T05:33:20.01951Z","shell.execute_reply.started":"2024-10-03T05:33:18.338617Z","shell.execute_reply":"2024-10-03T05:33:20.018624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feats_list = best_selected_features_df.loc[best_selected_features_df[\"count\"] >= 4][\"features\"].tolist()\nfeats_list = list(dict.fromkeys(feats_list))\n\ndef create_significant_vars_list(df_correlated, significant_values, threshold):\n    df_first_orders_features_columns=['ID','10Percentile', '90Percentile', 'Energy', 'Entropy', 'InterquartileRange', 'Kurtosis',\n                                      'Maximum', 'MeanAbsoluteDeviation', 'Mean', 'Median', 'Minimum', 'Range', 'RobustMeanAbsoluteDeviation',\n                                      'RootMeanSquared', 'Skewness', 'TotalEnergy', 'Uniformity', 'Variance']\n    significant_vars_to_plot = []\n    vars_to_exclude = []\n    groups_correlated_threshold = df_correlated[threshold].tolist()\n    groups_correlated_threshold = filter(lambda x: x!= \"\", groups_correlated_threshold)\n    groups_correlated_threshold = list(groups_correlated_threshold)\n    groups_correlated_threshold = list(map((lambda x : x.split(', ')), groups_correlated_threshold))\n            \n    for var in significant_values :\n        if var not in vars_to_exclude :\n            \n            if var in df_first_orders_features_columns :\n                non_fst_order_found = False\n                \n                for corr_vars in groups_correlated_threshold :\n                    if var in corr_vars :\n                        for i in corr_vars :\n                            if i not in df_first_orders_features_columns and not non_fst_order_found :\n                                significant_vars_to_plot.append(i)\n                                non_fst_order_found = True\n                            else :\n                                if i != var :\n                                    vars_to_exclude.append(i)\n                                    \n                        if non_fst_order_found :\n                            vars_to_exclude.append(var)\n                        else :\n                            significant_vars_to_plot.append(var)\n                \n            else :\n                significant_vars_to_plot.append(var)\n                \n                for corr_vars in groups_correlated_threshold :\n                    if var in corr_vars :\n                        for i in corr_vars :\n                            if i in significant_values :\n                                vars_to_exclude.append(i)\n    \n    return significant_vars_to_plot\n\nsignificant_values = significant_values_df[significant_values_df['Significant'] == \"⬆️\"].index.tolist()\nsignificant_values = list(set(significant_values) & set(feats_list))\n\nsignificant_features_avoid_overfitting_df = create_significant_vars_list(correlated_features_table, significant_values, 0.90)\n\nsignificant_features = pd.DataFrame({\"feature\":significant_features_avoid_overfitting_df})\nsignificant_features","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:33:20.021031Z","iopub.execute_input":"2024-10-03T05:33:20.02153Z","iopub.status.idle":"2024-10-03T05:33:20.043153Z","shell.execute_reply.started":"2024-10-03T05:33:20.021484Z","shell.execute_reply":"2024-10-03T05:33:20.042366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"significant_vars_to_plot_copy = (var for var in significant_features_avoid_overfitting_df if var not in ['ngtdm_Coarseness','glrlm_GrayLevelNonUniformity','Maximum2DDiameterRow','Maximum2DDiameterColumn','gzlm_LargeAreaLowGrayLevelEmphasis','ngtdm_Strength','glrlm_GrayLevelNonUniformity','gzlm_GrayLevelNonUniformity','glrlm_GrayLevelNonUniformityNormalized', 'glrlm_LongRunLowGrayLevelEmphasis'])\nsignificant_vars_to_plot_copy = list(significant_vars_to_plot_copy)\n\ndisplay(HTML(\"<h3>Exploring of Feature Selection and Redundancy Prevention for Avoiding Overfitting:</h3>\"))\n\nfor var in significant_vars_to_plot_copy :\n    fonc_test_normality(dataset_df, var)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:33:20.04448Z","iopub.execute_input":"2024-10-03T05:33:20.044836Z","iopub.status.idle":"2024-10-03T05:33:29.861053Z","shell.execute_reply.started":"2024-10-03T05:33:20.044805Z","shell.execute_reply":"2024-10-03T05:33:29.860104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(HTML(\"<h3>Exploring of Density Comparison:</h3>\"))\n\nsignificant_values_combinations = []\n\nfor combination in combinations(significant_vars_to_plot_copy, 2):\n    significant_values_combinations.append(combination)\n    \n# Select the first 5 combinations\nselected_combinations = significant_values_combinations[:5]\n\nfor combination in selected_combinations :\n    create_kde_mgmt_pos_neg(dataset_df, combination[0], combination[1])","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-03T05:33:29.862215Z","iopub.execute_input":"2024-10-03T05:33:29.862531Z","iopub.status.idle":"2024-10-03T05:33:38.117598Z","shell.execute_reply.started":"2024-10-03T05:33:29.862498Z","shell.execute_reply":"2024-10-03T05:33:38.116684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nfrom sklearn.feature_selection import SelectKBest, f_regression\nfrom sklearn.preprocessing import StandardScaler\nimport numpy as np\n\ntarget = dataset_df[\"MGMT_value\"]\nfeatures = dataset_df.drop(\"MGMT_value\", axis=1)\n\n# Supposons que vous ayez votre jeu de données dans une variable X et les étiquettes correspondantes dans une variable y\nX, X_test, y, y_test = train_test_split(features, target, test_size=0.2, random_state=123)\n# Réduire la dimension en utilisant PCA\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)  # Mise à l'échelle des caractéristiques\n\npca = PCA(n_components=0.9)  # Sélectionner le nombre de composantes qui capturent 95% de la variance\nX_reduced = pca.fit_transform(X_scaled)\n\n# Sélectionner les caractéristiques les plus importantes en utilisant la régression F\nselector = SelectKBest(score_func=f_regression, k=2)  # Sélectionner les 10 meilleures caractéristiques\nX_selected = selector.fit_transform(X_reduced, y)\n\n# Obtenir les indices des caractéristiques sélectionnées\nselected_feature_indices = selector.get_support(indices=True)\n\n# Obtenir les noms des caractéristiques sélectionnées\nselected_feature_names = np.array(list(X.columns))[selected_feature_indices]\n\n# Supprimer les caractéristiques hautement corrélées\ncorrelation_matrix = np.corrcoef(X_selected, rowvar=False)\ncorrelation_mask = np.abs(correlation_matrix) < 0.9  # Définir le seuil de corrélation ici (0.9 dans cet exemple)\ncorrelation_mask = np.all(correlation_mask, axis=0)  # Sélectionner les caractéristiques qui satisfont le seuil de corrélation\n\nX_final = X_selected[:, correlation_mask]\n\n# Utiliser X_final pour entraîner votre modèle\n\nselected_feature_names","metadata":{"execution":{"iopub.status.busy":"2024-10-03T05:33:38.119034Z","iopub.execute_input":"2024-10-03T05:33:38.11969Z","iopub.status.idle":"2024-10-03T05:33:38.178081Z","shell.execute_reply.started":"2024-10-03T05:33:38.119646Z","shell.execute_reply":"2024-10-03T05:33:38.177202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sélectionner les variables avec les valeurs les plus proches de 1 pour 'skewness' et les plus faibles 'absolute_value'\nunivariate_analysis_normality_skewness_kurtosis_tests_top_50_df = describe_transposed.nsmallest(50, ['absolute_value', 'skewness_diff'])\nunivariate_analysis_normality_skewness_kurtosis_tests_top_50_df = univariate_analysis_normality_skewness_kurtosis_tests_top_50_df.drop(['skewness_diff', 'absolute_value'], axis=1)\n\ndisplay(HTML(\"<h3>Top 50 normal distribution:</h3>\"))\ndisplay(univariate_analysis_normality_skewness_kurtosis_tests_top_50_df)","metadata":{"execution":{"iopub.status.busy":"2024-10-03T05:33:38.17928Z","iopub.execute_input":"2024-10-03T05:33:38.1825Z","iopub.status.idle":"2024-10-03T05:33:38.210174Z","shell.execute_reply.started":"2024-10-03T05:33:38.182454Z","shell.execute_reply":"2024-10-03T05:33:38.209308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_copy = dataset_df.copy() ","metadata":{"execution":{"iopub.status.busy":"2024-10-03T05:33:38.213651Z","iopub.execute_input":"2024-10-03T05:33:38.215867Z","iopub.status.idle":"2024-10-03T05:33:38.221351Z","shell.execute_reply.started":"2024-10-03T05:33:38.215825Z","shell.execute_reply":"2024-10-03T05:33:38.220546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_copy.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-03T05:33:38.225075Z","iopub.execute_input":"2024-10-03T05:33:38.227073Z","iopub.status.idle":"2024-10-03T05:33:38.321026Z","shell.execute_reply.started":"2024-10-03T05:33:38.227031Z","shell.execute_reply":"2024-10-03T05:33:38.320089Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import necessary libraries\nfrom sklearn.model_selection import train_test_split, StratifiedKFold, cross_val_score, GridSearchCV\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.svm import OneClassSVM\nfrom sklearn.metrics import classification_report\nimport pandas as pd\nfrom sklearn.preprocessing import FunctionTransformer\nfrom sklearn_pandas import gen_features, DataFrameMapper\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.metrics import accuracy_score\n\nfrom sklearn.preprocessing import RobustScaler\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, RobustScaler","metadata":{"execution":{"iopub.status.busy":"2024-10-03T05:33:38.322207Z","iopub.execute_input":"2024-10-03T05:33:38.322819Z","iopub.status.idle":"2024-10-03T05:33:38.349529Z","shell.execute_reply.started":"2024-10-03T05:33:38.322777Z","shell.execute_reply":"2024-10-03T05:33:38.348888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Contrast\n#DifferenceVariance \n#DifferenceEntropy\n#DifferenceVariance \n#DifferenceEntropy\n#DifferenceAverage\n\n#LargeAreaHighGrayLevelEmphasis\n#LargeDependenceHighGrayLevelEmphasis\n#LargeAreaHighGrayLevelEmphasis\n#Kurtosis \n#Idmn\n\n#Correlation\n#LargeDependenceHighGrayLevelEmphasis\n#SumEntropy\n#HighGrayLevelEmphasis\n#LargeAreaHighGrayLevelEmphasis\n#JointAverage\n\nfeature_outliers = [\n    'Kurtosis',\n    'glrlm_LongRunHighGrayLevelEmphasis',\n    'gzlm_SizeZoneNonUniformityNormalized',\n    'MajorAxisLength',\n    'Imc2',\n    'TotalEnergy',\n    'gzlm_LargeAreaLowGrayLevelEmphasis',\n    'Idn',\n    'gzlm_SmallAreaLowGrayLevelEmphasis',\n    'gzlm_GrayLevelVariance'\n]\n\nfeature_non_outliers = ['InverseVariance',\n    'Range',\n    'MaximumProbability',\n    'Maximum3DDiameter',\n    'Maximum2DDiameterColumn',\n    'Entropy',\n    'Maximum2DDiameterRow']\n\nfeature_to_keep = [\"MGMT_value\"]\n\ndataset_copy = dataset_copy[feature_outliers + feature_non_outliers + feature_to_keep]","metadata":{"execution":{"iopub.status.busy":"2024-10-03T05:33:38.350673Z","iopub.execute_input":"2024-10-03T05:33:38.351276Z","iopub.status.idle":"2024-10-03T05:33:38.358006Z","shell.execute_reply.started":"2024-10-03T05:33:38.351233Z","shell.execute_reply":"2024-10-03T05:33:38.357081Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_copy.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-03T05:33:38.359203Z","iopub.execute_input":"2024-10-03T05:33:38.359533Z","iopub.status.idle":"2024-10-03T05:33:38.387427Z","shell.execute_reply.started":"2024-10-03T05:33:38.3595Z","shell.execute_reply":"2024-10-03T05:33:38.386514Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Slipt train test (Features & Targets)","metadata":{}},{"cell_type":"code","source":"target = dataset_copy[\"MGMT_value\"]\nfeatures = dataset_copy.drop(\"MGMT_value\", axis=1)\n\nX_train, X_test, y_train, y_test = train_test_split(features, target, test_size=0.2, random_state=123)","metadata":{"execution":{"iopub.status.busy":"2024-10-03T05:33:38.388582Z","iopub.execute_input":"2024-10-03T05:33:38.388882Z","iopub.status.idle":"2024-10-03T05:33:38.397723Z","shell.execute_reply.started":"2024-10-03T05:33:38.388851Z","shell.execute_reply":"2024-10-03T05:33:38.396851Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Pipeline","metadata":{}},{"cell_type":"code","source":"pipeline_steps = []","metadata":{"execution":{"iopub.status.busy":"2024-10-03T05:33:38.398821Z","iopub.execute_input":"2024-10-03T05:33:38.399168Z","iopub.status.idle":"2024-10-03T05:33:38.406766Z","shell.execute_reply.started":"2024-10-03T05:33:38.399124Z","shell.execute_reply":"2024-10-03T05:33:38.40595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Preprocessing","metadata":{}},{"cell_type":"code","source":"#-------------------------------#\n# Name\n#-------------------------------#\npreprocessing_name = \"preprocessing\"\n\n#-------------------------------#\n# Map\n#-------------------------------#\n\ntransformers = []\n#-------------------------------#\n# Maps all features into transformers based on their categories\n#-------------------------------#\n#transformers = map_all_features_into_transformers(all_features)\ntransformers.append((f\"ze_a_transformer\", RobustScaler(), feature_outliers))\ntransformers.append((f\"ze_b_transformer\", StandardScaler(), feature_non_outliers))\n\n#-------------------------------#\n# Create ColumnTransformer\n#-------------------------------#\npreprocessor = ColumnTransformer(transformers)\n\n#-------------------------------#\n# Pipeline first step\n#-------------------------------#\npipeline_steps.append((preprocessing_name, preprocessor))\n","metadata":{"execution":{"iopub.status.busy":"2024-10-03T05:33:38.407995Z","iopub.execute_input":"2024-10-03T05:33:38.408794Z","iopub.status.idle":"2024-10-03T05:33:38.416666Z","shell.execute_reply.started":"2024-10-03T05:33:38.408747Z","shell.execute_reply":"2024-10-03T05:33:38.415837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model","metadata":{}},{"cell_type":"code","source":"#-------------------------------#\n# Name\n#-------------------------------#\nmodel_name = \"model\"\n\n#-------------------------------#\n# Model\n#-------------------------------#\n# KNeighborsClassifier\n#from sklearn.neighbors import KNeighborsClassifier\n#model = KNeighborsClassifier()\n\n# DecisionTreeClassifier\n#from sklearn.tree import DecisionTreeClassifier\n#model = DecisionTreeClassifier()\n\n# SVC\n#from sklearn.svm import SVC\n#model = SVC()\n\n# DBSCAN\n#from sklearn.cluster import DBSCAN\n#model = DBSCAN()\n\n# RandomForestClassifier\nfrom sklearn.ensemble import RandomForestClassifier\n#model = RandomForestClassifier()\n\n# XGBoost\n#from xgboost import XGBClassifier\n#model = XGBClassifier()\n\n# LabelSpreading\n#from sklearn.semi_supervised import LabelSpreading\n#model = LabelSpreading()\n\n# LabelSpreading\n#from sklearn.cluster import KMeans\n#model = KMeans()\n\n# AgglomerativeClustering\n#from sklearn.cluster import AgglomerativeClustering\n#model = AgglomerativeClustering()\n\n# BaggingClassifier\nfrom sklearn.ensemble import BaggingClassifier\nmodel = BaggingClassifier(base_estimator=RandomForestClassifier())\n\n#-------------------------------#\n# Pipeline second step\n#-------------------------------#\npipeline_steps.append((model_name, model))","metadata":{"execution":{"iopub.status.busy":"2024-10-03T05:33:38.417688Z","iopub.execute_input":"2024-10-03T05:33:38.417967Z","iopub.status.idle":"2024-10-03T05:33:38.429928Z","shell.execute_reply.started":"2024-10-03T05:33:38.417937Z","shell.execute_reply":"2024-10-03T05:33:38.429129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Pipeline","metadata":{}},{"cell_type":"code","source":"pipeline = Pipeline(steps=pipeline_steps)\npipeline","metadata":{"execution":{"iopub.status.busy":"2024-10-03T05:33:38.431158Z","iopub.execute_input":"2024-10-03T05:33:38.431615Z","iopub.status.idle":"2024-10-03T05:33:38.469624Z","shell.execute_reply.started":"2024-10-03T05:33:38.431572Z","shell.execute_reply":"2024-10-03T05:33:38.468773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Hyperparameter tuning, cross-validation, optimization","metadata":{}},{"cell_type":"code","source":"# Define the parameter grid for GridSearchCV\n#param_grid = {\n#    'model__nu': [0.1, 0.2, 0.3],  # Values to try for the 'nu' parameter\n#    'model__gamma': [0.1, 0.2, 0.3]  # Values to try for the 'gamma' parameter, if applicable\n#}\n\n# KNeighborsClassifier\n#param_grid = {\n#    f'{model_name}__n_neighbors': list(range(1, 200)),  # Valeurs possibles pour le nombre de voisins\n#    f'{model_name}__weights': ['uniform', 'distance']  # Valeurs possibles pour les poids\n#}\n\n# DecisionTreeClassifier\n#param_grid = {\n#    f'{model_name}__criterion': ['gini', 'entropy'],  # Critère pour mesurer la qualité des splits\n#    f'{model_name}__max_depth': [None, 5, 10],  # Profondeur maximale de l'arbre\n#    f'{model_name}__min_samples_split': [2, 5, 10],  # Nombre minimal d'échantillons requis pour effectuer un split\n#    f'{model_name}__min_samples_leaf': [1, 2, 3],  # Nombre minimal d'échantillons requis dans une feuille\n#    f'{model_name}__max_features': ['auto', 'sqrt', 'log2'],  # Nombre maximal de caractéristiques à considérer pour chaque split\n#    f'{model_name}__min_impurity_decrease': [0.0, 0.1, 0.2],  # Seuil minimal pour effectuer un split basé sur l'impureté\n#}\n\n# SVC\n#param_grid = {\n#    f'{model_name}__C': [0.1, 1.0, 10.0],  # Paramètre de régularisation C\n#    f'{model_name}__kernel': ['linear', 'rbf'],  # Noyau du modèle SVC\n#    f'{model_name}__gamma': ['scale', 'auto']  # Paramètre gamma pour les noyaux rbf et poly\n#}\n\n\n#RandomForestClassifier\n#param_grid = {\n#    f\"{model_name}__criterion\": [\"gini\", \"entropy\"],\n#    f\"{model_name}__n_estimators\":  [5, 100, 200, 300],  # Nombre d'estimateurs dans le RandomForestClassifier\n#    f\"{model_name}__max_depth\":  [1, 5, 15, 50],  # Profondeur maximale des arbres dans le RandomForestClassifier\n#}\n\n# XGBClassifier\n#param_grid = {\n#    f\"{model_name}__n_estimators\": [100, 200, 300, 400, 500, 600, 700],  # Nombre d'estimateurs\n#    f\"{model_name}__max_depth\": [3, 4, 5, 6],  # Profondeur maximale des arbres\n#    f\"{model_name}__learning_rate\": [0.3, 0.2, 0.1, 0.01, 0.001]  # Taux d'apprentissage\n#}\n#param_grid = {\n    # Nombre d'estimateurs (nombre d'arbres dans le modèle)\n   # f\"{model_name}__n_estimators\": [50, 100, 200, 300, 350],\n\n    # Profondeur maximale des arbres\n   # f\"{model_name}__max_depth\": [3, 4, 5, 6],\n\n    # Taux d'apprentissage (contrôle la contribution de chaque arbre dans le modèle)\n    #f\"{model_name}__learning_rate\": [0.3, 0.2, 0.1, 0.01, 0.001],\n\n    # Sous-échantillonnage des exemples (contrôle la proportion d'échantillons utilisée pour la construction de chaque arbre)\n   # f\"{model_name}__subsample\": [0.6, 0.8, 1.0],\n\n    # Sous-échantillonnage des colonnes (features) (contrôle la proportion de features utilisée pour la construction de chaque arbre)\n    #f\"{model_name}__colsample_bytree\": [0.6, 0.8, 1.0],\n\n    # Valeur de pénalité pour les coupes dans l'arbre (contrôle la complexité de l'arbre)\n    #f\"{model_name}__gamma\": [0, 0.1, 0.2],\n\n    # Valeur de pénalité L1 pour les poids de l'arbre (contrôle la régularisation L1)\n   # f\"{model_name}__reg_alpha\": [0, 0.1, 0.5],\n\n    # Valeur de pénalité L2 pour les poids de l'arbre (contrôle la régularisation L2)\n    #f\"{model_name}__reg_lambda\": [0, 0.1, 0.5]\n#}\n\n\n# LabelSpreading\n#param_grid = {\n#    f\"{model_name}__kernel\": ['knn', 'rbf'],\n#    f\"{model_name}__gamma\": [0.1, 1.0, 2, 3, 4, 5, 10, 15, 20 , 25, 30],\n#    f\"{model_name}__n_neighbors\": [3, 5, 7, 5, 10, 15, 30, 50, 60, 100],\n#    f\"{model_name}__max_iter\": [40, 50, 100, 200, 300],\n#}\n\n\n# kmeans\n#param_grid = {\n#    f\"{model_name}__n_clusters\": [2],  # Nombre de clusters à essayer\n#    f\"{model_name}__init\": ['k-means++', 'random'],  # Méthode d'initialisation des centroides\n#    f\"{model_name}__max_iter\": [100, 200, 300, 400, 500 , 600]  # Nombre maximum d'itérations\n#}\n\n# agglomerative\n#param_grid = {\n#    f\"{model_name}__n_clusters\": [2],  # Nombre de clusters à essayer\n   # f\"{model_name}__affinity\": ['euclidean', 'l1', 'l2', 'manhattan', 'cosine'],  # Métrique de distance utilisée\n   # f\"{model_name}__linkage\": ['ward', 'complete', 'average', 'single'],  # Méthode de liaison pour former les clusters\n   # 'agglomerative__connectivity': [None, nearest_neighbors_graph],  # Matrice de connectivité pour contraindre les regroupements\n   # 'agglomerative__distance_threshold': [None, 0.5, 1.0, 1.5]  # Seuil de distance pour former les clusters\n#}\n\n\n# f\"{model_name}__n_estimators\": [50, 100, 200],\n# Nombre d'estimateurs dans le BaggingClassifier.\n# Plus le nombre d'estimateurs est élevé, plus le modèle est robuste, mais cela augmente le temps d'entraînement.\n\n# f\"{model_name}__max_samples\": [0.5, 0.8, 1.0],\n# La proportion des échantillons d'entraînement à utiliser pour chaque estimateur.\n# Une valeur inférieure à 1.0 crée des échantillons d'entraînement bootstrap.\n\n# f\"{model_name}__max_features\": [0.5, 0.8, 1.0],\n# La proportion des caractéristiques à utiliser pour chaque estimateur.\n# Une valeur inférieure à 1.0 effectue une sélection aléatoire des caractéristiques.\n\n# f\"{model_name}__base_estimator__max_depth\": [None, 5, 10],\n# La profondeur maximale de chaque arbre de décision de RandomForestClassifier.\n# Une valeur None signifie que les arbres sont développés jusqu'à ce que toutes les feuilles soient pures ou jusqu'à ce que toutes les feuilles contiennent un nombre minimum d'échantillons.\n\n# f\"{model_name}__base_estimator__min_samples_split\": [2, 5, 10],\n# Le nombre minimum d'échantillons requis pour scinder un nœud interne dans chaque arbre de décision.\n# Une valeur plus élevée peut empêcher le surapprentissage en obligeant les arbres à avoir un nombre minimum d'échantillons dans chaque nœud.\n\n# f\"{model_name}__base_estimator__min_samples_leaf\": [1, 2, 4]\n# Le nombre minimum d'échantillons requis dans une feuille d'arbre.\n# Une valeur plus élevée peut également aider à prévenir le surapprentissage en limitant la croissance de l'arbre.\n\n# Define the parameter grid for GridSearchCV\nparam_grid = {\n    f\"{model_name}__n_estimators\": [50, 100, 200],\n    f\"{model_name}__max_samples\": [0.5, 0.8, 1.0],\n    f\"{model_name}__max_features\": [0.5, 0.8, 1.0],\n    f\"{model_name}__base_estimator__max_depth\": [None, 5, 10],\n    f\"{model_name}__base_estimator__min_samples_split\": [2, 5, 10],\n    f\"{model_name}__base_estimator__min_samples_leaf\": [1, 2, 4]\n}","metadata":{"execution":{"iopub.status.busy":"2024-10-03T05:33:38.471211Z","iopub.execute_input":"2024-10-03T05:33:38.47159Z","iopub.status.idle":"2024-10-03T05:33:38.482131Z","shell.execute_reply.started":"2024-10-03T05:33:38.471533Z","shell.execute_reply":"2024-10-03T05:33:38.481274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Search Hyperparame","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import KFold\n\ncross_validator = KFold(n_splits=10)\n\ngrid_search = GridSearchCV(pipeline, param_grid, cv=cross_validator, error_score='raise')","metadata":{"execution":{"iopub.status.busy":"2024-10-03T05:33:38.483276Z","iopub.execute_input":"2024-10-03T05:33:38.483599Z","iopub.status.idle":"2024-10-03T05:33:38.494476Z","shell.execute_reply.started":"2024-10-03T05:33:38.483546Z","shell.execute_reply":"2024-10-03T05:33:38.493615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid_search.fit(X_train , y_train)","metadata":{"execution":{"iopub.status.busy":"2024-10-03T15:11:26.582113Z","iopub.execute_input":"2024-10-03T15:11:26.582483Z","iopub.status.idle":"2024-10-03T15:11:26.605366Z","shell.execute_reply.started":"2024-10-03T15:11:26.582445Z","shell.execute_reply":"2024-10-03T15:11:26.604125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Score","metadata":{}},{"cell_type":"code","source":"# Get the best parameters and best score from GridSearchCV\nbest_params = grid_search.best_params_\nbest_score = grid_search.best_score_\nprint(best_params, \"\\n\")\nprint(best_score, \"\\n\")\n \n# Obtenez les prédictions du meilleur modèle sur X_train\ny_train_pred = grid_search.best_estimator_.predict(X_train)\n# Générez le rapport de classification pour les prédictions sur X_train\nclassification_rep = classification_report(y_train, y_train_pred)\nprint(classification_rep)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Predict\n","metadata":{}},{"cell_type":"code","source":"# Evaluate the predictions\naccuracy = grid_search.score(X_test, y_test)\n# Affichez le score d'exactitude\nprint(\"Accuracy Score on Test Set:\", accuracy, \"\\n\")\n\n# Use the best model for predictions\nbest_model = grid_search.best_estimator_\npredictions = best_model.predict(X_test)\n\n# Evaluate the predictions\nprint(classification_report(y_test, predictions))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"----","metadata":{}},{"cell_type":"code","source":"import itertools\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.optimizers import Adamax\nfrom sklearn.metrics import accuracy_score, confusion_matrix\nfrom sklearn.preprocessing import OneHotEncoder\nfrom imblearn.over_sampling import RandomOverSampler\nfrom sklearn.model_selection import KFold\nfrom keras.utils import to_categorical\n\n'''\ndef build_model(layer_size, activation, dropout):\n    model = Sequential()\n    model.add(Dense(layer_size, activation=activation, input_shape=(input_dim,)))\n    model.add(Dropout(dropout))\n    model.add(Dense(num_classes, activation='softmax'))\n    return model\n'''\n\ndef build_model(layer_size, activation, dropout):\n    model = Sequential()\n    model.add(Dense(layer_size, activation=activation, input_shape=(input_dim,)))\n    model.add(Dropout(dropout))\n    model.add(BatchNormalization())\n    model.add(Dense(layer_size, activation=activation))\n    model.add(Dropout(dropout))\n    model.add(BatchNormalization())\n    #model.add(Dense(num_classes, activation='softmax'))\n    model.add(Dense(num_classes, activation='sigmoid'))\n    return model\n\nnum_classes = 2\n\ndataset_copy = dataset_df.copy() \n\nfeature_to_keep = [\"MGMT_value\"]\n\nfeature_outliers = ['glrlm_LongRunHighGrayLevelEmphasis','MajorAxisLength','gzlm_SizeZoneNonUniformityNormalized','Kurtosis','Idn',\n'Idmn','gldm_LargeDependenceHighGrayLevelEmphasis','Mean','MinorAxisLenth','glrlm_RunPercentage']\nfeature_non_outliers = ['InverseVariance','Maximum3DDiameter']\n\ndataset_copy = dataset_copy[feature_outliers + feature_non_outliers + feature_to_keep]\n\ntarget = dataset_copy[\"MGMT_value\"]\nfeatures = dataset_copy.drop(\"MGMT_value\", axis=1)\n\n# Encodage One-Hot des étiquettes\nonehot_encoder = OneHotEncoder(sparse=False)\ntarget_encoded = onehot_encoder.fit_transform(target.values.reshape(-1, 1))\n\n# Convertir le codage one-hot en classe binaire\ny_binary = np.argmax(target_encoded, axis=1)\n\n# Compter les occurrences de chaque classe dans y_binary\nclass_counts = np.bincount(y_binary)\n\n# Afficher les répartitions des classes\nfor cls, count in enumerate(class_counts):\n    print(f\"Classe {cls}: {count} échantillons\")\n\n# Diviser les données en ensembles d'entraînement et de test\nX_train, X_test, y_train, y_test = train_test_split(\n    features, y_binary, test_size=0.2, random_state=42\n)\n\n\n# Création d'une instance du sur-échantillonneur RandomOverSampler\noversampler = RandomOverSampler(random_state=42)\n\n# Appliquer le sur-échantillonnage sur vos données\nX_train, y_train = oversampler.fit_resample(X_train, y_train)\n\n# Normaliser les données d'entraînement et de test\nscaler = StandardScaler()\nX_train = scaler.fit_transform(X_train)\nX_test = scaler.transform(X_test)\n\n# Dimension d'entrée du modèle\ninput_dim = X_train.shape[1]\n\n# Liste des paramètres à tester\n# Meilleurs paramètres : (512, 'relu', 0.2, 256, 0.01)\n# Meilleure précision : 0.598313745856285\n\n# AUTRE\n#Accuracy: 0.5555555555555556 | Parameters: 512, relu, 1e-05, 512, 0.001\n#Best Parameters: {'layer_size': 512, 'activation': 'relu', 'dropout': 1e-05, 'batch_size': 512, 'learning_rate': 0.001}\n#Confusion Matrix:\n#array([[ 3, 49],\n#       [ 3, 62]])\n\n#Meilleurs paramètres : (512, 'relu', 1e-05, 256, 0.001)\n#Meilleure précision : 0.5630980551242828\n#Confusion Matrix:\n#array([[122, 132],\n#       [138, 116]])\n'''\nlayer_sizes = [512]\nactivations = ['relu']\ndropouts = [0.2]\nbatch_sizes = [256]\nlearning_rates = [0.01]\n'''\n'''\nlayer_sizes = [512]\nactivations = ['relu']\ndropouts = [1e-05]\nbatch_sizes = [512]\nlearning_rates = [0.001]\n'''\n\nlayer_sizes = [64]#, 128, 256, 512]\nactivations = ['relu','sigmoid']#,'leakyrelu']\ndropouts = [1e-05, 0.0001,0.001]#, 0.01, 0.1, 0.2,0.5,0.8]\nbatch_sizes = [16,32,64]#,128,256,512]\nlearning_rates = [0.0001,0.001,0.01]#, 0.1,0.2]\n\nepochs = 10\n\n# Effectuer une recherche par grille\nbest_accuracy = 0.0\nbest_params = {}\nconfusion_matrix_best = None\n\nkf = KFold(n_splits=10, shuffle=True)  # Utilisation de la validation croisée avec 10 plis\n\nfor params in itertools.product(layer_sizes, activations, dropouts, batch_sizes, learning_rates):\n    layer_size, activation, dropout, batch_size, learning_rate = params\n\n    accuracies = []  # Stocker les précisions pour chaque pli\n    y_pred_fold = []  # Stocker les prédictions pour chaque pli\n\n\n    for train_index, val_index in kf.split(X_train):\n        X_train_fold, X_val_fold = X_train[train_index], X_train[val_index]\n        y_train_fold, y_val_fold = y_train[train_index], y_train[val_index]\n\n        # Convertir les étiquettes en encodage one-hot\n        y_train_fold = to_categorical(y_train_fold)\n        y_val_fold = to_categorical(y_val_fold)\n        \n        # Construire et entraîner le modèle avec les paramètres actuels\n        model = build_model(layer_size, activation, dropout)\n        model.compile(optimizer=Adam(learning_rate=learning_rate), loss='categorical_crossentropy', metrics=['accuracy'])\n        #model.compile(optimizer=Adamax(learning_rate=learning_rate), loss='categorical_crossentropy', metrics=['accuracy'])\n        \n        history = model.fit(X_train_fold, y_train_fold, batch_size=batch_size, epochs=epochs, validation_data=(X_val_fold, y_val_fold), verbose=0)\n\n        # Évaluer les performances du modèle sur l'ensemble de validation actuel\n        accuracy = model.evaluate(X_val_fold, y_val_fold)[1]\n        accuracies.append(accuracy)\n        \n        # Obtenir les prédictions sur l'ensemble de validation actuel\n        y_pred = model.predict(X_val_fold)\n        y_pred = np.argmax(y_pred, axis=1)\n        y_pred_fold.extend(y_pred)\n\n    # Calculer la moyenne des précisions sur tous les plis\n    mean_accuracy = sum(accuracies) / len(accuracies)\n\n    # Vérifier si les performances actuelles sont les meilleures\n    if mean_accuracy > best_accuracy:\n        best_accuracy = mean_accuracy\n        best_params = params\n        y_pred_best = np.array(y_pred_fold)\n        confusion_matrix_best = confusion_matrix(y_train, y_pred_best)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Afficher les meilleurs paramètres et la meilleure précision\nprint(\"Meilleurs paramètres :\", best_params)\nprint(\"Meilleure précision :\", best_accuracy)\n\nlayer_size, activation, dropout, batch_size, learning_rate = best_params\n# Construire et entraîner le modèle avec les paramètres actuels\nmodel = build_model(layer_size, activation, dropout)\nmodel.compile(optimizer=Adam(learning_rate=learning_rate), loss='categorical_crossentropy', metrics=['accuracy'])\nmodel.fit(X_train, to_categorical(y_train), batch_size=batch_size, epochs=100)\n\ny_pred = np.argmax(model.predict(X_test), axis=1)\nprint(\"Confusion Matrix:\")\ndisplay(confusion_matrix(y_test, y_pred))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Juillet 2023 DAVID : Autre tentatives d'approches sur modèles ML\n----","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport itertools\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom matplotlib import offsetbox\nfrom mpl_toolkits import mplot3d\n\nfrom sklearn.model_selection import train_test_split, KFold, cross_val_score, GridSearchCV, StratifiedKFold, RandomizedSearchCV\nfrom sklearn.preprocessing import StandardScaler, RobustScaler, MinMaxScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.manifold import LocallyLinearEmbedding, MDS, Isomap, TSNE\nfrom sklearn.decomposition import PCA\nfrom sklearn.metrics import confusion_matrix, classification_report, make_scorer, accuracy_score, f1_score\nfrom sklearn.metrics import roc_curve, roc_auc_score\nfrom sklearn.metrics import make_scorer, precision_score, recall_score\n\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.feature_selection import RFECV, SelectKBest, f_classif\nfrom imblearn.over_sampling import SMOTE\nfrom sklearn.calibration import CalibratedClassifierCV\nfrom sklearn.svm import SVC\nfrom xgboost.core import DMatrix\nimport xgboost as xgb\n\ndataset_copy = dataset_df.copy() \nexcluded_features_column_names = [\"LeastAxisLength\", \"Flatness\", \"gldm_GrayLevelNonUniformityNormalized\", \"gldm_DependencePercentage\"]\ndataset_df.drop(excluded_features_column_names, axis=1, inplace=True)\n\n# Patient BraTS21ID now is ID, and ID of Dataset\ndataset_copy = dataset_copy.set_index('ID')\n# Drop patient\nexcluded_patient_ids = [109, 123, 709] #, 11, 351, 353, 442, 445, 554, 556, 568, 570, 572, 578, 581, 584, 587, 589, 593, 594, 596, 610]\ndataset_copy = dataset_copy.drop(index=excluded_patient_ids)\n\nX = dataset_copy.drop(\"MGMT_value\",axis=1)\ny = dataset_copy['MGMT_value']\n\n\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"MODELE 1","metadata":{}},{"cell_type":"code","source":"pca = PCA()\npca.fit(X)\nplt.bar(range(len(pca.explained_variance_ratio_)), pca.explained_variance_ratio_)\nplt.xlabel('Composante principale')\nplt.ylabel('Ratio de variance expliquée')\nplt.show()\n\n#display(pca.n_components_)\ndisplay(pca.score(X))\npca = PCA(n_components=10)\n\nX = pca.fit_transform(X)\n\n# Division en ensembles de train et de test\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Utilisation de StratifiedKFold pour préserver les proportions des classes lors de la validation croisée\ncv = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n'''Utilisation de la classe StratifiedKFold pour effectuer une validation croisée plus fine \ntout en préservant les proportions des classes lors de la division des données. \nUtilisation de GridSearchCV avec une grille de paramètres plus spécifique pour trouver les \nmeilleurs hyperparamètres du modèle RandomForest. Ajout d'une étape de calibrage du modèle \navec CalibratedClassifierCV pour ajuster les seuils de décision et améliorer la performance \nde classification. Ce code ajoute la sélection des caractéristiques en utilisant SelectKBest \navec le test F pour choisir les meilleures caractéristiques. Les caractéristiques fortement \ncorrélées sont identifiées et exclues avant la sélection des meilleures caractéristiques. \nSMOTE pour augmenter artificiellement le jeu de données SelectKBest pour sélection des \nvariables \"pertinentes\"'''\n\n'''\n# Augmentation artificielle des données avec SMOTE\nsmote = SMOTE(random_state=42)\nX_train, y_train = smote.fit_resample(X_train, y_train)\n\n\n# Prétraitement des données : Standardisation des caractéristiques\n\nscaler = StandardScaler()\nX_train = scaler.fit_transform(X_train)\nX_test = scaler.transform(X_test)\n\n# Sélection des meilleures caractéristiques avec SelectKBest\nk = get_k(selected_features)\nselector = SelectKBest(score_func=f_classif, k=k)  # Définissez le nombre de caractéristiques à sélectionner\n\nX_train = selector.fit_transform(X_train, y_train)\nX_test = selector.transform(X_test)\n\n# Affichage des caractéristiques sélectionnées\nprint(\"Features sélectionnées :\")\nselected_features = [feature for feature, selected in zip(X.columns, selector.get_support()) if selected]\nprint(selected_features)\n'''\n\n# Définition des hyperparamètres à optimiser avec une validation croisée plus fine\n'''\nparam_grid = {\n  'kernel':['rbf'],#['linear','poly','rbf','sigmoid'],\n  'gamma': [0, 0.1, 0.2],\n}\n'''\n'''\n# RandomForestClassifier\nparam_grid = {\n    'n_estimators': [100], #[100,200,300,400,500],\n    'max_depth': [10], #,50,100,200],\n}\n'''\n# the best\nparam_grid = {\n 'max_depth': [20], #[10,20,30,50],\n 'max_features': ['sqrt'],\n 'min_samples_leaf': [1],#[1,2,5,10],\n 'min_samples_split': [5],#[1,2,5,10],\n 'n_estimators': [100]#[100,200,300]\n}\n\n'''\nparam_grid = {\n  'n_estimators': [100],\n  'learning_rate' : [0.1],\n  'max_depth':[2]\n}\n#'''\n'''\n# XGBoost\nparam_grid = {\n    'n_estimators': [100], #, 200, 300],\n    'learning_rate': [0.1, 0.3, 1.0],#0.01, 0.001],\n    'max_depth': [3, 5, 6, 7],\n    'min_child_weight': [1, 3, 5],\n    'subsample': [0.5, 0.8, 0.9, 1.0],\n    'colsample_bytree': [0.5, 0.8, 0.9, 1.0],\n    'gamma': [0, 0.1, 0.2],\n    'reg_alpha': [0, 0.1, 0.2],\n    'reg_lambda': [0, 0.1, 0.2],\n    #'objective': ['binary_logistic'],\n    'criterion': ['friedman_mse', 'squared_error'],\n    'min_weight_fraction_leaf': [0, 0.5],\n    'random_state' : [42]\n}\n#'''\n'''\nparam_grid = {\n 'colsample_bytree': [0.8],\n 'gamma': [0],\n 'learning_rate': [0.1],\n 'max_depth': [5],\n 'min_child_weight': [1],\n 'n_estimators': [100],\n 'reg_alpha': [0.2],\n 'reg_lambda': [0],\n 'subsample': [0.8]}\n#'''\n\n# Sélection récursive de variables avec validation croisée\nrf = RandomForestClassifier(random_state=42)\n#rf = xgb.XGBClassifier(random_state=42)\n#rf = GradientBoostingClassifier()\n#rf = SVC()\n# Optimisation des hyperparamètres avec validation croisée plus fine\ngrid_search = GridSearchCV(rf, param_grid, cv=cv, scoring='accuracy', error_score='raise')\n#grid_search = RandomizedSearchCV(rf, param_distributions=param_grid, n_iter=10, cv=cv)\n\ngrid_search.fit(X_train, y_train)\n\n# Meilleurs paramètres trouvés\nbest_params = grid_search.best_params_\ndisplay(\"Best params : \", best_params)\nbest_rf= grid_search.best_estimator_\n# Entraînement du modèle avec les meilleurs paramètres\n#best_rf = RandomForestClassifier(random_state=42, **best_params)\n#best_rf = xgb.XGBClassifier(random_state=42, **best_params)\nbest_rf.fit(X_train, y_train)\ny_pred = best_rf.predict(X_test)\n'''\n# Conversion des données d'entraînement en objet DMatrix\ndtrain = xgb.DMatrix(X_train, label=y_train)\n# Conversion des données de test en objet DMatrix\ndtest = xgb.DMatrix(X_test)\nmodel = xgb.train(best_params, dtrain)\ny_pred = model.predict(dtest)\n\n# Convertir les prédictions en classes binaires\ny_pred = (y_pred >= 0.5).astype(int)\n'''\n\n# ===========================\n# Resultats\n# ===========================\ndisplay('Accuracy train :', accuracy_score(y_train, best_rf.predict(X_train)))\n\n# Calcul de l'accuracy\naccuracy = accuracy_score(y_test, y_pred)\ndisplay('Accuracy : ', accuracy)\n\n# Calcul du F1-score\nf1 = f1_score(y_test, y_pred)\n\n# Calcul de la matrice de confusion\nconfusion = confusion_matrix(y_test, y_pred)\n\nreport = classification_report(y_test, y_pred)\nprint(report)\n\n\n# Création de la figure\nplt.figure(figsize=(8, 6))\n\n# Affichage de la matrice de confusion avec heatmap\nsns.heatmap(confusion, annot=True, fmt='d', cmap='Blues')\n\n# Ajout des labels des axes\nplt.xlabel('Prédictions')\nplt.ylabel('Vraies étiquettes')\nplt.title('Matrice de confusion')\n\n# Affichage de la figure\nplt.show()\n\n# Calcul de la courbe ROC\nfpr, tpr, thresholds = roc_curve(y_test, y_pred)\n\n# Calcul de l'aire sous la courbe ROC\nauc = roc_auc_score(y_test, y_pred)\n\n# Affichage de la courbe ROC\nplt.plot(fpr, tpr, label='Courbe ROC (AUC = {:.2f})'.format(auc))\nplt.plot([0, 1], [0, 1], 'k--')  # Diagonale aléatoire\nplt.xlabel('Taux de faux positifs')\nplt.ylabel('Taux de vrais positifs')\nplt.title('Courbe ROC')\nplt.legend(loc='lower right')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"MODELE 2\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import GridSearchCV, train_test_split\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import RobustScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\n\n# Créer un pipeline de prétraitement\npreprocessor = Pipeline([\n    ('scaler', RobustScaler())\n])\n\n# Définir les métriques d'évaluation pour optimiser le réglage du seuil\nscoring = {\n    'Precision': make_scorer(precision_score),\n    'Recall': make_scorer(recall_score),\n    'F1': make_scorer(f1_score)\n}\n\n\n# Créer un dictionnaire avec les modèles à tester et leurs grilles de paramètres\nmodels = {\n    'Logistic Regression': {\n        'model': LogisticRegression(),\n        'params': {\n            'model__C': [0.001,0.01,0.1, 1.0, 10.0],\n            'model__penalty': [None, 'l2'],\n            ##'random_state' : [42],\n            ##'model__threshold': [0.3, 0.4, 0.5, 0.6, 0.7]\n        }\n    },\n'Support Vector Machine': {\n    'model': SVC(probability=True),\n    'params': {\n        'model__C': [0.0001, 0.001, 0.01, 0.1, 1.0, 10.0, 20.0, 50.0],\n        'model__kernel': ['linear', 'rbf','poly','sigmoid'],\n        'model__gamma': ['scale', 'auto'],\n        'model__degree': [2, 3, 4],\n        'model__shrinking': [True, False],\n        #'random_state' : [42],\n        #'model__threshold': [0.3, 0.4, 0.5, 0.6, 0.7]\n    }\n},\n    'Random Forest': {\n        'model': RandomForestClassifier(),\n        'params': {\n            'model__n_estimators': [100, 200],\n            'model__max_depth': [3, 5, 7,20],\n            'model__min_samples_split': [2, 5, 10],\n            #'random_state' : [42],\n            #'model__threshold': [0.3, 0.4, 0.5, 0.6, 0.7]\n        }\n    },\n\n  # XGBoost\n   'XGBoost' : {\n      'model' : xgb.XGBClassifier(),\n      'params' : {\n        'n_estimators': [100, 200, 300],\n        #'learning_rate': ['memory', 'steps', 'verbose'], #[0.1, 0.3, 1.0, 0.01, 0.001],\n        'max_depth': [3, 5, 6, 7],\n\n      }\n  }\n}\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Effectuer la recherche des meilleurs hyperparamètres pour chaque modèle\nfor model_name, model_info in models.items():\n    print(\"Entraînement du modèle:\", model_name)\n\n    # Définir le modèle du pipeline\n    model_info['model'] = Pipeline([\n        ('preprocessor', preprocessor),\n        ('model', model_info['model'])\n    ])\n\n    # Effectuer la recherche des hyperparamètres optimaux\n    grid_search = GridSearchCV(model_info['model'], model_info['params'], cv=cv, error_score='raise', scoring=scoring, refit='F1')\n    grid_search.fit(X_train, y_train)\n\n    # Évaluer le modèle sur l'ensemble de test\n    y_pred = grid_search.predict(X_test)\n\n\n    # Afficher les meilleurs paramètres et scores\n    print(\"Meilleurs paramètres : \", grid_search.best_params_)\n    print(\"Meilleur score F1 : \", grid_search.best_score_)\n\n    # Utiliser le modèle avec les meilleurs paramètres sur les données de test\n    best_model = grid_search.best_estimator_\n    y_pred_proba = best_model.predict_proba(X_test)[:, 1]\n\n    # Appliquer un seuil manuellement pour la prédiction\n    threshold = 0.5  # Seuil par défaut\n    y_pred = (y_pred_proba >= threshold).astype(int)\n\n    accuracy = accuracy_score(y_test, y_pred)\n    precision = precision_score(y_test, y_pred)\n    recall = recall_score(y_test, y_pred)\n    f1 = f1_score(y_test, y_pred)\n    print(\"Accuracy train : \", accuracy_score(y_train, grid_search.predict(X_train)))\n    # Afficher les résultats\n    print(\"Meilleurs hyperparamètres:\", grid_search.best_params_)\n    print(\"Accuracy test:\", accuracy)\n    print(\"Précision:\", precision)\n    print(\"Rappel:\", recall)\n    print(\"Score F1:\", f1)\n    print(\"-------------------------------------\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"MODELE 3 : DL","metadata":{}},{"cell_type":"code","source":"def build_model_conv(layer_size, activation, dropout, input_shape):\n    model = Sequential()\n    model.add(Conv1D(layer_size, kernel_size=3, activation=activation, input_shape=input_shape))\n    model.add(MaxPooling1D(pool_size=2))\n    model.add(Conv1D(layer_size * 2, kernel_size=3, activation='relu'))\n    model.add(MaxPooling1D(pool_size=2))\n    model.add(Flatten())\n    model.add(Dense(layer_size, activation=activation))\n    model.add(Dropout(dropout))\n    model.add(Dense(1, activation='sigmoid'))\n    return model\n\ndef build_model(layer_size, activation, dropout, input_shape) :\n    # Définition de l'entrée du modèle\n    inputs = Input(shape=input_shape)\n\n    # Couches du modèle\n    x = Dense(layer_size)(inputs)\n    x = BatchNormalization()(x)\n    x = Activation(activation)(x)\n\n    x = Dense(layer_size*2)(x)\n    x = BatchNormalization()(x)\n    x = Activation(activation)(x)\n\n    x = Dense(layer_size*2)(x)\n    x = BatchNormalization()(x)\n    x = Activation(activation)(x)\n\n    # Connexion résiduelle\n    residual = Dense(layer_size*2)(inputs)\n    residual = BatchNormalization()(residual)\n\n    # Ajout de la connexion résiduelle\n    x = Add()([x, residual])\n    x = Activation('relu')(x)\n\n    # Couche de sortie\n    outputs = Dense(1, activation='sigmoid')(x)\n\n    # Création du modèle\n    model = Model(inputs=inputs, outputs=outputs)\n\n    return model\n\n# Paramètres du modèle\nepoch = 150\n\nlayer_sizes = [128]\nactivations = ['relu']#,'leakyrelu']\ndropouts = [0.1]#, 0.01, 0.1, 0.2,0.5,0.8]\nbatch_sizes = [128]#,128,256,512]\nlearning_rates = [0.1]#, 0.1,0.2]\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\nscaler = RobustScaler()\nX_train = scaler.fit_transform(X_train)\nX_test = scaler.transform(X_test)\n\ny_train = y_train.reset_index(drop=True)\n\nbest_accuracy = 0.0\nbest_params = {}\n\n\n\nkf = KFold(n_splits=10, shuffle=True)\n\nfor params in itertools.product(layer_sizes, activations, dropouts, batch_sizes, learning_rates):\n    layer_size, activation, dropout, batch_size, learning_rate = params\n    accuracies = []\n\n    for train_index, val_index in kf.split(X_train):\n        X_train_fold, X_val_fold = X_train[train_index], X_train[val_index]\n        y_train_fold, y_val_fold = y_train[train_index], y_train[val_index]\n\n        input_shape = (X_train_fold.shape[1], 1)\n        #model = build_model_conv(layer_size, activation, dropout, input_shape)\n        model = build_model(layer_size, activation, dropout, input_shape)\n        model.compile(optimizer=Adam(learning_rate=learning_rate), loss='binary_crossentropy', metrics=['accuracy'])\n        model.fit(X_train_fold, y_train_fold, batch_size=batch_size, epochs=epoch, verbose=0)\n        accuracy = model.evaluate(X_val_fold, y_val_fold)[1]\n        accuracies.append(accuracy)\n\n    mean_accuracy = sum(accuracies) / len(accuracies)\n\n    if mean_accuracy > best_accuracy:\n        best_accuracy = mean_accuracy\n        best_params = params\n        display(best_accuracy, best_params)\n\n\n\nlayer_size, activation, dropout, batch_size, learning_rate = best_params\n\ninput_shape = (X_train.shape[1], 1)\nbest_model = build_model_conv(layer_size, activation, dropout, input_shape)\nbest_model.compile(optimizer=Adam(learning_rate=learning_rate), loss='binary_crossentropy', metrics=['accuracy'])\n\n# Créer une instance du callback History\nhistory = History()\n\nbest_model.fit(X_train, y_train, batch_size=batch_size, epochs=epoch, verbose=1, callbacks=[history])\ntest_loss, test_accuracy = best_model.evaluate(X_test, y_test)\n\n# Afficher les résultats des epochs\nloss = history.history['loss']\naccuracy = history.history['accuracy']\n\n# Afficher les résultats\nepochs = range(1, len(loss) + 1)\nplt.plot(epochs, loss, 'b', label='Training loss')\nplt.plot(epochs, accuracy, 'r', label='Training accuracy')\nplt.title('Training Loss and Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Loss / Accuracy')\nplt.legend()\nplt.show()\n\n\n\ny_pred_prob = best_model.predict(X_test)\n# Restreindre les valeurs entre 0 et 1\ny_pred_prob = np.clip(y_pred_prob, 0, 1)\n\n# Convertir les valeurs en entiers (0 ou 1)\ny_pred = (y_pred_prob > 0.5).astype(int)\n\n# Vérifier les valeurs uniques dans y_test et y_pred\nunique_values_test = np.unique(y_test)\nunique_values_pred = np.unique(y_pred)\n\nprint(\"Unique values in y_test:\", unique_values_test)\nprint(\"Unique values in y_pred:\", unique_values_pred)\n\ncm = confusion_matrix(y_test, y_pred)\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues')\nplt.xlabel('Predicted')\nplt.ylabel('Actual')\nplt.show()\n\ndisplay(best_accuracy, best_params)\n\n\nreport = classification_report(y_test, y_pred)\nprint(report)\n\n\n# Création de la figure\nplt.figure(figsize=(8, 6))\n\n# Affichage de la matrice de confusion avec heatmap\nsns.heatmap(confusion, annot=True, fmt='d', cmap='Blues')\n\n# Ajout des labels des axes\nplt.xlabel('Prédictions')\nplt.ylabel('Vraies étiquettes')\nplt.title('Matrice de confusion')\n\n# Affichage de la figure\nplt.show()\n\n\n\n# Calcul de la courbe ROC\nfpr, tpr, thresholds = roc_curve(y_test, y_pred)\n\n# Calcul de l'aire sous la courbe ROC\nauc = roc_auc_score(y_test, y_pred)\n\n# Affichage de la courbe ROC\nplt.plot(fpr, tpr, label='Courbe ROC (AUC = {:.2f})'.format(auc))\nplt.plot([0, 1], [0, 1], 'k--')  # Diagonale aléatoire\nplt.xlabel('Taux de faux positifs')\nplt.ylabel('Taux de vrais positifs')\nplt.title('Courbe ROC')\nplt.legend(loc='lower right')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}