{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import imageio\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport random\nimport pickle\nfrom scipy.ndimage import convolve, gaussian_filter\nfrom PIL import Image, ImageEnhance","metadata":{"execution":{"iopub.status.busy":"2023-01-23T18:10:51.432840Z","iopub.execute_input":"2023-01-23T18:10:51.433633Z","iopub.status.idle":"2023-01-23T18:10:51.442690Z","shell.execute_reply.started":"2023-01-23T18:10:51.433560Z","shell.execute_reply":"2023-01-23T18:10:51.440808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport os\nimport torch\nfrom torch import nn\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision\nfrom torchvision import transforms as T\nfrom tqdm import tqdm\nimport tqdm.notebook as tq\nfrom PIL import Image\nimport random\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import classification_report","metadata":{"execution":{"iopub.status.busy":"2023-01-23T18:10:51.445677Z","iopub.execute_input":"2023-01-23T18:10:51.447276Z","iopub.status.idle":"2023-01-23T18:10:51.457121Z","shell.execute_reply.started":"2023-01-23T18:10:51.447185Z","shell.execute_reply":"2023-01-23T18:10:51.455996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HEIGHT = 520\nWIDTH = 704\nLABEL_MAP = {\"astro\":0, \"cort\":1, \"shsy5y\":2}\nBATCH_SIZE = 64\nNEPOCHS = 10\nGRADIENT_ACCUMULATION_STEPS = 1\nMIXED_PRECISION = \"fp16\"\nLEARNING_RATE = 0.001","metadata":{"execution":{"iopub.status.busy":"2023-01-23T18:10:51.458497Z","iopub.execute_input":"2023-01-23T18:10:51.459962Z","iopub.status.idle":"2023-01-23T18:10:51.470757Z","shell.execute_reply.started":"2023-01-23T18:10:51.459917Z","shell.execute_reply":"2023-01-23T18:10:51.469357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def transform1(d, path):\n    this = d.groupby(\"id\")[\"cell_type\"].unique()\n    arr = np.empty(shape=(len(this),2),dtype=\"object\")\n    for i in range(len(this)):\n        arr[i,1] = this[i][0]\n        arr[i,0] = path + \"/\" + this.index[i] + \".png\"\n    return pd.DataFrame(arr, columns=(\"path\",\"class\"))\n\ndef transform2(path):\n    liss = os.listdir(path)\n    arr = np.empty(shape=(len(liss),2),dtype=\"object\")\n    for i in range(len(liss)):\n        arr[i,0] = os.path.join(path,liss[i])\n        v = liss[i].split(\"[\")[0]\n        if v==\"astros\":\n            v=\"astro\"\n        arr[i,1] = v\n    return pd.DataFrame(arr, columns=(\"path\",\"class\"))\n\nclass image_dataset(Dataset):\n    \n    def __init__(self, df, processing_function=None):\n        super().__init__()\n        self.df = df\n        self.processing_function = processing_function\n        \n        \n    def __len__(self):\n        return self.df.shape[0]\n    \n    def __getitem__(self, ind):\n        path, label = tuple(self.df.iloc[ind,:])\n        label = torch.as_tensor(LABEL_MAP[label], dtype=torch.float)\n        img = Image.open(path)\n        if self.processing_function!=None:\n            img = self.processing_function(img)\n        img = T.Resize(size=(HEIGHT,WIDTH))(img)\n        img = T.PILToTensor()(img)/255.0\n        return img, label\n    \n    \ndef process_image(img):\n    return T.functional.equalize(img)\n\ndef get_features(x):\n    mean = x.mean().item()\n    std = x.std().item()\n    x = process_image((x*255).type(torch.uint8))/255\n    mean_eq = x.mean().item()\n    std_eq = x.std().item()\n    return mean, std, mean_eq, std_eq\n\ndef create_tabular_dataset(traindata):\n    tabdata = np.zeros(shape=(len(traindata),5))\n    for i in range(len(traindata)):\n        x = traindata.__getitem__(i)\n        tabdata[i,:4] = get_features(x[0])\n        tabdata[i,4] = int(x[1].item())\n    return pd.DataFrame(tabdata, columns=[\"mean\",\"std\",\"mean_equalized\",\"std_equalized\",\"label\"])\n\ndef split(df, val_size=200):\n    val_samples = data.sample(n=val_size)\n    train_samples = df.drop(index=val_samples.index)\n    return train_samples, val_samples","metadata":{"execution":{"iopub.status.busy":"2023-01-23T18:10:51.472239Z","iopub.execute_input":"2023-01-23T18:10:51.472641Z","iopub.status.idle":"2023-01-23T18:10:51.493952Z","shell.execute_reply.started":"2023-01-23T18:10:51.472584Z","shell.execute_reply":"2023-01-23T18:10:51.492419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data1 = pd.read_csv(\"/kaggle/input/sartorius-cell-instance-segmentation/train.csv\")\ndata1_images = \"/kaggle/input/sartorius-cell-instance-segmentation/train\"\nsemi_dir = \"/kaggle/input/sartorius-cell-instance-segmentation/train_semi_supervised\"","metadata":{"execution":{"iopub.status.busy":"2023-01-23T18:10:51.497171Z","iopub.execute_input":"2023-01-23T18:10:51.497749Z","iopub.status.idle":"2023-01-23T18:10:51.917454Z","shell.execute_reply.started":"2023-01-23T18:10:51.497687Z","shell.execute_reply":"2023-01-23T18:10:51.916236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data1 = transform1(data1, data1_images)\ndata2 = transform2(semi_dir)\ndata = pd.concat([data1,data2]).reset_index(drop=True)\ndata","metadata":{"execution":{"iopub.status.busy":"2023-01-23T18:10:51.919924Z","iopub.execute_input":"2023-01-23T18:10:51.920403Z","iopub.status.idle":"2023-01-23T18:10:52.007015Z","shell.execute_reply.started":"2023-01-23T18:10:51.920356Z","shell.execute_reply":"2023-01-23T18:10:52.004878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\ntraindata, valdata = split(data)\ntraindata = image_dataset(traindata, processing_function=None)\nvaldata = image_dataset(valdata, processing_function=None)\ntabdata = create_tabular_dataset(traindata)\nx, y = tabdata.drop(\"label\", axis=1), tabdata[\"label\"]\nxtrain, xtest, ytrain, ytest = train_test_split(x, y, test_size=0.1)\nrf = RandomForestClassifier().fit(xtrain,ytrain)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T18:10:52.008255Z","iopub.execute_input":"2023-01-23T18:10:52.008574Z","iopub.status.idle":"2023-01-23T18:11:24.329824Z","shell.execute_reply.started":"2023-01-23T18:10:52.008545Z","shell.execute_reply":"2023-01-23T18:11:24.328674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport random\nfrom scipy.ndimage import convolve\nfrom tqdm import tqdm\n\ndef get_voisins(i, j, shape):\n    \"\"\"Give list of neighbours next to actual point\n\n    Args:\n        i (_type_): _description_\n        j (_type_): _description_\n        shape (_type_): _description_\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n   \n    list_i = set([max(i-1, 0), i, min(i+1, shape[0]-1)])\n    list_j = set([max(j-1, 0), j, min(j+1, shape[1]-1)])\n    voisins = list()\n    for el_i in list_i:\n        for el_j in list_j:\n            voisins.append((el_i, el_j))\n    return voisins\n\n\ndef search_object(img, i, j):\n    \"\"\" Give a point, and return all linked points that are equals to 1\n    \n    Args:\n        img (_type_): _description_\n        i (_type_): _description_\n        j (_type_): _description_\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    new_object = list()\n    pile = [(i, j)]\n    img[i, j] = 0\n    while pile != []:\n        coord = pile.pop()\n        # print(len(pile))\n        if coord not in new_object:\n            new_object.append(coord)\n            neighbours = get_voisins(coord[0], coord[1], img.shape)\n            for (cur_i, cur_j) in neighbours:\n                if img[cur_i, cur_j] == 1:\n                    pile.append((cur_i, cur_j))\n                    img[cur_i, cur_j] = 0\n    return new_object, img\n\n\ndef extract_objects(img):\n    \"\"\"Search for all objects in the image\n\n    Args:\n        img (_type_): _description_\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    im = img.astype(float).copy()\n    objects = list()\n    (l, w) = im.shape\n    for i in range(l):\n        for j in range(w):\n            if im[i, j] >= 1:\n                new_object, im = search_object(im, i, j)\n                objects.append(new_object)\n    return objects\n\n\n\ndef objects_img(objects, img):\n    \"\"\"List of objects and img and return the object on a blank backcrean\n\n    Args:\n        objects (_type_): _description_\n        img (_type_): _description_\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    obj_img = np.zeros_like(img)\n    for obj in objects:\n        for (i, j) in obj:\n            obj_img[i, j] = 1\n    return obj_img\n\n\n\ndef object2format(object, img_shape):\n    \"\"\"Convert an object for the Kaggle submission format\n\n    Args:\n        object (_type_): _description_\n        img_shape (_type_): _description_\n\n    Returns:\n        _type_: _description_\n    \"\"\" \n    sorted_object = sorted(object, key=lambda x: (x[1], x[0]))\n    hauteur = img_shape[0]\n    object_ids = [j*hauteur + i + 1 for (i, j) in sorted_object]\n    final_format = list()\n    prev_idx = -1\n    count = 0\n    for idx in object_ids:\n        if idx == prev_idx+1:\n            count += 1\n        else:\n            if count:\n                final_format.append(count)\n            final_format.append(idx)\n            count = 1\n        prev_idx = idx\n    final_format.append(count)\n    return final_format\n\n\n\n\ndef objects2format(objects, img_shape):\n    \"\"\"Convert all objects for the Kaggle submission format\n\n    Args:\n        objects (_type_): _description_\n        img_shape (_type_): _description_\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    output = list()\n    for obj in objects:\n        output.append(object2format(obj, img_shape))\n    return output\n\n\n\ndef rle2mask(rle, shape = [520, 704]):\n    \"\"\"Convert the RLE to an img with a 1 at the point (mask of the cell)\n\n    Args:\n        rle (_type_): _description_\n        shape (list, optional): _description_. Defaults to [520, 704].\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    s = rle.split()\n    starts, lengths = [np.asarray(x, dtype = int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0] * shape[1], dtype = np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape)\n\ndef img_mask(img_id, df_train, shape=[520, 704]):\n    \"\"\"Get all anotations of an img\n\n    Args:\n        img_id (_type_): _description_\n        df_train (_type_): _description_\n        shape (list, optional): _description_. Defaults to [520, 704].\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    \n    img = np.zeros((shape[0], shape[1]))\n    for rle in df_train.loc[df_train.id == img_id, 'annotation']:\n        img += rle2mask(rle, shape)\n    return img\n\n\ndef mean_intensity(objects, img):\n    \"\"\"Compute mean intensity for each object\n\n    Args:\n        objects (_type_): _description_\n        img (_type_): _description_\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    intensity_list = list()\n    for obj in objects:\n        intensity = 0\n        for (i, j) in obj:\n            intensity += img[i, j]\n        intensity_list.append(intensity / len(obj))\n    return np.array(intensity_list)\n\n\n\ndef expend_object(object, img, border = 0, threshold=0.1, already_seen=set()):\n    \"\"\"Extend an object if the number of point added is above the treshold percentage\n\n    Args:\n        object (_type_): _description_\n        img (_type_): _description_\n        border (int, optional): _description_. Defaults to 0.\n        threshold (float, optional): _description_. Defaults to 0.1.\n        already_seen (_type_, optional): _description_. Defaults to set().\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    voisins = set()\n    for (i, j) in object:\n        cur_voisins = get_voisins(i, j, img.shape)\n        if (type(border) != int):\n            if border[i, j] == 1:\n                cur_voisins = [v for v in cur_voisins if border[v[0], v[1]] == 1]\n        voisins |= set(cur_voisins)\n    candidates = voisins - (set(object) | already_seen)\n    all_seen = set(object) | already_seen | voisins\n    if candidates == set():\n        return object, all_seen\n    kept_cadidates = list()\n    for (i, j) in candidates:\n        if img[i, j] == 1:\n            kept_cadidates.append((i, j))\n    density = len(kept_cadidates) / len(candidates)\n    if density >= threshold:\n        return object + kept_cadidates, all_seen\n    else:\n        return object, all_seen\n\n\n\n\ndef expend_max_object(object, img, threshold=0.5):\n    \"\"\"Extend an object until the number of point added is above the treshold percentage\n\n    Args:\n        object (_type_): _description_\n        img (_type_): _description_\n        threshold (float, optional): _description_. Defaults to 0.5.\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    \n    last_object = list()\n    while len(last_object) < len(object):\n        last_object = object\n        object, seen = expend_object(object, img, border = 0, threshold=threshold)\n    return object\n\ndef gradient(img):\n    \"\"\"Compute Sobel gradient\n\n    Args:\n        img (_type_): _description_\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    # TODO\n    sobel_x_img = convolve(img, np.array([[1, 0, -1], [2, 0, -2], [1, 0, -1]]))\n    sobel_y_img = convolve(img, np.array([[1, 2, 1], [0, 0, 0], [-1, -2, -1]]))\n\n    g_magnitude = np.sqrt(np.square(sobel_x_img) + np.square(sobel_y_img))\n    g_dir = np.arctan2(sobel_x_img, sobel_y_img)\n    return g_magnitude, g_dir\n\ndef expend_all_objects(objects, img, border=0, threshold=0.1):\n    \"\"\"Expend all objects of per one step\n\n    Args:\n        objects (_type_): _description_\n        img (_type_): _description_\n        border (int, optional): _description_. Defaults to 0.\n        threshold (float, optional): _description_. Defaults to 0.1.\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    copy_obj = objects.copy()\n    prev_seen = set()\n    seen = set([el for obj in copy_obj for el in obj])\n    fini_object_ids = list()\n    while prev_seen != seen:\n        prev_seen = seen\n        # print(len(seen))\n        for i in range(len(objects)):\n            if i not in fini_object_ids:\n                new_obj, seen = expend_object(copy_obj[i], img, border, threshold, seen)\n                if len(new_obj) != len(copy_obj[i]):\n                    copy_obj[i] = new_obj\n                else:\n                    fini_object_ids.append(i)\n        # seen = set([el for obj in copy_obj for el in obj])\n    return copy_obj\n\ndef plot_objects(list_objects, img):\n    \"\"\"Plot a list of object\n\n    Args:\n        list_objects (_type_): _description_\n        img (_type_): _description_\n    \"\"\"\n    for obj in list_objects:\n        coords = np.array([[i, j] for (i, j) in obj])\n        color = \"#\"+''.join([random.choice('ABCDEF0123456789') for i in range(6)])\n        plt.scatter(coords[:, 1] , coords[:, 0], c=color, s=0.1)\n    plt.imshow(np.zeros_like(img))\n    plt.show()\n\ndef plot_objects_ax(list_objects, img, ax):\n    \"\"\"Return the objects as an ax (to subplot)\n\n    Args:\n        list_objects (_type_): _description_\n        img (_type_): _description_\n        ax (_type_): _description_\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    for obj in list_objects:\n        coords = np.array([[i, j] for (i, j) in obj])\n        color = \"#\"+''.join([random.choice('ABCDEF0123456789') for i in range(6)])\n        ax.scatter(coords[:, 1] , coords[:, 0], c=color, s=0.1)\n    ax.imshow(np.zeros_like(img))\n    return ax\n\n\ndef object2image(object, img):\n    \"\"\"Compute 1 where the objects is located\n\n    Args:\n        object (_type_): _description_\n        img (_type_): _description_\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    obj_img = np.zeros_like(img)\n    for (i, j) in object:\n        obj_img[i, j] = 1\n    return obj_img\n\n\n\ndef remplis_trou(object, img):\n    \"\"\"Fill hole in the objects\n\n    Args:\n        object (_type_): _description_\n        img (_type_): _description_\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    obj_img = object2image(object, img)\n    non_intern = np.zeros_like(img)\n    start = (0, 0)\n    seen = set(object)\n    to_see = set([start])\n    while to_see != set():\n        next_step = set()\n        for (i, j) in to_see:\n            if obj_img[i, j] == 0:\n                non_intern[i, j] = 1\n            seen |= set([(i, j)])\n            next_step |= set(get_voisins(i, j, img.shape))\n        to_see = next_step - seen\n    return 1 - non_intern\n\n\n\n\ndef evaluation(image_objects, real_objects):\n    \"\"\"Return intersection matrix between predicted and true\n\n    Args:\n        image_objects (_type_): _description_\n        real_objects (_type_): _description_\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    intersection_matrix = np.zeros((len(image_objects), len(real_objects)))\n    for i in range(len(image_objects)):\n        for j in range(len(real_objects)):\n            intersection_matrix[i, j] = len(set(image_objects[i]) & set(real_objects[j])) / len(set(image_objects[i]) | set(real_objects[j]))\n    return intersection_matrix\n\n\n\ndef eval_thresh(intersection_matrix, threshold):\n    \"\"\"Compute the true positive rate within a give treshold\n\n    Args:\n        intersection_matrix (_type_): _description_\n        threshold (_type_): _description_\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    TP = 0\n    FP = 0\n    (l, w) = intersection_matrix.shape\n    for i in range(l):\n        if np.max(intersection_matrix[i]) >= threshold:\n            TP += 1\n        else:\n            FP += 1\n    FN = w - TP\n    return TP / (TP + FP + FN)\n\n\n\ndef eval_image(image_objects, real_objects):\n    \"\"\"Compute the mean of true positive rate\n\n    Args:\n        image_objects (_type_): _description_\n        real_objects (_type_): _description_\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    intersection_matrix = evaluation(image_objects, real_objects)\n    thresh_list = [0.5, 0.55, 0.6, 0.65, 0.7, 0.75, 0.8, 0.85, 0.9, 0.95]\n    scores = list()\n    for tresh in thresh_list:\n        scores.append(eval_thresh(intersection_matrix, tresh))\n    return np.mean(scores)\n\n\ndef real_object(rle, shape = [520, 704]):\n    \"\"\"Convert RLE into object\n\n    Args:\n        rle (_type_): _description_\n        shape (list, optional): _description_. Defaults to [520, 704].\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    obj = list()\n    (h, l) = shape\n    s = rle.split()\n    starts, lengths = [np.asarray(x, dtype = int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    for lo, hi in zip(starts, ends):\n        for j in range(lo, hi):\n            obj.append((j // l, j % l))\n    return obj\n\n\ndef real_objects(img_id, df_train, shape=[520, 704]):\n    \"\"\"Compute RLE into object for all objects\n\n    Args:\n        img_id (_type_): _description_\n        df_train (_type_): _description_\n        shape (list, optional): _description_. Defaults to [520, 704].\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    objects = list()\n    for rle in df_train.loc[df_train.id == img_id, 'annotation']:\n        objects.append(real_object(rle, shape))\n    return objects\n\n\ndef mean_img(image, kernel_size, threshold):\n    \"\"\"Function to smooth the image\n\n    Args:\n        image (_type_): _description_\n        kernel_size (_type_): _description_\n        threshold (_type_): _description_\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    (l, w) = image.shape\n    final_img = np.zeros_like(image)\n    for i in range(kernel_size, l-kernel_size):\n        for j in range(kernel_size, w-kernel_size):\n            counter = np.mean(image[i-kernel_size:i+kernel_size+1, j-kernel_size:j+kernel_size+1])\n            if counter <= threshold:\n                final_img[i-kernel_size:i+kernel_size+1, j-kernel_size:j+kernel_size+1] = 1\n    return final_img\n\n\n\ndef print_scores(true_objects, pred_objects, verbose=1):\n    \"\"\"Compute scores\n\n    Args:\n        true_objects (_type_): _description_\n        pred_objects (_type_): _description_\n        verbose (int, optional): _description_. Defaults to 1.\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    truth = set([el for o in true_objects for el in o])\n    pred = set([el for o in pred_objects for el in o])\n\n    TP = len(truth & pred)\n    FP = len(pred - truth)\n    FN = len(truth - pred)\n    TN = 704*520 - (TP+FP+FN)\n\n    precision = TP /(TP+FP)\n    recall = TP / (TP+FN)\n    acc = (TP + TN) / (TP + TN + FP + FN)\n    score = eval_image(pred_objects, true_objects)\n    f1score = 2*precision*recall/(precision+recall)\n    if verbose:\n        print('Précision : ', precision)\n        print('Recall : ', recall)\n        print('F score : ', f1score)\n        print('Accuracy : ', acc)\n        print(\"Kaggle score : \", score)\n    return precision, recall, f1score, acc, score\n\n#######\n# BASED ON VIC TD\n#######\n\ndef non_maximum_suppression(g_magnitude, g_dir):\n    \"\"\"Delete the non-maximum derivative\n\n    Args:\n        g_magnitude (_type_): _description_\n        g_dir (_type_): _description_\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    # TODO\n    g_max = np.zeros_like(g_magnitude)\n    h, w = g_magnitude.shape\n    pad_g = np.pad(g_magnitude, 1)\n    for i in range(1, h+1):\n        for j in range(1, w+1):\n            angle = g_dir[i-1, j-1]\n            if (np.pi/8 < angle and 3*np.pi / 4 > angle) or (5*np.pi/8 < angle and 7*np.pi / 4 > angle):\n                if pad_g[i, j-1] <= pad_g[i, j] and pad_g[i, j-1] <= pad_g[i, j]:\n                    g_max[i-1, j-1] = pad_g[i, j]\n                else:\n                    g_max[i-1, j-1] = 0\n            else:\n                if pad_g[i-1, j] <= pad_g[i, j] and pad_g[i+1, j] <= pad_g[i, j]:\n                    g_max[i-1, j-1] = pad_g[i, j]\n                else:\n                    g_max[i-1, j-1] = 0\n\n    return g_max\n\ndef double_thresholding(g_max, thresh_lo, thresh_hi):\n    \"\"\"Apply a double threshold\n\n    Args:\n        g_max (_type_): _description_\n        thresh_lo (_type_): _description_\n        thresh_hi (_type_): _description_\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    # TODO\n    thresh_img = np.zeros_like(g_max)\n    thresh_img[g_max >= thresh_hi] = 1\n    thresh_img[np.logical_and(g_max >= thresh_lo, g_max < thresh_hi)] = 0.5\n    return thresh_img\n\n\n\ndef connectivity(thresh_img):\n    \"\"\"Link close arête\n\n    Args:\n        thresh_img (_type_): _description_\n\n    Returns:\n        _type_: _description_\n    \"\"\"\n    # TODO\n    edge_img = np.zeros_like(thresh_img)\n    h, w = edge_img.shape\n    pad_img = np.pad(thresh_img, 1)\n    for i in range(1, h+1):\n        for j in range(1, w+1):\n            if pad_img[i, j] > 0.6:\n                edge_img[i-1, j-1] = 1\n            elif pad_img[i, j] > 0.1 and any([pad_img[i-1, j] == 1, pad_img[i+1, j] == 1, pad_img[i, j-1] == 1, pad_img[i, j+1] == 1]):\n                edge_img[i-1, j-1] = 1\n\n    return edge_img\n\ndef classify(ids,model=rf):\n    test_data = pd.DataFrame([['/kaggle/input/sartorius-cell-instance-segmentation/test/' + ids + '.png',\"cort\"]],columns=[\"path\",\"label\"])\n    test_data = image_dataset(test_data,processing_function=None)\n    tab_test_data = create_tabular_dataset(test_data)\n    result = model.predict(tab_test_data[[\"mean\",\"std\",\"mean_equalized\",\"std_equalized\"]])\n\n    return result[0]","metadata":{"execution":{"iopub.status.busy":"2023-01-23T18:11:24.332719Z","iopub.execute_input":"2023-01-23T18:11:24.333103Z","iopub.status.idle":"2023-01-23T18:11:24.414841Z","shell.execute_reply.started":"2023-01-23T18:11:24.333069Z","shell.execute_reply":"2023-01-23T18:11:24.412959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from skimage.filters import threshold_otsu\nimport imageio\nfrom PIL import Image, ImageEnhance\nfrom scipy.ndimage import convolve, gaussian_filter\n\ndef predict_contours(ig, i2, sigma=1, threshold=0.9):\n    \"\"\"\n    Extract the borders of the cells using gradient and fluorescent border.\n    ig: image with only black pixels with value 1\n    i2: image with only white pixels with value 1\n    \"\"\"\n    # Get edges thanks to gradient\n    g_magnitude, g_dir = gradient(gaussian_filter(ig, sigma))\n    g_max = non_maximum_suppression(g_magnitude, g_dir)\n    thresh_img = double_thresholding(g_max, thresh_lo=threshold, thresh_hi=threshold)\n    edge_img = connectivity(thresh_img)\n    # Get edges thanks to fluorescence\n    fluo_img = connectivity(i2)\n    # Combine both technics\n    contours = ((fluo_img + edge_img) > 0).astype(int)\n    return contours\n\n\ndef predict_shsy5y(image_id, contrast=16, thresh_contrast=10, sigma=2, seg_kernel=3, seg_tresh=0.1,\n    surface_kernel=1, surface_tresh=0.35, expend_tresh=0.2):\n    \"\"\"\n    Predict the number, place and shape of each cell of type shsy5y\n    \"\"\"\n    #Get image and apply contrast and thresholds\n    X = imageio.v2.imread('/kaggle/input/sartorius-cell-instance-segmentation/test/' + image_id + '.png')\n    _img = np.asarray(ImageEnhance.Contrast(Image.fromarray(X)).enhance(contrast))\n    ig = (_img < thresh_contrast).astype(float)\n    i2 = (_img > 250).astype(int)\n\n    # Smooth image\n    gaussian_img = gaussian_filter(ig, sigma=sigma)\n    # Apply Otsu threshold to keep only center of the cells\n    segmentation = gaussian_img > threshold_otsu(gaussian_img)\n    # Expend areas of interest\n    mim_seg = mean_img(segmentation, seg_kernel, seg_tresh)\n    if np.mean(mim_seg) > 0.5:\n        mim_seg = mean_img(1 - segmentation, seg_kernel, seg_tresh)\n    # Remove the edges to be able to dinstinguish objects\n    contours = predict_contours(ig, i2, threshold=0.5)\n    mim_seg = ((mim_seg - contours) > 0).astype(int)\n    # Extract objects and remove the smallest ones\n    seg_objects = extract_objects(mim_seg)\n    seg_objects = [o for o in seg_objects if len(o) > 30]\n    # Extract the surface covered by cells\n    mim = mean_img(_img > thresh_contrast, surface_kernel, surface_tresh)\n    # Expend the objects to their max surface\n    contours = (_img > 250).astype(int)\n    fill_objects_3 = expend_all_objects(seg_objects, mim, contours, expend_tresh)\n    return fill_objects_3\n\ndef predict_astro(image_id, contrast=16, thresh_contrast=100, sigma=1, seg_kernel=1, seg_tresh=0.1,\n    surface_kernel=1, surface_tresh=0.65):\n    \"\"\"\n    Predict the number, place and shape of each cell of type astro\n    \"\"\"\n    #Get image and apply contrast and thresholds\n    X = imageio.v2.imread('/kaggle/input/sartorius-cell-instance-segmentation/test/' + image_id + '.png')\n    _img = np.asarray(ImageEnhance.Contrast(Image.fromarray(X)).enhance(contrast))\n    ig = (_img < thresh_contrast).astype(float)\n    i2 = (_img > 250).astype(int)\n\n    # Smooth image\n    gaussian_img = gaussian_filter(ig, sigma=sigma)\n    # Apply Otsu threshold to keep only center of the cells\n    segmentation = gaussian_img > threshold_otsu(gaussian_img)\n    # Expend areas of interest\n    mim_seg = mean_img(segmentation, seg_kernel, seg_tresh)\n    if np.mean(mim_seg) > 0.5:\n        mim_seg = mean_img(1 - segmentation, seg_kernel, seg_tresh)\n    # Remove the edges to be able to dinstinguish objects\n    contours = predict_contours(ig, i2)\n    mim_seg = ((mim_seg - contours) > 0).astype(int)\n    # Extract objects and remove the smallest ones\n    seg_objects = extract_objects(mim_seg)\n    seg_objects = [o for o in seg_objects if len(o) > 50]\n    # Extract the surface covered by cells\n    mim = mean_img(_img > thresh_contrast, surface_kernel, surface_tresh)\n    # Expend the objects to their max surface\n    contours = (_img > 250).astype(int)\n    fill_objects_3 = expend_all_objects(seg_objects, mim, contours, 0.1)\n    return fill_objects_3\n\n\ndef predict_cort(image_id, contrast=16, thresh_contrast=10, filter_size=5, convol_tresh=0.4, sigma=0.5,\n    surface_kernel=3, surface_tresh=0.35, expend_tresh=0.1):\n    \"\"\"\n    Predict the number, place and shape of each cell of type cort\n    \"\"\"\n    #Get image and apply contrast and thresholds\n    X = imageio.v2.imread('/kaggle/input/sartorius-cell-instance-segmentation/test/' + image_id + '.png')\n    _img = np.asarray(ImageEnhance.Contrast(Image.fromarray(X)).enhance(contrast))\n    ig = (_img < thresh_contrast).astype(float)\n    i2 = (_img > 250).astype(int)\n\n    # Create the filter\n    n = filter_size\n    filtre = np.ones((n, n)) / (n*n)\n\n    # Apply filters to erase noise\n    filt = convolve(ig, filtre) > convol_tresh\n    bin_img = gaussian_filter(filt, sigma)\n    # Extract objects\n    objects = extract_objects(bin_img)\n\n    # Extract the surface covered by cells\n    mim = mean_img(_img > thresh_contrast, surface_kernel, surface_tresh)\n\n    # Expend the objects \n    expended_objects = list()\n    for object in objects:\n        final_object = expend_max_object(object, mim, expend_tresh)\n        expended_objects.append(final_object)\n\n    # Update the objects image to keep only interesting areas\n    new_bin = np.zeros_like(ig)\n    for obj in expended_objects:\n        for [i, j] in obj:\n            new_bin[i, j] = 1\n\n    # Remove the edges to be able to dinstinguish objects\n    contours = predict_contours(ig, i2, sigma=2)\n    mim_seg = ((new_bin - contours) > 0).astype(int)\n    # Extract the new objects and remove the smallest ones\n    new_objects = extract_objects(mim_seg)\n    new_objects = [o for o in new_objects if len(o)>20]\n    # Expend the final objects\n    mim = mean_img(_img > thresh_contrast, 3, 0.45)\n    new_objects = expend_all_objects(new_objects, mim, connectivity(i2), 0.1)\n    return new_objects","metadata":{"execution":{"iopub.status.busy":"2023-01-23T18:11:24.416687Z","iopub.execute_input":"2023-01-23T18:11:24.417445Z","iopub.status.idle":"2023-01-23T18:11:24.447711Z","shell.execute_reply.started":"2023-01-23T18:11:24.417408Z","shell.execute_reply":"2023-01-23T18:11:24.446671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id_1 = '7ae19de7bc2a'\nclassify(image_id_1)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T18:11:24.450050Z","iopub.execute_input":"2023-01-23T18:11:24.450414Z","iopub.status.idle":"2023-01-23T18:11:24.501301Z","shell.execute_reply.started":"2023-01-23T18:11:24.450382Z","shell.execute_reply":"2023-01-23T18:11:24.500115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(image_id):\n    \n    type_cell = classify(image_id)\n    if type_cell==0.:\n        objects_pred = predict_astro(image_id)\n    elif type_cell==1.:\n        objects_pred = predict_cort(image_id)\n    else:\n        objects_pred = predict_shsy5y(image_id)\n    \n    return(objects_pred)\n\n\ndef rle_encoding(objects_pred):\n    img = np.zeros((HEIGHT,WIDTH))\n    x = object2image(objects_pred,img)\n    dots = np.where(x.flatten() == 1)[0]\n    run_lengths = []\n    prev = -2\n    for b in dots:\n        if (b>prev+1): run_lengths.extend((b + 1, 0))\n        run_lengths[-1] += 1\n        prev = b\n    return ' '.join(map(str, run_lengths))","metadata":{"execution":{"iopub.status.busy":"2023-01-23T18:11:24.502875Z","iopub.execute_input":"2023-01-23T18:11:24.503811Z","iopub.status.idle":"2023-01-23T18:11:24.515509Z","shell.execute_reply.started":"2023-01-23T18:11:24.503754Z","shell.execute_reply":"2023-01-23T18:11:24.514197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/sartorius-cell-instance-segmentation/sample_submission.csv')\nids = df['id'].to_list()\nobjects = list()\nfinal_ids = list()\nfor i in tqdm(ids):\n    pred_objects = predict(i)\n    for p in pred_objects:\n        objects.append(rle_encoding(p))\n    final_ids += [i]*len(pred_objects)\n\ndf = pd.DataFrame()\ndf['id'] = final_ids\ndf['predicted'] = objects\ndf","metadata":{"execution":{"iopub.status.busy":"2023-01-23T18:11:24.516867Z","iopub.execute_input":"2023-01-23T18:11:24.517497Z","iopub.status.idle":"2023-01-23T18:13:58.895868Z","shell.execute_reply.started":"2023-01-23T18:11:24.517454Z","shell.execute_reply":"2023-01-23T18:13:58.894684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T18:13:58.897494Z","iopub.execute_input":"2023-01-23T18:13:58.898352Z","iopub.status.idle":"2023-01-23T18:13:58.910655Z","shell.execute_reply.started":"2023-01-23T18:13:58.898317Z","shell.execute_reply":"2023-01-23T18:13:58.909532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" ","metadata":{},"execution_count":null,"outputs":[]}]}