{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":6243,"databundleVersionId":868544,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import matplotlib.pyplot as plt\n%matplotlib inline\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport math\nfrom sklearn import mixture\nfrom sklearn.utils import shuffle\nfrom skimage import measure\nfrom glob import glob\nimport os\nfrom multiprocessing import Pool, cpu_count\nfrom functools import partial\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Conv2D, MaxPooling2D, Flatten, Dropout\nfrom keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split\n#from keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nTRAIN_DATA = \"../input/intel-mobileodt-cervical-cancer-screening/train/train\"\nTEST_DATA = \"../input/intel-mobileodt-cervical-cancer-screening/test/test\"\nADDITIONAL_DATA = \"../input/intel-mobileodt-cervical-cancer-screening/additional\"\n\ntypes = ['Type_1', 'Type_2', 'Type_3']\ntype_ids = []\n\nfor type in enumerate(types):\n    type_i_files = glob(os.path.join(TRAIN_DATA, type[1], \"*.jpg\"))\n    type_i_ids = np.array([s[len(TRAIN_DATA)+8:-4] for s in type_i_files])\n    type_ids.append(type_i_ids[:5])\n    \nprint(type_ids)\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-23T15:45:17.310491Z","iopub.execute_input":"2024-05-23T15:45:17.310892Z","iopub.status.idle":"2024-05-23T15:45:17.341201Z","shell.execute_reply.started":"2024-05-23T15:45:17.310861Z","shell.execute_reply":"2024-05-23T15:45:17.339972Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom skimage.io import imread, imshow\nimport cv2\n\n%matplotlib inline\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\nimport plotly.tools as tls\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input/intel-mobileodt-cervical-cancer-screening\"]).decode(\"utf8\"))","metadata":{"execution":{"iopub.status.busy":"2024-05-23T15:45:17.343818Z","iopub.execute_input":"2024-05-23T15:45:17.344402Z","iopub.status.idle":"2024-05-23T15:45:17.358764Z","shell.execute_reply.started":"2024-05-23T15:45:17.344365Z","shell.execute_reply":"2024-05-23T15:45:17.357631Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Helper Functions","metadata":{}},{"cell_type":"code","source":"def get_filename(image_id, image_type):\n    data_path = {\n        \"Type_1\": os.path.join(TRAIN_DATA, image_type),\n        \"Type_2\": os.path.join(TRAIN_DATA, image_type),\n        \"Type_3\": os.path.join(TRAIN_DATA, image_type),\n        \"Test\": TEST_DATA,\n        \"AType_1\": os.path.join(ADDITIONAL_DATA, image_type),\n        \"AType_2\": os.path.join(ADDITIONAL_DATA, image_type),\n        \"AType_3\": os.path.join(ADDITIONAL_DATA, image_type)\n    }.get(image_type, None)\n\n    if data_path is None:\n        raise Exception(\"Image type '%s' is not recognized\" % image_type)\n\n    return os.path.join(data_path, \"{}.jpg\".format(image_id))\n\ndef get_image_data(image_id, image_type):\n    fname = get_filename(image_id, image_type)\n    img = cv2.imread(fname)\n    if img is None:\n        raise Exception(\"Failed to read image : %s, %s\" % (image_id, image_type))\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    return img\n","metadata":{"execution":{"iopub.status.busy":"2024-05-23T15:45:17.361520Z","iopub.execute_input":"2024-05-23T15:45:17.362346Z","iopub.status.idle":"2024-05-23T15:45:17.371323Z","shell.execute_reply.started":"2024-05-23T15:45:17.362313Z","shell.execute_reply":"2024-05-23T15:45:17.370002Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Segmentation Functions\n","metadata":{}},{"cell_type":"code","source":"def maxHist(hist):\n    maxArea = (0, 0, 0)\n    height = []\n    position = []\n    for i in range(len(hist)):\n        if (len(height) == 0):\n            if (hist[i] > 0):\n                height.append(hist[i])\n                position.append(i)\n        else:\n            if (hist[i] > height[-1]):\n                height.append(hist[i])\n                position.append(i)\n            elif (hist[i] < height[-1]):\n                while (height[-1] > hist[i]):\n                    maxHeight = height.pop()\n                    area = maxHeight * (i - position[-1])\n                    if (area > maxArea[0]):\n                        maxArea = (area, position[-1], i)\n                    last_position = position.pop()\n                    if (len(height) == 0):\n                        break\n                position.append(last_position)\n                if (len(height) == 0):\n                    height.append(hist[i])\n                elif(height[-1] < hist[i]):\n                    height.append(hist[i])\n                else:\n                    position.pop()\n    while (len(height) > 0):\n        maxHeight = height.pop()\n        last_position = position.pop()\n        area =  maxHeight * (len(hist) - last_position)\n        if (area > maxArea[0]):\n            maxArea = (area, len(hist), last_position)\n    return maxArea\n\ndef maxRect(img):\n    maxArea = (0, 0, 0)\n    addMat = np.zeros(img.shape)\n    for r in range(img.shape[0]):\n        if r == 0:\n            addMat[r] = img[r]\n            area = maxHist(addMat[r])\n            if area[0] > maxArea[0]:\n                maxArea = area + (r,)\n        else:\n            addMat[r] = img[r] + addMat[r-1]\n            addMat[r][img[r] == 0] *= 0\n            area = maxHist(addMat[r])\n            if area[0] > maxArea[0]:\n                maxArea = area + (r,)\n    return (int(maxArea[3] + 1 - maxArea[0] / abs(maxArea[1] - maxArea[2])), maxArea[2], maxArea[3], maxArea[1], maxArea[0])\n\ndef cropCircle(img):\n    if(img.shape[0] > img.shape[1]):\n        tile_size = (int(img.shape[1] * 256 / img.shape[0]), 256)\n    else:\n        tile_size = (256, int(img.shape[0] * 256 / img.shape[1]))\n\n    img = cv2.resize(img, dsize=tile_size)\n    gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n    _, thresh = cv2.threshold(gray, 10, 255, cv2.THRESH_BINARY)\n    _, contours, _ = cv2.findContours(thresh.copy(), cv2.RETR_TREE, cv2.CHAIN_APPROX_NONE)\n    main_contour = sorted(contours, key=cv2.contourArea, reverse=True)[0]\n\n    ff = np.zeros((gray.shape[0], gray.shape[1]), 'uint8')\n    cv2.drawContours(ff, main_contour, -1, 1, 15)\n    ff_mask = np.zeros((gray.shape[0] + 2, gray.shape[1] + 2), 'uint8')\n    cv2.floodFill(ff, ff_mask, (int(gray.shape[1] / 2), int(gray.shape[0] / 2)), 1)\n\n    rect = maxRect(ff)\n    rectangle = [min(rect[0], rect[2]), max(rect[0], rect[2]), min(rect[1], rect[3]), max(rect[1], rect[3])]\n    img_crop = img[rectangle[0]:rectangle[1], rectangle[2]:rectangle[3]]\n    return [img_crop, rectangle, tile_size]\n","metadata":{"execution":{"iopub.status.busy":"2024-05-23T15:45:17.373537Z","iopub.execute_input":"2024-05-23T15:45:17.374329Z","iopub.status.idle":"2024-05-23T15:45:17.402413Z","shell.execute_reply.started":"2024-05-23T15:45:17.374285Z","shell.execute_reply":"2024-05-23T15:45:17.401316Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Extraction in Lab Color Space","metadata":{}},{"cell_type":"code","source":"def Ra_space(img, Ra_ratio, a_threshold):\n    imgLab = cv2.cvtColor(img, cv2.COLOR_RGB2LAB)\n    w = img.shape[0]\n    h = img.shape[1]\n    Ra = np.zeros((w * h, 2))\n    for i in range(w):\n        for j in range(h):\n            R = math.sqrt((w / 2 - i) ** 2 + (h / 2 - j) ** 2)\n            Ra[i * h + j, 0] = R\n            Ra[i * h + j, 1] = min(imgLab[i][j][1], a_threshold)\n    Ra[:, 0] /= max(Ra[:, 0])\n    Ra[:, 0] *= Ra_ratio\n    Ra[:, 1] /= max(Ra[:, 1])\n    return Ra\n","metadata":{"execution":{"iopub.status.busy":"2024-05-23T15:45:17.404419Z","iopub.execute_input":"2024-05-23T15:45:17.405612Z","iopub.status.idle":"2024-05-23T15:45:17.419965Z","shell.execute_reply.started":"2024-05-23T15:45:17.405577Z","shell.execute_reply":"2024-05-23T15:45:17.418707Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Clustering and Region Selection","metadata":{}},{"cell_type":"code","source":"def get_and_crop_image(image_id, image_type):\n    img = get_image_data(image_id, image_type)\n    initial_shape = img.shape\n    [img, rectangle_cropCircle, tile_size] = cropCircle(img)\n    Ra = Ra_space(img, 1.0, 150)\n    a_channel = np.reshape(Ra[:, 1], (img.shape[0], img.shape[1]))\n\n    g = mixture.GaussianMixture(n_components=2, covariance_type='diag', random_state=0, init_params='kmeans')\n    image_array_sample = shuffle(Ra, random_state=0)[:1000]\n    g.fit(image_array_sample)\n    labels = g.predict(Ra)\n    labels += 1\n\n    labels_2D = np.reshape(labels, (img.shape[0], img.shape[1]))\n    gg_labels_regions = measure.regionprops(labels_2D, intensity_image=a_channel)\n    gg_intensity = [prop.mean_intensity for prop in gg_labels_regions]\n    cervix_cluster = gg_intensity.index(max(gg_intensity)) + 1\n\n    mask = np.zeros((img.shape[0] * img.shape[1], 1), 'uint8')\n    mask[labels == cervix_cluster] = 255\n    mask_2D = np.reshape(mask, (img.shape[0], img.shape[1]))\n    cc_labels = measure.label(mask_2D, background=0)\n    regions = measure.regionprops(cc_labels)\n    areas = [prop.area for prop in regions]\n\n    regions_label = [prop.label for prop in regions]\n    largestCC_label = regions_label[areas.index(max(areas))]\n    mask_largestCC = np.zeros((img.shape[0], img.shape[1]), 'uint8')\n    mask_largestCC[cc_labels == largestCC_label] = 255\n\n    img_masked = img.copy()\n    img_masked[mask_largestCC == 0] = (0, 0, 0)\n    img_masked_gray = cv2.cvtColor(img_masked, cv2.COLOR_RGB2GRAY)\n\n    _, thresh_mask = cv2.threshold(img_masked_gray, 0, 255, 0)\n    kernel = np.ones((11, 11), np.uint8)\n    thresh_mask = cv2.dilate(thresh_mask, kernel, iterations=1)\n    thresh_mask = cv2.erode(thresh_mask, kernel, iterations=2)\n    _, contours_mask, _ = cv2.findContours(thresh_mask.copy(), cv2.RETR_TREE, cv2.CHAIN_APPROX_NONE)\n\n    main_contour = sorted(contours_mask, key=cv2.contourArea, reverse=True)[0]\n    cv2.drawContours(img, main_contour, -1, 255, 3)\n\n    x, y, w, h = cv2.boundingRect(main_contour)\n    rectangle = [x + rectangle_cropCircle[2], y + rectangle_cropCircle[0], w, h, initial_shape[0], initial_shape[1], tile_size[0], tile_size[1]]\n\n    return [image_id, img, rectangle]\n","metadata":{"execution":{"iopub.status.busy":"2024-05-23T15:45:17.443853Z","iopub.execute_input":"2024-05-23T15:45:17.444321Z","iopub.status.idle":"2024-05-23T15:45:17.463004Z","shell.execute_reply.started":"2024-05-23T15:45:17.444277Z","shell.execute_reply":"2024-05-23T15:45:17.462152Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":" # Process and Visualize Segmented Images","metadata":{}},{"cell_type":"code","source":"def parallelize_image_cropping(image_ids):\n    out = open('rectangles.csv', \"w\")\n    out.write(\"image_id,type,x,y,w,h,img_shp_0_init,img_shape1_init,img_shp_0,img_shp_1\\n\")\n    p = Pool(cpu_count())\n    for type in enumerate(types):\n        partial_get_and_crop = partial(get_and_crop_image, image_type=type[1])\n        ret = p.map(partial_get_and_crop, image_ids[type[0]])\n        for i in range(len(ret)):\n            out.write(image_ids[type[0]][i])\n            out.write(',' + str(type[1]))\n            out.write(',' + str(ret[i][2][0]))\n            out.write(',' + str(ret[i][2][1]))\n            out.write(',' + str(ret[i][2][2]))\n            out.write(',' + str(ret[i][2][3]))\n            out.write(',' + str(ret[i][2][4]))\n            out.write(',' + str(ret[i][2][5]))\n            out.write(',' + str(ret[i][2][6]))\n            out.write(',' + str(ret[i][2][7]))\n            out.write('\\n')\n            img = get_image_data(image_ids[type[0]][i], type[1])\n            if(img.shape[0] > img.shape[1]):\n                tile_size = (int(img.shape[1] * 256 / img.shape[0]), 256)\n            else:\n                tile_size = (256, int(img.shape[0] * 256 / img.shape[1]))\n            img = cv2.resize(img, dsize=tile_size)\n            cv2.rectangle(img, (ret[i][2][0], ret[i][2][1]), (ret[i][2][0] + ret[i][2][2], ret[i][2][1] + ret[i][2][3]), 255, 2)\n            plt.imshow(img)\n            plt.show()\n        ret = []\n    out.close()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-23T15:45:17.464975Z","iopub.execute_input":"2024-05-23T15:45:17.465351Z","iopub.status.idle":"2024-05-23T15:45:17.481667Z","shell.execute_reply.started":"2024-05-23T15:45:17.465320Z","shell.execute_reply":"2024-05-23T15:45:17.480471Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Build and Train Classification Model","metadata":{}},{"cell_type":"code","source":"def load_images_and_labels(type_ids):\n    images = []\n    labels = []\n    for idx, type in enumerate(types):\n        for image_id in type_ids[idx]:\n            img = get_image_data(image_id, type)\n            img = cv2.resize(img, (128, 128))\n            images.append(img)\n            labels.append(idx)\n    images = np.array(images)\n    labels = np.array(labels)\n    return images, labels\n\nimages, labels = load_images_and_labels(type_ids)\nlabels = to_categorical(labels, num_classes=len(types))\nprint(labels)\n\nX_train, X_test, y_train, y_test = train_test_split(images, labels, test_size=0.33, random_state=42)\n\ndatagen = ImageDataGenerator(rotation_range=40, width_shift_range=0.2, height_shift_range=0.2, shear_range=0.2, zoom_range=0.2, horizontal_flip=True, fill_mode='nearest')\ndatagen.fit(X_train)\n\nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(128, 128, 3)),\n    MaxPooling2D((2, 2)),\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Conv2D(128, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Flatten(),\n    Dense(512, activation='relu'),\n    Dropout(0.5),\n    Dense(len(types), activation='softmax')\n])\n\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\nhistory = model.fit(datagen.flow(X_train, y_train, batch_size=32), epochs=50, validation_data=(X_test, y_test))\n","metadata":{"execution":{"iopub.status.busy":"2024-05-23T15:45:17.483009Z","iopub.execute_input":"2024-05-23T15:45:17.484103Z","iopub.status.idle":"2024-05-23T15:45:51.398606Z","shell.execute_reply.started":"2024-05-23T15:45:17.484069Z","shell.execute_reply":"2024-05-23T15:45:51.397001Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}}]}