{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.9","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":23870,"databundleVersionId":1781260,"sourceType":"competition"},{"sourceId":2023250,"sourceType":"datasetVersion","datasetId":1211002}],"dockerImageVersionId":30066,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# # # Data set Analysis","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\n# Input X-ray patch and kernel\ninput_image = np.array([[100, 120, 110, 90, 80],\n                        [105, 130, 125, 95, 85],\n                        [115, 140, 150, 100, 90],\n                        [110, 135, 145, 105, 95],\n                        [100, 125, 130, 95, 85]])\n\nkernel = np.array([[-1, 0, 1],\n                   [-1, 0, 1],\n                   [-1, 0, 1]])\n\n# Plot\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(10, 5))\n\n# Input image\nax1.imshow(input_image, cmap='gray', vmin=0, vmax=255)\nax1.set_title(\"Input X-ray Patch (5×5)\")\nfor i in range(5):\n    for j in range(5):\n        ax1.text(j, i, f\"{input_image[i, j]}\", ha='center', va='center', color='red')\n\n# Kernel\nax2.imshow(kernel, cmap='coolwarm', vmin=-1, vmax=1)\nax2.set_title(\"Kernel (3×3)\")\nfor i in range(3):\n    for j in range(3):\n        ax2.text(j, i, f\"{kernel[i, j]}\", ha='center', va='center', color='black')\n\nplt.tight_layout()\nplt.savefig(\"convolution_example.png\", dpi=300)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-11T14:19:19.715327Z","iopub.execute_input":"2026-09-11T14:19:19.715691Z","iopub.status.idle":"2026-09-11T14:19:20.758327Z","shell.execute_reply.started":"2026-09-11T14:19:19.715651Z","shell.execute_reply":"2026-09-11T14:19:20.757504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten\nfrom keras.layers import Conv2D, MaxPooling2D\nfrom keras.utils import to_categorical\nfrom keras.preprocessing import image\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\nimport plotly.graph_objs as go\nfrom plotly.offline import init_notebook_mode, iplot\nfrom plotly import tools\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2026-09-11T14:19:25.135132Z","iopub.execute_input":"2026-09-11T14:19:25.135427Z","iopub.status.idle":"2026-09-11T14:19:25.142209Z","shell.execute_reply.started":"2026-09-11T14:19:25.135402Z","shell.execute_reply":"2026-09-11T14:19:25.141532Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv('../input/ranzcr-clip-catheter-line-classification/train.csv') \n\nlabels  = (['ETT - Abnormal', 'ETT - Borderline',\n       'ETT - Normal', 'NGT - Abnormal', 'NGT - Borderline',\n       'NGT - Incompletely Imaged', 'NGT - Normal', 'CVC - Abnormal',\n       'CVC - Borderline', 'CVC - Normal', 'Swan Ganz Catheter Present'])\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:19:25.549688Z","iopub.execute_input":"2026-09-11T14:19:25.549979Z","iopub.status.idle":"2026-09-11T14:19:25.623187Z","shell.execute_reply.started":"2026-09-11T14:19:25.549955Z","shell.execute_reply":"2026-09-11T14:19:25.622270Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:19:33.537002Z","iopub.execute_input":"2026-09-11T14:19:33.537411Z","iopub.status.idle":"2026-09-11T14:19:33.542611Z","shell.execute_reply.started":"2026-09-11T14:19:33.537369Z","shell.execute_reply":"2026-09-11T14:19:33.541734Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Tubes coexistence in the same image :","metadata":{}},{"cell_type":"code","source":"\n#sample.groupby(['StudyInstanceUID']).sum(1).value_counts(sort=True)\n\ndf_train[labels].sum(1).value_counts(sort=True)","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:19:34.624979Z","iopub.execute_input":"2026-09-11T14:19:34.625257Z","iopub.status.idle":"2026-09-11T14:19:34.634832Z","shell.execute_reply.started":"2026-09-11T14:19:34.625233Z","shell.execute_reply":"2026-09-11T14:19:34.633615Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train[labels].sum(1).value_counts(sort=True).plot.bar()\nplt.xlabel(\"Occurrence number\")\nplt.ylabel(\"Simultaneous label number\")\n\n","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:19:37.632383Z","iopub.execute_input":"2026-09-11T14:19:37.632787Z","iopub.status.idle":"2026-09-11T14:19:37.776680Z","shell.execute_reply.started":"2026-09-11T14:19:37.632755Z","shell.execute_reply":"2026-09-11T14:19:37.775757Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\ntarg_cts=df_train.iloc[:,1:-2].sum(axis=0)\nfig = plt.figure(figsize=(12,6))\nsns.barplot(y=targ_cts.sort_values(ascending=False).index, x=targ_cts.sort_values(ascending=False).values, palette='flare')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:19:37.920957Z","iopub.execute_input":"2026-09-11T14:19:37.921311Z","iopub.status.idle":"2026-09-11T14:19:38.102880Z","shell.execute_reply.started":"2026-09-11T14:19:37.921277Z","shell.execute_reply":"2026-09-11T14:19:38.102168Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DataSet Imbalance :","metadata":{}},{"cell_type":"code","source":"df_train[labels].mean()","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:19:38.926348Z","iopub.execute_input":"2026-09-11T14:19:38.926727Z","iopub.status.idle":"2026-09-11T14:19:38.936364Z","shell.execute_reply.started":"2026-09-11T14:19:38.926685Z","shell.execute_reply":"2026-09-11T14:19:38.935562Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train[labels].mean().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:19:42.567177Z","iopub.execute_input":"2026-09-11T14:19:42.567562Z","iopub.status.idle":"2026-09-11T14:19:42.756525Z","shell.execute_reply.started":"2026-09-11T14:19:42.567525Z","shell.execute_reply":"2026-09-11T14:19:42.755737Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Height number of annotations for the same patient :","metadata":{}},{"cell_type":"code","source":"df_train['PatientID'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:19:43.558229Z","iopub.execute_input":"2026-09-11T14:19:43.558582Z","iopub.status.idle":"2026-09-11T14:19:43.572336Z","shell.execute_reply.started":"2026-09-11T14:19:43.558544Z","shell.execute_reply":"2026-09-11T14:19:43.571546Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ETT Tubes :","metadata":{}},{"cell_type":"code","source":"df_train[labels[:3]].value_counts().rename('Counts').reset_index()","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:19:48.218530Z","iopub.execute_input":"2026-09-11T14:19:48.218815Z","iopub.status.idle":"2026-09-11T14:19:48.234489Z","shell.execute_reply.started":"2026-09-11T14:19:48.218789Z","shell.execute_reply":"2026-09-11T14:19:48.233764Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# NG Tubes :","metadata":{}},{"cell_type":"code","source":"df_train[labels[3:7]].value_counts().rename('Counts').reset_index()\n","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:19:49.557023Z","iopub.execute_input":"2026-09-11T14:19:49.557298Z","iopub.status.idle":"2026-09-11T14:19:49.576642Z","shell.execute_reply.started":"2026-09-11T14:19:49.557274Z","shell.execute_reply":"2026-09-11T14:19:49.575800Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# CVC Tubes :","metadata":{}},{"cell_type":"code","source":"df_train[labels[7:10]].value_counts().rename('Counts').reset_index()","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:16:58.973730Z","iopub.execute_input":"2026-09-11T14:16:58.974048Z","iopub.status.idle":"2026-09-11T14:16:58.988544Z","shell.execute_reply.started":"2026-09-11T14:16:58.974016Z","shell.execute_reply":"2026-09-11T14:16:58.987859Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocesing :","metadata":{}},{"cell_type":"code","source":"from skimage import exposure\nfrom skimage.util import random_noise\ndef randRange(a, b):\n    return np.random.rand() * (b - a) + a\n\n\ndef AHE(img):\n    img_adapteq = exposure.equalize_adapthist(img, clip_limit=0.03)\n    #var = randRange(0.005, 0.01)\n    #img_adapteq=  random_noise(img_adapteq, var=var)\n    return img_adapteq\n#datagen = ImageDataGenerator(rotation_range=30, horizontal_flip=0.5, preprocessing_function=AHE)","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:16:58.989550Z","iopub.execute_input":"2026-09-11T14:16:58.989794Z","iopub.status.idle":"2026-09-11T14:16:59.204594Z","shell.execute_reply.started":"2026-09-11T14:16:58.989774Z","shell.execute_reply":"2026-09-11T14:16:59.203763Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df1 = pd.read_csv('../input/extract/train-Copy1.csv') \ndef append_ext(fn):\n    return fn+\".jpg\"\n\n\ntrain_df1[\"StudyInstanceUID\"]=train_df1[\"StudyInstanceUID\"].apply(append_ext)\ntrain_image = []\nfor i in tqdm(range(train_df1.shape[0]-1500)):\n    img = image.load_img(\"../input/ranzcr-clip-catheter-line-classification/train/\"+train_df1['StudyInstanceUID'][i],target_size=(640,640,3))\n    img = image.img_to_array(img)\n    img = img/255\n    train_image.append(img)\n         \nX = np.array(train_image)\n\ndf_lab =train_df1[['ETT - Abnormal', 'ETT - Borderline',\n       'ETT - Normal', 'NGT - Abnormal', 'NGT - Borderline',\n       'NGT - Incompletely Imaged', 'NGT - Normal', 'CVC - Abnormal',\n       'CVC - Borderline', 'CVC - Normal', 'Swan Ganz Catheter Present']]\nlabels = np.array(df_lab)\n#print (labels)\n\nX_train, X_test= train_test_split(X, test_size=0.2, random_state=42)\ny_train, y_test= train_test_split(labels, test_size=0.2, random_state=42)\nX_train = X_train.astype('float32')\nX_test = X_test.astype('float32')\n# print (labels_numeric)\n\nprint (X_train.shape)\n\nprint (y_train.shape)\n\nprint (X_test.shape)\n\nprint (y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:16:59.205890Z","iopub.execute_input":"2026-09-11T14:16:59.206171Z","iopub.status.idle":"2026-09-11T14:17:15.995867Z","shell.execute_reply.started":"2026-09-11T14:16:59.206145Z","shell.execute_reply":"2026-09-11T14:17:15.995004Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#\"\"\"\n#for i in range(40,90):\nplt.figure(figsize=(30,12))\nplt.subplot(141)\nplt.imshow((X_train[40]))\nplt.xticks([])\nplt.yticks([])\nplt.title('original image 1')\nplt.subplot(142)\nplt.imshow(AHE(X_train[40]))\nplt.xticks([])\nplt.yticks([])\nplt.title('image 1 with histogram equalization')\nplt.subplot(143)\nplt.imshow((X_train[120]))\nplt.xticks([])\nplt.yticks([])\nplt.title('original image 2')\nplt.subplot(144)\nplt.imshow(AHE(X_train[120]))\nplt.xticks([])\nplt.yticks([])\nplt.title('image 2 with histogram equalization')\nplt.show()\n#\"\"\"","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:17:15.997460Z","iopub.execute_input":"2026-09-11T14:17:15.997866Z","iopub.status.idle":"2026-09-11T14:17:17.095427Z","shell.execute_reply.started":"2026-09-11T14:17:15.997824Z","shell.execute_reply":"2026-09-11T14:17:17.094576Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport os\nimport tensorflow as tf\nimport tensorflow.keras.layers as tfl\nfrom tensorflow.keras import losses\nfrom tensorflow.keras.layers import Flatten ,Dense, Dropout\nfrom tensorflow.keras.preprocessing import image_dataset_from_directory\nfrom tensorflow.keras.layers.experimental.preprocessing import RandomFlip, RandomRotation,RandomCrop,RandomContrast,Normalization\n\n#load data\nimport pandas as pd\ntrain_df = pd.read_csv('../input/ranzcr-clip-catheter-line-classification/train.csv')\n#sample_df.shape\ndef append_ext(fn):\n    return fn+\".jpg\"\n\ntrain_df[\"StudyInstanceUID\"]=train_df[\"StudyInstanceUID\"].apply(append_ext)\n\ntrain_df.shape","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:17:17.096749Z","iopub.execute_input":"2026-09-11T14:17:17.097031Z","iopub.status.idle":"2026-09-11T14:17:17.166181Z","shell.execute_reply.started":"2026-09-11T14:17:17.096999Z","shell.execute_reply":"2026-09-11T14:17:17.165429Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BATCH_SIZE = 8\nIMG_SIZE = (380, 380)\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nlabel=['ETT - Abnormal', 'ETT - Borderline',\n       'ETT - Normal', 'NGT - Abnormal', 'NGT - Borderline',\n       'NGT - Incompletely Imaged', 'NGT - Normal', 'CVC - Abnormal',\n       'CVC - Borderline', 'CVC - Normal', 'Swan Ganz Catheter Present']\n\ndatagen=ImageDataGenerator(validation_split=0.20,\n                           rotation_range=0.3,\n                           horizontal_flip= True,\n                           rescale=1./255.)\n                          \n\ntrain_dataset=datagen.flow_from_dataframe(\n    dataframe=train_df,\n    directory=\"../input/ranzcr-clip-catheter-line-classification/train/\",\n    x_col=\"StudyInstanceUID\",\n    y_col=label,\n    subset=\"training\",\n    batch_size=BATCH_SIZE,\n    color_mode='rgb',\n    labels_mode ='binary',\n    class_mode='raw',\n    target_size=IMG_SIZE,    \n    shuffle=1024,\n    seed=42,\n    interpolation=\"bilinear\")\n\nvalidation_dataset=datagen.flow_from_dataframe(\ndataframe=train_df,\ndirectory=\"../input/ranzcr-clip-catheter-line-classification/train\",\nx_col=\"StudyInstanceUID\",\ny_col=label,\nsubset=\"validation\",\nbatch_size=BATCH_SIZE,\ncolor_mode='rgb',\nlabels_mode ='binary',\nclass_mode='raw',\ntarget_size=IMG_SIZE,\n#shuffle=1024,\nshuffle=False,\nseed=42,\ninterpolation=\"bilinear\")\n","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:17:17.167166Z","iopub.execute_input":"2026-09-11T14:17:17.167370Z","iopub.status.idle":"2026-09-11T14:18:40.637401Z","shell.execute_reply.started":"2026-09-11T14:17:17.167348Z","shell.execute_reply":"2026-09-11T14:18:40.635458Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image, _ = train_dataset.next()\n\nplt.figure(figsize=(10, 10))\nfirst_image = image[10]\n\nfor i in range(3):\n    ax = plt.subplot(3, 3, i + 1)\n    #augmented_image = data_augmentation(tf.expand_dims(first_image, 0))\n    augmented_image = tf.expand_dims(image[i], 0)\n    plt.imshow(augmented_image[0])\n    plt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:18:40.638530Z","iopub.status.idle":"2026-09-11T14:18:40.638948Z","shell.execute_reply":"2026-09-11T14:18:40.638777Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimage_shape = (380,380)\ninput_shape = image_shape + (3,)\nim_size =380\nfrom tensorflow.keras import Model, initializers, regularizers\n\n#from tensorflow.keras.applications.efficientnet import EfficientNetB5\nfrom tensorflow.keras.applications.resnet_v2 import ResNet50V2\n\nmodelB7 = tf.keras.Sequential([ResNet50V2(input_shape=(im_size, im_size, 3),\n                                                weights='imagenet',\n                                                include_top=False)])\n                                                #drop_connect_rate=0.7),\n                            # tf.keras.layers.GlobalAveragePooling2D()])\n    \n\n    \n    \ninputs = tf.keras.Input(shape=input_shape) \n\n#x = data_augmenter()(inputs)    \nx = modelB7(inputs) \nx = tf.keras.layers.GlobalAveragePooling2D()(x) \n#x =  Flatten()(x)\nx = Dropout(0.5)(x)\n#x = tfl.GlobalAveragePooling2D()(x)\noutputs = tfl.Dense(11,activation='sigmoid')(x)\n    \n\nmodel = tf.keras.Model(inputs, outputs)    \nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=3e-5),\n    loss='binary_crossentropy',\n    metrics=[tf.keras.metrics.AUC(multi_label=True)])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:18:40.639950Z","iopub.status.idle":"2026-09-11T14:18:40.640507Z","shell.execute_reply":"2026-09-11T14:18:40.640202Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"steps_per_epoch = 24067 // BATCH_SIZE\ncheckpoint = tf.keras.callbacks.ModelCheckpoint(\n    'model.h5', save_best_only=True, monitor='val_auc', mode='max')\nlr_reducer = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor='val_auc', patience=3, min_lr=1e-6, mode='max')","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:18:40.641660Z","iopub.status.idle":"2026-09-11T14:18:40.642170Z","shell.execute_reply":"2026-09-11T14:18:40.641904Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"initial_learning_rate = 1e-4\ndef lr_exp_decay(epoch, lr):\n    k = 0.3\n    return initial_learning_rate * tf.math.exp(-k*epoch)\n\nhistory = model.fit(\n    train_dataset,\n    verbose=True,\n    epochs=20,\n    #initial_epoch=history.epoch[-1],\n    #callbacks=[checkpoint],\n    callbacks=[checkpoint, tf.keras.callbacks.LearningRateScheduler(lr_exp_decay, verbose=1)],\n    steps_per_epoch=steps_per_epoch,\n    validation_data=validation_dataset)\n#model.save('P2-01.h5')\n","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:18:40.643134Z","iopub.status.idle":"2026-09-11T14:18:40.643768Z","shell.execute_reply":"2026-09-11T14:18:40.643421Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.load('P2-01.h5')","metadata":{"execution":{"iopub.status.busy":"2026-09-11T14:18:40.644862Z","iopub.status.idle":"2026-09-11T14:18:40.645384Z","shell.execute_reply":"2026-09-11T14:18:40.645106Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}