{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10418,"databundleVersionId":862236,"sourceType":"competition"}],"dockerImageVersionId":30747,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os \nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\n\n%matplotlib inline\n\nfrom skimage.io import imread\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n\ntrain_folder='/kaggle/input/human-protein-atlas-image-classification/train' #path for train folder\ncells_imgs=os.listdir(train_folder)\n\nprotein_df=pd.read_csv('/kaggle/input/human-protein-atlas-image-classification/train.csv') \n\nprint(protein_df.head(2))\nprint(protein_df.info()) \nprint(f'\\n cells_img_ids: {len(cells_imgs)/4}')\n","metadata":{"execution":{"iopub.status.busy":"2024-09-22T03:58:14.152688Z","iopub.execute_input":"2024-09-22T03:58:14.153056Z","iopub.status.idle":"2024-09-22T03:58:27.466922Z","shell.execute_reply.started":"2024-09-22T03:58:14.153024Z","shell.execute_reply":"2024-09-22T03:58:27.465909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"current_rand_state=np.random.get_state() #get the current random state","metadata":{"execution":{"iopub.status.busy":"2024-09-22T03:58:49.507641Z","iopub.execute_input":"2024-09-22T03:58:49.508457Z","iopub.status.idle":"2024-09-22T03:58:49.512857Z","shell.execute_reply.started":"2024-09-22T03:58:49.508425Z","shell.execute_reply":"2024-09-22T03:58:49.511786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 28 different labels for different unique locations where proteins can be present\n\nprotein_locs = {\n    0:  \"Nucleoplasm\",  \n    1:  \"Nuclear membrane\",   \n    2:  \"Nucleoli\",   \n    3:  \"Nucleoli fibrillar center\",   \n    4:  \"Nuclear speckles\",\n    5:  \"Nuclear bodies\",   \n    6:  \"Endoplasmic reticulum\",   \n    7:  \"Golgi apparatus\",   \n    8:  \"Peroxisomes\",   \n    9:  \"Endosomes\",   \n    10:  \"Lysosomes\",   \n    11:  \"Intermediate filaments\",   \n    12:  \"Actin filaments\",   \n    13:  \"Focal adhesion sites\",   \n    14:  \"Microtubules\",   \n    15:  \"Microtubule ends\",   \n    16:  \"Cytokinetic bridge\",   \n    17:  \"Mitotic spindle\",   \n    18:  \"Microtubule organizing center\",   \n    19:  \"Centrosome\",   \n    20:  \"Lipid droplets\",   \n    21:  \"Plasma membrane\",   \n    22:  \"Cell junctions\",   \n    23:  \"Mitochondria\",   \n    24:  \"Aggresome\",   \n    25:  \"Cytosol\",   \n    26:  \"Cytoplasmic bodies\",   \n    27:  \"Rods & rings\"\n}","metadata":{"execution":{"iopub.status.busy":"2024-09-22T03:59:03.543591Z","iopub.execute_input":"2024-09-22T03:59:03.544517Z","iopub.status.idle":"2024-09-22T03:59:03.552524Z","shell.execute_reply.started":"2024-09-22T03:59:03.544475Z","shell.execute_reply":"2024-09-22T03:59:03.551311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import MultiLabelBinarizer\n\nfrom skmultilearn.model_selection import iterative_train_test_split  \n#to split multicalss-multilabel data while preserving proportions of unique 28 labels\n\n\nprotein_ids=protein_df['Id'].values.reshape(-1,1) \n#get ids array and reshape to 2D to feed into split function \n\nlabels = [list(map(int, target.split())) for target in protein_df['Target']] \n#split target strings,chnage to int and make a list \n\nmlb=MultiLabelBinarizer() #mlb object to access the methods of M..L..B..() class\nbinary_labels=mlb.fit_transform(labels)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-09-22T03:59:09.581842Z","iopub.execute_input":"2024-09-22T03:59:09.582514Z","iopub.status.idle":"2024-09-22T03:59:09.821822Z","shell.execute_reply.started":"2024-09-22T03:59:09.582479Z","shell.execute_reply":"2024-09-22T03:59:09.821046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_counts=binary_labels.sum(axis=0) \nperc_proteins=np.array((labels_counts/labels_counts.sum()*100).round(1))\n# sum across each column to get counts of all unique protein locations in the dataset\n\nprint(f'labels_counts: \\n{labels_counts}')\n\nprint(f'percentages of proteins before split: \\n {(labels_counts/labels_counts.sum()*100).round(1)}') \n# percentages of all 28 protein locations in the dataset before split","metadata":{"execution":{"iopub.status.busy":"2024-09-22T08:56:11.028439Z","iopub.execute_input":"2024-09-22T08:56:11.029399Z","iopub.status.idle":"2024-09-22T08:56:11.037012Z","shell.execute_reply.started":"2024-09-22T08:56:11.029361Z","shell.execute_reply":"2024-09-22T08:56:11.036116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_sorted_indxs=np.argsort(labels_counts)[::-1]\nlabels_counts_sorted=labels_counts[labels_sorted_indxs]\n# print(labels_sorted_indxs)\n# print(labels_counts_sorted)\n# print(labels_counts)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T03:59:16.668554Z","iopub.execute_input":"2024-09-22T03:59:16.669095Z","iopub.status.idle":"2024-09-22T03:59:16.673427Z","shell.execute_reply.started":"2024-09-22T03:59:16.669064Z","shell.execute_reply":"2024-09-22T03:59:16.672311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def text_bars(bars):\n    for bar in bars:\n        plt.text(\n        bar.get_width(), \n        bar.get_y() + bar.get_height() / 2, \n        f'{int(bar.get_width())}', \n        va='center', \n        ha='left', \n        fontsize=14,\n        color='teal',\n        weight='bold'\n    )","metadata":{"execution":{"iopub.status.busy":"2024-09-22T03:59:19.074010Z","iopub.execute_input":"2024-09-22T03:59:19.074755Z","iopub.status.idle":"2024-09-22T03:59:19.079978Z","shell.execute_reply.started":"2024-09-22T03:59:19.074721Z","shell.execute_reply":"2024-09-22T03:59:19.079010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def outer_ax_invis():\n    ax = plt.gca()\n    ax.spines['top'].set_visible(False)\n    ax.spines['right'].set_visible(False)\n    ax.spines['left'].set_visible(False)\n    ax.spines['bottom'].set_visible(False)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T03:59:21.644043Z","iopub.execute_input":"2024-09-22T03:59:21.644797Z","iopub.status.idle":"2024-09-22T03:59:21.649592Z","shell.execute_reply.started":"2024-09-22T03:59:21.644765Z","shell.execute_reply":"2024-09-22T03:59:21.648661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Reference notebook for few exploration ideas -https://www.kaggle.com/code/allunia/protein-atlas-exploration-and-baseline That's a cool notebook!","metadata":{}},{"cell_type":"code","source":"# to visualise the count of different unique proteins in the dataset\n\nproteins_names_sorted=[protein_locs[i] for  i in labels_sorted_indxs]\n\n\ndef protein_bars(labels_counts_sorted):\n    plt.figure(figsize=(20,18))\n    bars=plt.barh( proteins_names_sorted,labels_counts_sorted, color='orange')\n    text_bars(bars)\n    plt.title('Counts of Protein Locations',fontsize='14',weight='bold',color='teal')\n    plt.xticks(color='crimson',weight='bold')\n    plt.yticks(color='crimson',fontsize='10',weight='bold')\n    plt.gca().invert_yaxis()\n    outer_ax_invis()\n    plt.tight_layout()\n    plt.show()\n    \nprotein_bars(labels_counts_sorted)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T03:59:24.331617Z","iopub.execute_input":"2024-09-22T03:59:24.332460Z","iopub.status.idle":"2024-09-22T03:59:25.188386Z","shell.execute_reply.started":"2024-09-22T03:59:24.332408Z","shell.execute_reply":"2024-09-22T03:59:25.187488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#to check percentages of different combinations of labels\nlabels_combination=binary_labels.sum(axis=1)\nval,counts=np.unique(labels_combination,return_counts=True)\nlabels_comb_perc=np.round(counts*100/labels_combination.shape[0],1)\n# print(labels_comb_perc)\nplt.bar(val,labels_comb_perc,color='deeppink')\nouter_ax_invis()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-22T03:59:31.134533Z","iopub.execute_input":"2024-09-22T03:59:31.134960Z","iopub.status.idle":"2024-09-22T03:59:31.435804Z","shell.execute_reply.started":"2024-09-22T03:59:31.134923Z","shell.execute_reply":"2024-09-22T03:59:31.434801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport seaborn as sns\n\nnum_classes = 28  \n\nproteins_names=[protein_locs[i] for i in range(28)]\n\n# co_occurrence_matrix = np.dot(binary_labels[labels_combination>1].T, binary_labels[labels_combination>1])\nmorethan1_proteins = binary_labels[labels_combination>1]\nmorethan1_proteins=morethan1_proteins.astype(float)\ncorrelation_matrix = np.corrcoef(binary_labels.T)\n\n\nplt.figure(figsize=(20, 18))\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm', vmin=-1, vmax=1, xticklabels=proteins_names, yticklabels=proteins_names)\nplt.title('Protein Co-Occurrence Heatmap')\nplt.xticks(rotation=90)\nplt.yticks(rotation=0)\nplt.tight_layout()\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-09-22T03:59:49.752128Z","iopub.execute_input":"2024-09-22T03:59:49.752746Z","iopub.status.idle":"2024-09-22T03:59:52.419403Z","shell.execute_reply.started":"2024-09-22T03:59:49.752715Z","shell.execute_reply":"2024-09-22T03:59:52.418465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for 1 sample\ndef binary_to_num(binary_labels):\n    bool_labels=(binary_labels==1)  #map binary labels to boolean format\n#     print(bool_labels)\n\n    protein_keys_arr=np.array(list(protein_locs.keys())) #numerical keys for all protein locs.\n\n    present_labels_num=protein_keys_arr[bool_labels] #list of protein locs. keys where true\n\n    present_labels= [protein_locs[label] for label in present_labels_num]  #list of labels in categorical format\n    \n    \n    return present_labels\n","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:06:56.022470Z","iopub.execute_input":"2024-09-22T04:06:56.022828Z","iopub.status.idle":"2024-09-22T04:06:56.028391Z","shell.execute_reply.started":"2024-09-22T04:06:56.022798Z","shell.execute_reply":"2024-09-22T04:06:56.027415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" \n    def plot_1cross4_randids(sample_ids,sample_labels,num_images,cmap='copper'):\n        rng_indxs=np.random.randint(0,len(sample_ids),num_images,dtype=int)\n        channels = [\"blue\", \"green\", \"red\", \"yellow\"]\n\n            \n        plt.figure(figsize=(20,18))\n        for i,rng_indx in enumerate(rng_indxs):\n            for j,channel in enumerate(channels):\n                image_path = os.path.join(train_folder, f\"{sample_ids[rng_indx,0]}_{channel}.png\")\n                img_arr = imread(image_path)\n                plt.subplot(num_images,4,i*4+j+1)\n                plt.imshow(img_arr, cmap=f'{cmap}')\n                plt.title(f\"{binary_to_num(sample_labels[rng_indx,:])}, {img_arr.max(),img_arr.min()}\")\n                plt.axis('off')\n        plt.tight_layout()\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:12:31.306908Z","iopub.execute_input":"2024-09-22T04:12:31.307572Z","iopub.status.idle":"2024-09-22T04:12:31.315328Z","shell.execute_reply.started":"2024-09-22T04:12:31.307541Z","shell.execute_reply":"2024-09-22T04:12:31.314161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef unique_labels_explore(label_name,protein_ids,binary_labels):\n    label_index=next(index for index,value in protein_locs.items() if value==label_name)\n    current_protein_ids=protein_ids[binary_labels[:,label_index]==1]\n    current_binary_labels=binary_labels[binary_labels[:,label_index]==1]\n#     print(current_binary_labels[:5])\n    num_images=5\n    plot_1cross4_randids(current_protein_ids,current_binary_labels,num_images=5)\n    \n\nunique_labels_explore('Mitochondria',protein_ids,binary_labels)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:12:46.258819Z","iopub.execute_input":"2024-09-22T04:12:46.259655Z","iopub.status.idle":"2024-09-22T04:12:49.278312Z","shell.execute_reply.started":"2024-09-22T04:12:46.259622Z","shell.execute_reply":"2024-09-22T04:12:49.276989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.set_state(current_rand_state)\n# print(current_rand_state)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:12:59.653411Z","iopub.execute_input":"2024-09-22T04:12:59.654390Z","iopub.status.idle":"2024-09-22T04:12:59.658908Z","shell.execute_reply.started":"2024-09-22T04:12:59.654344Z","shell.execute_reply":"2024-09-22T04:12:59.657957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils import shuffle\n\n\nnp.random.seed(32)  #change state for shuffle\n\n\ntrain_ids,train_labels, val_ids, val_labels= iterative_train_test_split(protein_ids, binary_labels, test_size=0.20)\n#split the dataset into train and validation sets\n\nnp.random.set_state(current_rand_state)  #set random state to default state\n","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:13:02.345957Z","iopub.execute_input":"2024-09-22T04:13:02.346789Z","iopub.status.idle":"2024-09-22T04:13:05.389530Z","shell.execute_reply.started":"2024-09-22T04:13:02.346759Z","shell.execute_reply":"2024-09-22T04:13:05.388696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'train_ids:{len(train_ids)}\\n')\nprint(f'val_ids:{len(val_ids)}\\n')\n\nprint(f'total_ids:{len(train_ids)+len(val_ids)}\\n')\n\n\n#to check if percentages of all unique labels remain same across all sets\nlabels_counts_train=train_labels.sum(axis=0)\nlabels_counts_val=val_labels.sum(axis=0)\nprint(f'labels_counts_train: \\n {labels_counts_train}')\nprint(f'% of train proteins after split:\\n{(labels_counts_train/labels_counts_train.sum()*100).round(1)}')\nprint(f'labels_counts_val: \\n {labels_counts_val}')\nprint(f'% of val proteins after split:\\n{(labels_counts_val/labels_counts_val.sum()*100).round(1)}')","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:13:10.770952Z","iopub.execute_input":"2024-09-22T04:13:10.771315Z","iopub.status.idle":"2024-09-22T04:13:10.780781Z","shell.execute_reply.started":"2024-09-22T04:13:10.771288Z","shell.execute_reply":"2024-09-22T04:13:10.779719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Results conclude that the method used for splitting correctly splits the data while preserving the proportions of unique classes across all sets.","metadata":{}},{"cell_type":"code","source":"\n#function to randomly display few samples for each channel\ndef display_random_images(sample_ids,train_folder,num_images):\n        \"\"\"\n        Visualize few image arrays for each channel and compare\n\n        parameters\n        ----------\n        sample_ids: 2-D array , size: samples,1\n        train_folder: string\n          path to train_folder contating all images\n        num_images: int\n          no of sample ids to be visualised basis selection\n  \n        \"\"\"\n        \n        plot_1cross4_randids(sample_ids,binary_labels,num_images,cmap='Oranges')","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:13:19.409101Z","iopub.execute_input":"2024-09-22T04:13:19.409810Z","iopub.status.idle":"2024-09-22T04:13:19.414605Z","shell.execute_reply.started":"2024-09-22T04:13:19.409776Z","shell.execute_reply":"2024-09-22T04:13:19.413655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_random_images(protein_ids,train_folder,5)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:13:24.515707Z","iopub.execute_input":"2024-09-22T04:13:24.516418Z","iopub.status.idle":"2024-09-22T04:13:27.456227Z","shell.execute_reply.started":"2024-09-22T04:13:24.516387Z","shell.execute_reply":"2024-09-22T04:13:27.455224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.set_state(current_rand_state)\n# print(current_rand_state)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:14:05.739989Z","iopub.execute_input":"2024-09-22T04:14:05.740806Z","iopub.status.idle":"2024-09-22T04:14:05.744932Z","shell.execute_reply.started":"2024-09-22T04:14:05.740772Z","shell.execute_reply":"2024-09-22T04:14:05.743865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Inferences from above visulations and analysis-\n1. Green channel must be kept as it contains the protein of interest to be classified.\n2. Red and Yellow channel representing microtubules and endoplasmic reticulum are almost same in structure\n3. Blue channel representing nucleus structure is unique from red and yellow channels.\n\nAs we are using blue,red and yellow channels to locate the protein for that sample . two channels are almost same so we can keep one channel out of red / yellow to reduce compute time as the same structure won't be giving much extra information for determining the location of protein.","metadata":{}},{"cell_type":"code","source":"def merge_channels(paths):\n    blue=imread(paths[0])\n    blue=cv2.resize(blue,(256,256))\n    blue=blue.astype(np.float32)\n    \n    green=imread(paths[1])\n    green=cv2.resize(green,(256,256))\n    green=green.astype(np.float32)\n\n    red=imread(paths[2])\n    red=cv2.resize(red,(256,256))\n    red=red.astype(np.float32)\n    \n    yellow=plt.imread(paths[3])\n    yellow=cv2.resize(yellow,(256,256))\n    \n    merged_img=cv2.merge((red,green,blue))\n#     merged_img=merged_img.astype(np.float32)\n    merged_img=merged_img/255.0\n#     merged_img=(merged_img-merged_img.min())/(merged_img.max()-merged_img.min())\n    \n\n#     print(merged_img[:10,:10,:])\n\n#     merged_img=(merged_img-0.5)*2\n    \n#     merged_imgg/=127.5\n#     merged_imgg-=1\n    \n#     print(merged_img[:10,:10,:])\n    \n    return merged_img","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:17:33.685444Z","iopub.execute_input":"2024-09-22T04:17:33.686053Z","iopub.status.idle":"2024-09-22T04:17:33.693595Z","shell.execute_reply.started":"2024-09-22T04:17:33.686020Z","shell.execute_reply":"2024-09-22T04:17:33.692578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:15:19.092368Z","iopub.execute_input":"2024-09-22T04:15:19.093068Z","iopub.status.idle":"2024-09-22T04:15:31.368325Z","shell.execute_reply.started":"2024-09-22T04:15:19.093036Z","shell.execute_reply":"2024-09-22T04:15:31.367352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#it's for plotting 1 cross 3 to check few contrast and brightness effects\ndef plot_1cross5_randids(sample_ids,sample_labels,num_images,pred_labels=None):\n        rng_indxs=np.random.randint(0,len(sample_ids),num_images,dtype=int)\n#         rng_indxs=np.array([0,1,2,3])\n        channels = [\"blue\", \"green\", \"red\", \"yellow\"]\n        aug_factors=[1.0,1.3,1.5]\n\n        def set_aug(img,j):\n            if j==0:\n                img=tf.image.adjust_contrast(img,1.5)\n            elif j==1:\n                img=tf.image.adjust_brightness(img,0.04)\n                img=tf.image.adjust_contrast(img,1.7)          \n            elif j==3:\n                img=tf.image.adjust_saturation(img,1.5)\n            return img    \n        \n            \n        plt.figure(figsize=(15,20))\n        for i,samp_id in enumerate(rng_indxs):\n            paths_per_id=[]\n            for channel in ['blue','green','red','yellow']:\n                img_path=os.path.join(train_folder,f'{sample_ids[samp_id,0]}_{channel}.png')\n                paths_per_id.append(img_path)\n            img=merge_channels(paths_per_id)\n            for j in range(3):\n                aug_img=set_aug(img,j)\n                plt.subplot(num_images,3,i*3+j+1)\n                plt.imshow(aug_img)\n                plt.axis('off')\n                plt.title(f\" {binary_to_num(sample_labels[samp_id,:])}\\n predicted:{ 'Not Trained' if pred_labels is None else binary_to_num(pred_labels[samp_id,:])} \")\n        plt.show()\n        plt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:26:52.217474Z","iopub.execute_input":"2024-09-22T04:26:52.217869Z","iopub.status.idle":"2024-09-22T04:26:52.229133Z","shell.execute_reply.started":"2024-09-22T04:26:52.217839Z","shell.execute_reply":"2024-09-22T04:26:52.227935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_1cross5_randids(protein_ids,binary_labels,4)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:26:55.408485Z","iopub.execute_input":"2024-09-22T04:26:55.409104Z","iopub.status.idle":"2024-09-22T04:26:57.512154Z","shell.execute_reply.started":"2024-09-22T04:26:55.409071Z","shell.execute_reply":"2024-09-22T04:26:57.511231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.set_state(current_rand_state)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:27:51.823975Z","iopub.execute_input":"2024-09-22T04:27:51.824984Z","iopub.status.idle":"2024-09-22T04:27:51.830725Z","shell.execute_reply.started":"2024-09-22T04:27:51.824942Z","shell.execute_reply":"2024-09-22T04:27:51.829539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#class to encase function for dataset pipeline \nclass ImageDatasetPipeline():\n    #constructor method to intialise and define instance(self --> refers to current object ) variables\n    def __init__(self, image_ids, labels, image_folder, batch_size=32, image_size=(256,256),augment=False,mode=None):\n        self.image_ids = image_ids[:,0]  \n        self.labels = labels\n        self.image_folder = image_folder\n        self.batch_size = batch_size\n        self.image_size = image_size\n        self.augment=augment\n        self.mode=mode\n   \n\n    def resize(self,img):\n        img=cv2.resize(img,self.image_size)\n        return img\n    \n    def norm(self,img):\n        img=(img-img.min())/(img.max()-img.min())\n        return img\n    \n\n    def random_geometric_transform(self,image):\n        image = tf.image.random_flip_left_right(image)\n        image = tf.image.random_flip_up_down(image)\n        return image\n\n    def apply_augmentation(self,image):\n        image=tf.image.adjust_brightness(image,0.05)\n        image=tf.image.random_contrast(image,1.3,1.7)  \n        image = self.random_geometric_transform(image)\n        return image\n\n        \n    def data_generation(self, image_id, label):\n            \"\"\"\n            takes single image id and label as input and outputs the scaled,resized and merged version\n\n            parameters\n            ----------\n            self: current instance\n            image_id: bytes/byte string,as it was a string tensor so got converted to bytes by numpy_func\n              path to one image id\n            label: array, size- 1,28\n              labels array for one target \n              \n            \"\"\"\n            \n            channel_paths=[]\n            for channel in ['blue','green','red','yellow']:\n                channel_paths.append(os.path.join(self.image_folder, f\"{image_id.decode('utf-8')}_{channel}.png\"))   \n            blue_imgarr=imread(channel_paths[0])\n            blue_imgarr=self.resize(blue_imgarr).astype(np.float32)\n            \n            green_imgarr=imread(channel_paths[1])\n            green_imgarr=self.resize(green_imgarr).astype(np.float32)\n            \n            red_imgarr=imread(channel_paths[2])\n            red_imgarr=self.resize(red_imgarr).astype(np.float32)\n            \n            merged_img=cv2.merge((red_imgarr, green_imgarr,blue_imgarr )).astype(np.float32)\n            \n            if self.mode==None:\n                merged_img/=255.0\n                merged_img=self.norm(merged_img)\n            elif self.mode=='v2':\n                merged_img/=127.5\n                merged_img-=1\n           \n            merged_img_fin=tf.cast(self.apply_augmentation(merged_img), tf.float64) if self.augment else tf.cast(merged_img,tf.float64)\n            return merged_img_fin,label\n        \n        \n    def create_tf_dataset(self):\n        \"\"\"\n        method to make dataset feasible for parallel processing and \n        for scalable integration with tf models\n        \n        \"\"\"\n\n        dataset = tf.data.Dataset.from_tensor_slices((self.image_ids, self.labels))\n        #slices the dataset along first dimension\n        \n#         np.random.seed(31)\n        \n#         dataset = dataset.shuffle(buffer_size=1000) \n        #for a batch select elems randomly from first 1000 elems with replacement\n        \n#         print(f'before shape{dataset.element_spec}')\n        expected_image_shape = (256, 256, 3)  \n        expected_label_shape = (28,)  \n        \n        #to map each element of datset tensors to numpy and convert outputs of data_generation back to tensors\n\n        def map_data(image_id,label):\n            image, label= tf.numpy_function(\n                self.data_generation,\n                [image_id, label],\n                [tf.float64, tf.int64],\n            )\n            image.set_shape(expected_image_shape)\n            label.set_shape(expected_label_shape)\n            return image, label\n        \n        dataset = dataset.map(\n            map_data,\n            num_parallel_calls=tf.data.AUTOTUNE\n        )\n\n        # Batch the dataset and prefetch for performance\n        dataset = dataset.batch(self.batch_size).prefetch(tf.data.AUTOTUNE)\n        \n        return dataset\n    \n# np.random.set_state(current_rand_state)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:30:15.641221Z","iopub.execute_input":"2024-09-22T04:30:15.641933Z","iopub.status.idle":"2024-09-22T04:30:15.661579Z","shell.execute_reply.started":"2024-09-22T04:30:15.641875Z","shell.execute_reply":"2024-09-22T04:30:15.660524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs_folder_path = '/kaggle/input/human-protein-atlas-image-classification/train'\n\n# Create data pipeline instances\ntrain_generator = ImageDatasetPipeline(train_ids, train_labels, imgs_folder_path, batch_size=32, image_size=(256,256),augment=True)\nval_generator = ImageDatasetPipeline(val_ids, val_labels, imgs_folder_path, batch_size=32, image_size=(256,256),augment=True)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:30:25.971200Z","iopub.execute_input":"2024-09-22T04:30:25.972049Z","iopub.status.idle":"2024-09-22T04:30:25.977047Z","shell.execute_reply.started":"2024-09-22T04:30:25.972019Z","shell.execute_reply":"2024-09-22T04:30:25.976103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = train_generator.create_tf_dataset() #access the method using object\nval_dataset=val_generator.create_tf_dataset()","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:30:28.822336Z","iopub.execute_input":"2024-09-22T04:30:28.822757Z","iopub.status.idle":"2024-09-22T04:30:28.947623Z","shell.execute_reply.started":"2024-09-22T04:30:28.822725Z","shell.execute_reply":"2024-09-22T04:30:28.946729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_dataset.element_spec)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:30:32.677970Z","iopub.execute_input":"2024-09-22T04:30:32.678620Z","iopub.status.idle":"2024-09-22T04:30:32.683677Z","shell.execute_reply.started":"2024-09-22T04:30:32.678588Z","shell.execute_reply":"2024-09-22T04:30:32.682628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for X_batch, y_batch in train_dataset.take(3):\n    X_batch = X_batch.numpy() \n    y_batch = y_batch.numpy()\n\nnum_labels=mlb.inverse_transform(y_batch)\n\n# to check four samples\nfor i in range(5):\n    plt.figure(figsize=(4,4))\n    plt.title(f'{tuple(protein_locs[elem] for elem in num_labels[i])}', fontsize=9)\n    plt.imshow(X_batch[i])\n    plt.axis('off')\n    \n    plt.tight_layout()\n    plt.show()\n    ","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:30:37.598345Z","iopub.execute_input":"2024-09-22T04:30:37.598705Z","iopub.status.idle":"2024-09-22T04:30:40.360722Z","shell.execute_reply.started":"2024-09-22T04:30:37.598677Z","shell.execute_reply":"2024-09-22T04:30:40.359800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To check and try out preprocessing steps for transfer learning","metadata":{}},{"cell_type":"code","source":"IMG_SHAPE=(256,256,3)\nbase_resmodel1=tf.keras.applications.ResNet50V2(\n    include_top=False,\n    weights='imagenet',\n    input_shape=IMG_SHAPE\n)","metadata":{"execution":{"iopub.status.busy":"2024-09-16T11:39:13.673403Z","iopub.execute_input":"2024-09-16T11:39:13.674243Z","iopub.status.idle":"2024-09-16T11:39:15.984162Z","shell.execute_reply.started":"2024-09-16T11:39:13.674208Z","shell.execute_reply":"2024-09-16T11:39:15.983356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preprocess_input_adapt=tf.keras.applications.resnet_v2.preprocess_input","metadata":{"execution":{"iopub.status.busy":"2024-09-13T09:53:04.010065Z","iopub.execute_input":"2024-09-13T09:53:04.010955Z","iopub.status.idle":"2024-09-13T09:53:04.01522Z","shell.execute_reply.started":"2024-09-13T09:53:04.010921Z","shell.execute_reply":"2024-09-13T09:53:04.014232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_batch,label_batch=next(iter(train_dataset))\nfeatures=base_resmodel1(image_batch)\nprint(features.shape)","metadata":{"execution":{"iopub.status.busy":"2024-09-16T06:06:36.648208Z","iopub.execute_input":"2024-09-16T06:06:36.648568Z","iopub.status.idle":"2024-09-16T06:06:39.007808Z","shell.execute_reply.started":"2024-09-16T06:06:36.648522Z","shell.execute_reply":"2024-09-16T06:06:39.006916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(image_batch[1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_resmodel.trainable=False","metadata":{"execution":{"iopub.status.busy":"2024-09-13T09:53:14.070142Z","iopub.execute_input":"2024-09-13T09:53:14.070505Z","iopub.status.idle":"2024-09-13T09:53:14.080181Z","shell.execute_reply.started":"2024-09-13T09:53:14.070476Z","shell.execute_reply":"2024-09-13T09:53:14.079266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import GlobalAveragePooling2D,Dense","metadata":{"execution":{"iopub.status.busy":"2024-09-10T10:04:09.721829Z","iopub.execute_input":"2024-09-10T10:04:09.72219Z","iopub.status.idle":"2024-09-10T10:04:09.726758Z","shell.execute_reply.started":"2024-09-10T10:04:09.72216Z","shell.execute_reply":"2024-09-10T10:04:09.725742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_1d=GlobalAveragePooling2D()(features)\nprint(features_1d.shape)","metadata":{"execution":{"iopub.status.busy":"2024-09-10T10:03:46.87639Z","iopub.execute_input":"2024-09-10T10:03:46.876775Z","iopub.status.idle":"2024-09-10T10:03:46.88432Z","shell.execute_reply.started":"2024-09-10T10:03:46.876747Z","shell.execute_reply":"2024-09-10T10:03:46.883453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_batch=Dense(28)(features_1d)\nprint(predicted_batch.shape)","metadata":{"execution":{"iopub.status.busy":"2024-09-10T10:06:18.551349Z","iopub.execute_input":"2024-09-10T10:06:18.551773Z","iopub.status.idle":"2024-09-10T10:06:18.563676Z","shell.execute_reply.started":"2024-09-10T10:06:18.551742Z","shell.execute_reply":"2024-09-10T10:06:18.562571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Actual transfer learning and finetuning starts here!!**","metadata":{}},{"cell_type":"code","source":"# without preprocessing as it changes images to (-1,1) which is not desirable \n\nfrom tensorflow.keras.layers import Conv2D, BatchNormalization, Activation, Dropout, MaxPooling2D,GlobalAveragePooling2D, Flatten, Dense,Input\nfrom tensorflow.keras.models import load_model,Model\nfrom tensorflow.keras import Model\nfrom tensorflow.keras.losses import BinaryCrossentropy \nfrom tensorflow.keras.metrics import Precision, Recall ,AUC,BinaryAccuracy\nfrom tensorflow.keras.regularizers import l2\n\n\nclass ProteinLocationsCNN:\n    def __init__(self, input_shape=(256, 256, 3), num_classes=28, batch_size=32,tune=False,weights_path=\"/kaggle/working/protein_cnn.weights.h5\"):\n        self.input_shape = input_shape\n        self.num_classes = num_classes\n        self.batch_size = batch_size\n        self.weights_path = weights_path\n        self.finetune=tune\n        self.model = self.base_model_adapt()\n        \n        \n    def base_resmodel(self):\n        base_resmodel=tf.keras.applications.ResNet50V2(\n        include_top=False,\n        weights='imagenet',\n        input_shape=self.input_shape\n        )\n        return base_resmodel\n        \n    def base_model_summary(self):\n        return self.base_resmodel().summary()\n        \n    def base_model_adapt(self):\n        base_resmodel=self.base_resmodel()\n        \n        inputs=Input(shape=self.input_shape)\n        x=inputs\n        \n        print(\"Number of layers in the base model: \", len(base_resmodel.layers))\n        if self.finetune:\n            base_resmodel.trainable = True\n#             finetune_at=140\n#             for layer in base_resmodel.layers[:finetune_at]:\n#                 layer.trainable=False\n            x=base_resmodel(x)\n        else:\n            base_resmodel.trainable=False\n            x=base_resmodel(x,training=False)\n            \n        x=GlobalAveragePooling2D()(x)\n#         x=Dropout(0.20)(x)\n        \n        x=Dense(512,kernel_regularizer=l2(0.01))(x)\n        x=BatchNormalization()(x)\n        x=Activation('relu')(x)\n        x=Dropout(0.5)(x)\n        \n        outputs= Dense(28, activation='linear')(x) #it will be converted to sigmoid later in compile\n\n        model =Model(inputs,outputs)\n        return model\n    \n    def compile_model(self, learning_rate=0.0001):\n        \"\"\"\n        Compiles the model with appropriate loss function, optimizer, and metrics.\n        \"\"\"\n        self.model.compile(\n            optimizer=tf.keras.optimizers.Adam(learning_rate=learning_rate),\n            loss=BinaryCrossentropy(from_logits=True), \n            metrics=[Precision(thresholds=0.5), Recall(thresholds=0.5),AUC(from_logits=True),BinaryAccuracy(threshold=0.5)]\n        )\n        \n    def summary_model(self):\n        \"\"\"\n        Summary of model with layers and parameters\n        \"\"\"\n        return self.model.summary()\n    \n    \n    def train_model(self, train_dataset, val_dataset, epochs=100, use_saved_weights=True):\n        \"\"\"\n        Trains the model. If `use_saved_weights` is True, it loads the saved weights and continues training.\n        \"\"\"\n        if use_saved_weights and os.path.exists(self.weights_path):\n            print(f\"Loading weights from {self.weights_path}\")\n            self.model.load_weights(self.weights_path)\n        \n        callbacks = [\n            tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=7, restore_best_weights=True),\n            tf.keras.callbacks.ModelCheckpoint(self.weights_path,save_weights_only=True,save_best_only=True),\n            tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.1,patience=5, min_lr=0.00001)\n        ]\n        \n        history = self.model.fit(\n            train_dataset,\n            validation_data=val_dataset,\n            epochs=epochs,\n            callbacks=callbacks,\n        )\n        return history\n    \n    def predict_model(self, test_dataset):\n        \"\"\"\n        Predicts the labels for the given dataset using the trained model.\n        \"\"\"\n        prediction_logits = self.model.predict(test_dataset)\n#         prediction_probs = self.model.predict(test_dataset)\n        prediction_probs=tf.nn.sigmoid(prediction_logits)\n        return prediction_probs\n    ","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:57:19.201878Z","iopub.execute_input":"2024-09-22T04:57:19.202706Z","iopub.status.idle":"2024-09-22T04:57:19.221922Z","shell.execute_reply.started":"2024-09-22T04:57:19.202674Z","shell.execute_reply":"2024-09-22T04:57:19.220871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"protein_cnn_resbase=ProteinLocationsCNN(input_shape=(256, 256, 3), num_classes=28, batch_size=32,tune=True,weights_path=\"/kaggle/working/protein_cnn.weights.h5\")","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:57:24.202106Z","iopub.execute_input":"2024-09-22T04:57:24.202552Z","iopub.status.idle":"2024-09-22T04:57:25.337022Z","shell.execute_reply.started":"2024-09-22T04:57:24.202515Z","shell.execute_reply":"2024-09-22T04:57:25.336094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"protein_cnn_resbase.base_model_summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"protein_cnn_resbase.summary_model()","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:57:28.556430Z","iopub.execute_input":"2024-09-22T04:57:28.556819Z","iopub.status.idle":"2024-09-22T04:57:28.588723Z","shell.execute_reply.started":"2024-09-22T04:57:28.556789Z","shell.execute_reply":"2024-09-22T04:57:28.587693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"protein_cnn_resbase.compile_model()","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:57:32.026089Z","iopub.execute_input":"2024-09-22T04:57:32.026459Z","iopub.status.idle":"2024-09-22T04:57:32.050919Z","shell.execute_reply.started":"2024-09-22T04:57:32.026429Z","shell.execute_reply":"2024-09-22T04:57:32.050164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history=protein_cnn_resbase.train_model(train_dataset, val_dataset, epochs=100, use_saved_weights=True)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T04:57:35.283054Z","iopub.execute_input":"2024-09-22T04:57:35.283418Z","iopub.status.idle":"2024-09-22T07:44:32.797523Z","shell.execute_reply.started":"2024-09-22T04:57:35.283389Z","shell.execute_reply":"2024-09-22T07:44:32.792782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"auc = history.history['auc_2']\nval_auc =history.history['val_auc_2']\n\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nprecision = history.history['precision_2']\nval_precision =history.history['val_precision_2']\n\nrecall = history.history['recall_2']\nval_recall = history.history['val_recall_2']\n\nbin_acc=history.history['binary_accuracy']\nval_bin_acc=history.history['val_binary_accuracy']","metadata":{"execution":{"iopub.status.busy":"2024-09-22T07:50:11.252666Z","iopub.execute_input":"2024-09-22T07:50:11.253059Z","iopub.status.idle":"2024-09-22T07:50:11.259116Z","shell.execute_reply.started":"2024-09-22T07:50:11.253029Z","shell.execute_reply":"2024-09-22T07:50:11.258128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metrics_all=[('loss',loss),('val_loss',val_loss),('bin_acc',bin_acc),('val_bin_acc',val_bin_acc),('auc',auc),('val_auc',val_auc),('precision',precision),('val_precision',val_precision),('recall',recall),('val_recall',val_recall)]","metadata":{"execution":{"iopub.status.busy":"2024-09-22T07:50:32.894856Z","iopub.execute_input":"2024-09-22T07:50:32.895693Z","iopub.status.idle":"2024-09-22T07:50:32.900733Z","shell.execute_reply.started":"2024-09-22T07:50:32.895659Z","shell.execute_reply":"2024-09-22T07:50:32.899721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,axs=plt.subplots(5,1,figsize=(15,20))\n\nfor i,j in zip(range(0,9,2),range(0,5)):\n    axs[j].plot(metrics_all[i][1],label='train')\n    axs[j].plot(metrics_all[i+1][1],label='validation')\n    axs[j].set_title(f'{metrics_all[i][0]},{metrics_all[i+1][0]}')\n    axs[j].legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-22T07:50:44.929194Z","iopub.execute_input":"2024-09-22T07:50:44.930061Z","iopub.status.idle":"2024-09-22T07:50:46.393173Z","shell.execute_reply.started":"2024-09-22T07:50:44.930025Z","shell.execute_reply":"2024-09-22T07:50:46.392129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_probs_train =protein_cnn_resbase.predict_model(train_dataset)\n# prediction_probs_train=tf.nn.sigmoid(prediction_logits_train)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T07:51:20.369351Z","iopub.execute_input":"2024-09-22T07:51:20.369833Z","iopub.status.idle":"2024-09-22T07:56:39.930683Z","shell.execute_reply.started":"2024-09-22T07:51:20.369794Z","shell.execute_reply":"2024-09-22T07:56:39.929627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(pred_probs_train[:5])\n# print(pred_probs_train[:5].numpy()>0.5)\npreds_train=(pred_probs_train.numpy()>0.5).astype(int)\n# print(preds_train[:5])","metadata":{"execution":{"iopub.status.busy":"2024-09-22T07:59:01.461206Z","iopub.execute_input":"2024-09-22T07:59:01.461829Z","iopub.status.idle":"2024-09-22T07:59:01.467734Z","shell.execute_reply.started":"2024-09-22T07:59:01.461796Z","shell.execute_reply":"2024-09-22T07:59:01.466615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_y_trues=[]\nfor _,y_true in train_dataset:\n    train_y_trues.append(y_true.numpy())\n\ntrain_y_trues_np=np.concatenate(train_y_trues,axis=0)\n# print(train_y_trues)\n    ","metadata":{"execution":{"iopub.status.busy":"2024-09-22T07:59:04.956179Z","iopub.execute_input":"2024-09-22T07:59:04.956580Z","iopub.status.idle":"2024-09-22T08:03:22.149995Z","shell.execute_reply.started":"2024-09-22T07:59:04.956547Z","shell.execute_reply":"2024-09-22T08:03:22.148951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_probs_val =protein_cnn_resbase.predict_model(val_dataset)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T08:03:46.475026Z","iopub.execute_input":"2024-09-22T08:03:46.475695Z","iopub.status.idle":"2024-09-22T08:05:08.057307Z","shell.execute_reply.started":"2024-09-22T08:03:46.475664Z","shell.execute_reply":"2024-09-22T08:05:08.056446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_val=(pred_probs_val.numpy()>0.5).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T08:06:37.165096Z","iopub.execute_input":"2024-09-22T08:06:37.166543Z","iopub.status.idle":"2024-09-22T08:06:37.173617Z","shell.execute_reply.started":"2024-09-22T08:06:37.166498Z","shell.execute_reply":"2024-09-22T08:06:37.172745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_y_trues=[]\nfor _,y_true in val_dataset:\n    val_y_trues.append(y_true.numpy())\n\nval_y_trues_np=np.concatenate(val_y_trues,axis=0)\n# print(train_y_trues)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T08:06:41.698601Z","iopub.execute_input":"2024-09-22T08:06:41.699607Z","iopub.status.idle":"2024-09-22T08:07:45.693289Z","shell.execute_reply.started":"2024-09-22T08:06:41.699569Z","shell.execute_reply":"2024-09-22T08:07:45.692207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_y_trues_np.shape)\nprint(preds_train.shape)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T08:08:19.723457Z","iopub.execute_input":"2024-09-22T08:08:19.723795Z","iopub.status.idle":"2024-09-22T08:08:19.728793Z","shell.execute_reply.started":"2024-09-22T08:08:19.723771Z","shell.execute_reply":"2024-09-22T08:08:19.727935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(train_y_trues_np[30:36])\n# print(preds_train[30:36])","metadata":{"execution":{"iopub.status.busy":"2024-09-22T08:08:13.671104Z","iopub.execute_input":"2024-09-22T08:08:13.671767Z","iopub.status.idle":"2024-09-22T08:08:13.675856Z","shell.execute_reply.started":"2024-09-22T08:08:13.671735Z","shell.execute_reply":"2024-09-22T08:08:13.674849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n\ntrain_conf_matrices=[]\ntrain_proteins_precision=[]\ntrain_proteins_recall=[]\n\nfor i in range(len(proteins_names)):\n    conf_matrix=confusion_matrix(train_y_trues_np[:,i],preds_train[:,i])\n    train_conf_matrices.append(conf_matrix)\n    \n    precision=round(conf_matrix[1][1]/(conf_matrix[1][1]+conf_matrix[0][1]),2)\n    train_proteins_precision.append(precision)\n    \n    recall=round(conf_matrix[1][1]/(conf_matrix[1][1]+conf_matrix[1][0]),2)\n    train_proteins_recall.append(recall)\n    \n    sns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap='Reds')\n    plt.title(proteins_names[i],fontsize=10)\n    plt.ylabel('True Labels')\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-09-22T08:11:06.559240Z","iopub.execute_input":"2024-09-22T08:11:06.560106Z","iopub.status.idle":"2024-09-22T08:11:11.729657Z","shell.execute_reply.started":"2024-09-22T08:11:06.560071Z","shell.execute_reply":"2024-09-22T08:11:11.728571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n\nval_conf_matrices=[]\nval_proteins_precision=[]\nval_proteins_recall=[]\n\nfor i in range(len(proteins_names)):\n    plt.figure(figsize=(2,2))\n    conf_matrix=confusion_matrix(val_y_trues_np[:,i],preds_val[:,i])\n    val_conf_matrices.append(conf_matrix)\n    \n    precision=round(conf_matrix[1][1]/(conf_matrix[1][1]+conf_matrix[0][1]),2)\n    val_proteins_precision.append(precision)\n    \n    recall=round(conf_matrix[1][1]/(conf_matrix[1][1]+conf_matrix[1][0]),2)\n    val_proteins_recall.append(recall)\n    \n    sns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap='Reds')\n    plt.title(proteins_names[i],fontsize=10)\n    plt.ylabel('True Labels')\n    plt.show()\n\n \n","metadata":{"execution":{"iopub.status.busy":"2024-09-22T08:10:46.600550Z","iopub.execute_input":"2024-09-22T08:10:46.601423Z","iopub.status.idle":"2024-09-22T08:10:54.365574Z","shell.execute_reply.started":"2024-09-22T08:10:46.601387Z","shell.execute_reply":"2024-09-22T08:10:54.364661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'train_proteins_precision:{train_proteins_precision}')\nprint(f'train_proteins_recall:{train_proteins_recall}')\nprint(f'val_proteins_precision:{val_proteins_precision}')\nprint(f'val_proteins_precision:{val_proteins_recall}')","metadata":{"execution":{"iopub.status.busy":"2024-09-22T08:12:24.732710Z","iopub.execute_input":"2024-09-22T08:12:24.733397Z","iopub.status.idle":"2024-09-22T08:12:24.740416Z","shell.execute_reply.started":"2024-09-22T08:12:24.733364Z","shell.execute_reply":"2024-09-22T08:12:24.739498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"F1_train=2*(np.array(train_proteins_precision)*np.array(train_proteins_recall))/(np.array(train_proteins_precision)+np.array(train_proteins_recall))\nF1_train=np.round(F1_train,2)\nprint('F1_train',F1_train,2)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T08:48:33.071853Z","iopub.execute_input":"2024-09-22T08:48:33.072764Z","iopub.status.idle":"2024-09-22T08:48:33.078976Z","shell.execute_reply.started":"2024-09-22T08:48:33.072732Z","shell.execute_reply":"2024-09-22T08:48:33.077919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"F1_val=2*(np.array(val_proteins_precision)*np.array(val_proteins_recall))/(np.array(val_proteins_precision)+np.array(val_proteins_recall))\nF1_val=np.round(F1_val,2)\nprint('F1_val: ',F1_val,2)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T08:47:49.962021Z","iopub.execute_input":"2024-09-22T08:47:49.962923Z","iopub.status.idle":"2024-09-22T08:47:49.969041Z","shell.execute_reply.started":"2024-09-22T08:47:49.962863Z","shell.execute_reply":"2024-09-22T08:47:49.968108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nmetric_df = pd.DataFrame({\n    'Proteins': list(protein_locs.values()),\n    'Train_prec': train_proteins_precision,\n    'Val_prec': val_proteins_precision,\n    'Train_recall': train_proteins_recall,\n    'Val_recall': val_proteins_recall,\n    'TrainF1':F1_train,\n    'ValF1':F1_val,\n    'Protein_perc.':perc_proteins\n   \n}, index=list(protein_locs.keys()))\n\npd.set_option('display.max_columns', None)  \npd.set_option('display.width', 1000)  ","metadata":{"execution":{"iopub.status.busy":"2024-09-22T08:52:04.693265Z","iopub.execute_input":"2024-09-22T08:52:04.693743Z","iopub.status.idle":"2024-09-22T08:52:04.704145Z","shell.execute_reply.started":"2024-09-22T08:52:04.693714Z","shell.execute_reply":"2024-09-22T08:52:04.701724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(metric_df)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T08:52:07.614223Z","iopub.execute_input":"2024-09-22T08:52:07.615086Z","iopub.status.idle":"2024-09-22T08:52:07.632463Z","shell.execute_reply.started":"2024-09-22T08:52:07.615052Z","shell.execute_reply":"2024-09-22T08:52:07.631461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_val_dataset=val_dataset.take(6)\nfor X_samp,y_trues in sub_val_dataset:\n    X_samp=X_samp\n    y_trues=y_trues.numpy()\nprint(X_samp.shape)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T09:00:24.852086Z","iopub.execute_input":"2024-09-22T09:00:24.852833Z","iopub.status.idle":"2024-09-22T09:00:26.948637Z","shell.execute_reply.started":"2024-09-22T09:00:24.852796Z","shell.execute_reply":"2024-09-22T09:00:26.947657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions=protein_cnn_resbase.predict_model(X_samp)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T09:00:32.138586Z","iopub.execute_input":"2024-09-22T09:00:32.139282Z","iopub.status.idle":"2024-09-22T09:00:35.759063Z","shell.execute_reply.started":"2024-09-22T09:00:32.139249Z","shell.execute_reply":"2024-09-22T09:00:35.758251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(predictions[0])","metadata":{"execution":{"iopub.status.busy":"2024-09-22T09:00:38.397844Z","iopub.execute_input":"2024-09-22T09:00:38.398338Z","iopub.status.idle":"2024-09-22T09:00:38.410546Z","shell.execute_reply.started":"2024-09-22T09:00:38.398306Z","shell.execute_reply":"2024-09-22T09:00:38.409425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_binary=(predictions > 0.5).numpy().astype(int)\n# print(preds_binary)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T09:00:41.449969Z","iopub.execute_input":"2024-09-22T09:00:41.450354Z","iopub.status.idle":"2024-09-22T09:00:41.470772Z","shell.execute_reply.started":"2024-09-22T09:00:41.450325Z","shell.execute_reply":"2024-09-22T09:00:41.470046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(preds_binary[1])\nprint(y_trues[1])","metadata":{"execution":{"iopub.status.busy":"2024-09-22T09:00:53.827592Z","iopub.execute_input":"2024-09-22T09:00:53.828235Z","iopub.status.idle":"2024-09-22T09:00:53.833859Z","shell.execute_reply.started":"2024-09-22T09:00:53.828203Z","shell.execute_reply":"2024-09-22T09:00:53.832648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_labels=mlb.inverse_transform(preds_binary)\nnum_trues=mlb.inverse_transform(y_trues)\n\n# to check four samples\nfor i in range(32):\n    plt.figure(figsize=(4,4))\n    plt.title(f'{tuple(protein_locs[elem] for elem in num_labels[i])}\\n true: {tuple(protein_locs[elem] for elem in num_trues[i])}', fontsize=9)\n    plt.imshow(X_samp[i])\n    plt.axis('off')\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-22T09:01:21.275391Z","iopub.execute_input":"2024-09-22T09:01:21.275783Z","iopub.status.idle":"2024-09-22T09:01:28.989383Z","shell.execute_reply.started":"2024-09-22T09:01:21.275750Z","shell.execute_reply":"2024-09-22T09:01:28.988387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**To check prediction uniquely for few some protein locations/classes to get a very baisc idea of human level performance and to see if anything can be done to improve the  model performance**","metadata":{}},{"cell_type":"code","source":"def unique_labels(label_name,protein_ids,binary_labels):\n    label_index=next(index for index,value in protein_locs.items() if value==label_name)\n    curr_protein_ids=protein_ids[binary_labels[:,label_index]==1]\n    curr_binary_labels=binary_labels[binary_labels[:,label_index]==1]\n    return curr_protein_ids,curr_binary_labels\ncurr_protein_ids,curr_protein_labels=unique_labels(\"Cytosol\",protein_ids,binary_labels)\n# print(curr_protein_ids[:5])\n","metadata":{"execution":{"iopub.status.busy":"2024-09-22T09:10:53.432978Z","iopub.execute_input":"2024-09-22T09:10:53.433821Z","iopub.status.idle":"2024-09-22T09:10:53.441922Z","shell.execute_reply.started":"2024-09-22T09:10:53.433790Z","shell.execute_reply":"2024-09-22T09:10:53.440947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_generator_cytosol = ImageDatasetPipeline(curr_protein_ids, curr_protein_labels, imgs_folder_path, batch_size=32, image_size=(256,256),augment=True)\n\nunique_preds_cytosol=val_generator_cytosol.create_tf_dataset()","metadata":{"execution":{"iopub.status.busy":"2024-09-18T11:21:55.580679Z","iopub.execute_input":"2024-09-18T11:21:55.581055Z","iopub.status.idle":"2024-09-18T11:21:55.605570Z","shell.execute_reply.started":"2024-09-18T11:21:55.581026Z","shell.execute_reply":"2024-09-18T11:21:55.604646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_probs_cytosol=protein_cnn_resbase.predict_model(unique_preds_cytosol)","metadata":{"execution":{"iopub.status.busy":"2024-09-18T11:21:59.203699Z","iopub.execute_input":"2024-09-18T11:21:59.204539Z","iopub.status.idle":"2024-09-18T11:23:46.570619Z","shell.execute_reply.started":"2024-09-18T11:21:59.204496Z","shell.execute_reply":"2024-09-18T11:23:46.569631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cytosol_preds_labels=(pred_probs_cytosol>0.5).numpy().astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-09-18T11:23:59.323143Z","iopub.execute_input":"2024-09-18T11:23:59.324053Z","iopub.status.idle":"2024-09-18T11:23:59.328883Z","shell.execute_reply.started":"2024-09-18T11:23:59.324017Z","shell.execute_reply":"2024-09-18T11:23:59.327977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_1cross5_randids(curr_protein_ids,curr_protein_labels,4,pred_labels=cytosol_preds_labels)","metadata":{"execution":{"iopub.status.busy":"2024-09-18T11:24:03.508904Z","iopub.execute_input":"2024-09-18T11:24:03.509546Z","iopub.status.idle":"2024-09-18T11:24:05.361039Z","shell.execute_reply.started":"2024-09-18T11:24:03.509512Z","shell.execute_reply":"2024-09-18T11:24:05.360079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nbs_protein_ids,nbs_protein_labels=unique_labels(\"Nuclear bodies\",protein_ids,binary_labels)","metadata":{"execution":{"iopub.status.busy":"2024-09-18T11:24:35.688185Z","iopub.execute_input":"2024-09-18T11:24:35.689034Z","iopub.status.idle":"2024-09-18T11:24:35.694626Z","shell.execute_reply.started":"2024-09-18T11:24:35.689002Z","shell.execute_reply":"2024-09-18T11:24:35.693567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_generator_nbs = ImageDatasetPipeline(nbs_protein_ids, nbs_protein_labels, imgs_folder_path, batch_size=32, image_size=(256,256),augment=True)\n\nunique_preds_nbs=val_generator_nbs.create_tf_dataset()","metadata":{"execution":{"iopub.status.busy":"2024-09-18T11:24:38.614938Z","iopub.execute_input":"2024-09-18T11:24:38.615294Z","iopub.status.idle":"2024-09-18T11:24:38.639371Z","shell.execute_reply.started":"2024-09-18T11:24:38.615265Z","shell.execute_reply":"2024-09-18T11:24:38.638305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_probs_nbs=protein_cnn_resbase.predict_model(unique_preds_nbs)","metadata":{"execution":{"iopub.status.busy":"2024-09-18T11:24:41.990080Z","iopub.execute_input":"2024-09-18T11:24:41.990465Z","iopub.status.idle":"2024-09-18T11:25:17.971768Z","shell.execute_reply.started":"2024-09-18T11:24:41.990431Z","shell.execute_reply":"2024-09-18T11:25:17.970775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nbs_preds_labels=(pred_probs_nbs>0.5).numpy().astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-09-18T11:25:56.086038Z","iopub.execute_input":"2024-09-18T11:25:56.086415Z","iopub.status.idle":"2024-09-18T11:25:56.091825Z","shell.execute_reply.started":"2024-09-18T11:25:56.086366Z","shell.execute_reply":"2024-09-18T11:25:56.090808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_1cross5_randids(nbs_protein_ids,nbs_protein_labels,4,pred_labels=nbs_preds_labels)","metadata":{"execution":{"iopub.status.busy":"2024-09-18T11:25:59.606554Z","iopub.execute_input":"2024-09-18T11:25:59.606908Z","iopub.status.idle":"2024-09-18T11:26:01.507737Z","shell.execute_reply.started":"2024-09-18T11:25:59.606878Z","shell.execute_reply":"2024-09-18T11:26:01.506853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* For nuclear bodies it kinda looks like model is worse than human level performance and it's predicting nuclear speckles and nucleoplasm.\n* Well not very true as i predicted to have nucleoplasm and nuclear bodies both but actually it's nuclear bodies only correctly predicted by the model. But it's clearly missing to preidct nuclear bodies when it clearly is present.\n* Also for 1 random image it seems mucleoplasm is present as well predicted by the model but actually it's not.\n","metadata":{}},{"cell_type":"code","source":"nspecks_protein_ids,nspecks_protein_labels=unique_labels(\"Nuclear speckles\",protein_ids,binary_labels)","metadata":{"execution":{"iopub.status.busy":"2024-09-18T11:26:18.326192Z","iopub.execute_input":"2024-09-18T11:26:18.326918Z","iopub.status.idle":"2024-09-18T11:26:18.332472Z","shell.execute_reply.started":"2024-09-18T11:26:18.326881Z","shell.execute_reply":"2024-09-18T11:26:18.331568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_generator_nspecks = ImageDatasetPipeline(nspecks_protein_ids, nspecks_protein_labels, imgs_folder_path, batch_size=32, image_size=(256,256),augment=True)\n\nunique_preds_nspecks=val_generator_nspecks.create_tf_dataset()","metadata":{"execution":{"iopub.status.busy":"2024-09-18T11:26:29.111240Z","iopub.execute_input":"2024-09-18T11:26:29.112076Z","iopub.status.idle":"2024-09-18T11:26:29.131596Z","shell.execute_reply.started":"2024-09-18T11:26:29.112043Z","shell.execute_reply":"2024-09-18T11:26:29.130820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_probs_nspecks=protein_cnn_resbase.predict_model(unique_preds_nspecks)","metadata":{"execution":{"iopub.status.busy":"2024-09-18T11:26:32.111667Z","iopub.execute_input":"2024-09-18T11:26:32.112382Z","iopub.status.idle":"2024-09-18T11:26:58.598619Z","shell.execute_reply.started":"2024-09-18T11:26:32.112350Z","shell.execute_reply":"2024-09-18T11:26:58.597781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nspecks_preds_labels=(pred_probs_nspecks>0.5).numpy().astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-09-18T11:27:01.269328Z","iopub.execute_input":"2024-09-18T11:27:01.269722Z","iopub.status.idle":"2024-09-18T11:27:01.280505Z","shell.execute_reply.started":"2024-09-18T11:27:01.269691Z","shell.execute_reply":"2024-09-18T11:27:01.279430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_1cross5_randids(nspecks_protein_ids,nspecks_protein_labels,4,pred_labels=nspecks_preds_labels)","metadata":{"execution":{"iopub.status.busy":"2024-09-18T11:27:03.669334Z","iopub.execute_input":"2024-09-18T11:27:03.669715Z","iopub.status.idle":"2024-09-18T11:27:05.456933Z","shell.execute_reply.started":"2024-09-18T11:27:03.669685Z","shell.execute_reply":"2024-09-18T11:27:05.455900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Okay nuclear speckles is usually when there is usually very thin of nucleus dia oval shapes around nucleus with pointed ends. like sometimes it's not that clear .\n* Also nuclear bodies and nuclear speckles do not seem to be occuring together.\n","metadata":{}},{"cell_type":"code","source":"nplasm_protein_ids,nplasm_protein_labels=unique_labels(\"Nucleoplasm\",protein_ids,binary_labels)","metadata":{"execution":{"iopub.status.busy":"2024-09-18T11:27:19.885784Z","iopub.execute_input":"2024-09-18T11:27:19.886126Z","iopub.status.idle":"2024-09-18T11:27:19.893009Z","shell.execute_reply.started":"2024-09-18T11:27:19.886100Z","shell.execute_reply":"2024-09-18T11:27:19.892190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_generator_nplasm = ImageDatasetPipeline(nplasm_protein_ids, nplasm_protein_labels, imgs_folder_path, batch_size=32, image_size=(256,256),augment=True)\n\nunique_preds_nplasm=val_generator_nplasm.create_tf_dataset()","metadata":{"execution":{"iopub.status.busy":"2024-09-18T11:27:24.233049Z","iopub.execute_input":"2024-09-18T11:27:24.233429Z","iopub.status.idle":"2024-09-18T11:27:24.256329Z","shell.execute_reply.started":"2024-09-18T11:27:24.233387Z","shell.execute_reply":"2024-09-18T11:27:24.255450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_probs_nplasm=protein_cnn_resbase.predict_model(unique_preds_nplasm)","metadata":{"execution":{"iopub.status.busy":"2024-09-18T11:27:28.260341Z","iopub.execute_input":"2024-09-18T11:27:28.261048Z","iopub.status.idle":"2024-09-18T11:30:15.749236Z","shell.execute_reply.started":"2024-09-18T11:27:28.261014Z","shell.execute_reply":"2024-09-18T11:30:15.748423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nplasm_preds_labels=(pred_probs_nplasm>0.5).numpy().astype(int)\nplot_1cross5_randids(nplasm_protein_ids,nplasm_protein_labels,4,pred_labels=nplasm_preds_labels)","metadata":{"execution":{"iopub.status.busy":"2024-09-18T11:33:55.690983Z","iopub.execute_input":"2024-09-18T11:33:55.691345Z","iopub.status.idle":"2024-09-18T11:33:57.587484Z","shell.execute_reply.started":"2024-09-18T11:33:55.691316Z","shell.execute_reply":"2024-09-18T11:33:57.586541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* I Cannot see a clear pattern between nucleoplasm and cytosol.\n* Maybe nucleoplasm is when both in nucleus and surrounding it there is more green density.\n* while in cytosol for only surrounding area but it's not very clear\n* many times nucleplasm occurs with cytosol and it also occurs with some other nucleus related classes.","metadata":{}},{"cell_type":"code","source":"mito_protein_ids,mito_protein_labels=unique_labels(\"Mitochondria\",protein_ids,binary_labels)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-22T09:11:05.127195Z","iopub.execute_input":"2024-09-22T09:11:05.127801Z","iopub.status.idle":"2024-09-22T09:11:05.133751Z","shell.execute_reply.started":"2024-09-22T09:11:05.127768Z","shell.execute_reply":"2024-09-22T09:11:05.132844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_generator_mito = ImageDatasetPipeline(mito_protein_ids, mito_protein_labels, imgs_folder_path, batch_size=32, image_size=(256,256),augment=True)\nunique_preds_mito=val_generator_mito.create_tf_dataset()\npred_probs_mito=protein_cnn_resbase.predict_model(unique_preds_mito)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T09:11:08.603314Z","iopub.execute_input":"2024-09-22T09:11:08.604518Z","iopub.status.idle":"2024-09-22T09:11:49.249749Z","shell.execute_reply.started":"2024-09-22T09:11:08.604483Z","shell.execute_reply":"2024-09-22T09:11:49.248865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mito_preds_labels=(pred_probs_mito>0.5).numpy().astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T09:12:13.478801Z","iopub.execute_input":"2024-09-22T09:12:13.479206Z","iopub.status.idle":"2024-09-22T09:12:13.484769Z","shell.execute_reply.started":"2024-09-22T09:12:13.479176Z","shell.execute_reply":"2024-09-22T09:12:13.483704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_1cross5_randids(mito_protein_ids,mito_protein_labels,4,pred_labels=mito_preds_labels)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-22T09:12:19.198230Z","iopub.execute_input":"2024-09-22T09:12:19.198602Z","iopub.status.idle":"2024-09-22T09:12:21.142475Z","shell.execute_reply.started":"2024-09-22T09:12:19.198571Z","shell.execute_reply":"2024-09-22T09:12:21.141577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Okay so in mitochondria around nucleus sharp green dots indicating it's presence.","metadata":{}},{"cell_type":"code","source":"lipid_protein_ids,lipid_protein_labels=unique_labels(\"Lipid droplets\",protein_ids,binary_labels)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T09:14:04.023230Z","iopub.execute_input":"2024-09-22T09:14:04.024075Z","iopub.status.idle":"2024-09-22T09:14:04.029028Z","shell.execute_reply.started":"2024-09-22T09:14:04.024040Z","shell.execute_reply":"2024-09-22T09:14:04.028103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_generator_lipid = ImageDatasetPipeline(lipid_protein_ids, lipid_protein_labels, imgs_folder_path, batch_size=32, image_size=(256,256),augment=True)\nunique_preds_lipid=val_generator_lipid.create_tf_dataset()\npred_probs_lipid=protein_cnn_resbase.predict_model(unique_preds_lipid)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T09:14:06.820425Z","iopub.execute_input":"2024-09-22T09:14:06.820823Z","iopub.status.idle":"2024-09-22T09:14:13.611672Z","shell.execute_reply.started":"2024-09-22T09:14:06.820791Z","shell.execute_reply":"2024-09-22T09:14:13.610847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lipid_preds_labels=(pred_probs_lipid>0.5).numpy().astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-09-22T09:14:29.830236Z","iopub.execute_input":"2024-09-22T09:14:29.830594Z","iopub.status.idle":"2024-09-22T09:14:29.836148Z","shell.execute_reply.started":"2024-09-22T09:14:29.830568Z","shell.execute_reply":"2024-09-22T09:14:29.835230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_1cross5_randids(lipid_protein_ids,lipid_protein_labels,4,pred_labels=lipid_preds_labels)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-22T09:15:48.858256Z","iopub.execute_input":"2024-09-22T09:15:48.858642Z","iopub.status.idle":"2024-09-22T09:15:52.057063Z","shell.execute_reply.started":"2024-09-22T09:15:48.858610Z","shell.execute_reply":"2024-09-22T09:15:52.056185Z"},"trusted":true},"execution_count":null,"outputs":[]}]}