{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom tensorflow.keras.applications.resnet50 import ResNet50, preprocess_input\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.models import load_model\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.utils import load_img, img_to_array, array_to_img\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications.imagenet_utils import preprocess_input, decode_predictions\n\nimport cv2\nimport time\nfrom tqdm.notebook import tqdm\n\nfrom sklearn.metrics.pairwise import cosine_distances,pairwise_distances,cosine_similarity","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-02T18:05:12.907970Z","iopub.execute_input":"2022-12-02T18:05:12.908422Z","iopub.status.idle":"2022-12-02T18:05:20.103006Z","shell.execute_reply.started":"2022-12-02T18:05:12.908383Z","shell.execute_reply":"2022-12-02T18:05:20.101528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get Train Data","metadata":{}},{"cell_type":"code","source":"TRAIN_PATH = r\"/kaggle/input/sartorius-cell-instance-segmentation/train\"","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:05:33.358329Z","iopub.execute_input":"2022-12-02T18:05:33.358788Z","iopub.status.idle":"2022-12-02T18:05:33.363779Z","shell.execute_reply.started":"2022-12-02T18:05:33.358753Z","shell.execute_reply":"2022-12-02T18:05:33.362600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Amount of train images: \", len(os.listdir(TRAIN_PATH)))","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:06:03.501126Z","iopub.execute_input":"2022-12-02T18:06:03.502097Z","iopub.status.idle":"2022-12-02T18:06:03.509317Z","shell.execute_reply.started":"2022-12-02T18:06:03.502052Z","shell.execute_reply":"2022-12-02T18:06:03.508360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/sartorius-cell-instance-segmentation/train.csv\")\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:52:00.214611Z","iopub.execute_input":"2022-12-02T18:52:00.214986Z","iopub.status.idle":"2022-12-02T18:52:00.554865Z","shell.execute_reply.started":"2022-12-02T18:52:00.214954Z","shell.execute_reply":"2022-12-02T18:52:00.553772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"image_path\"] = TRAIN_PATH + \"/\" + train_df[\"id\"] + \".png\"","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:52:19.851464Z","iopub.execute_input":"2022-12-02T18:52:19.852611Z","iopub.status.idle":"2022-12-02T18:52:19.873594Z","shell.execute_reply.started":"2022-12-02T18:52:19.852537Z","shell.execute_reply":"2022-12-02T18:52:19.872598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_df = train_df.drop_duplicates(subset=[\"id\", \"image_path\"]).reset_index(drop=True)\ntmp_df[\"annotation\"] = train_df.groupby(\"id\")[\"annotation\"].agg(list).reset_index(drop=True)\ntrain_df = tmp_df.copy()\nprint(\"New train df shape:\", train_df.shape)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:52:55.863927Z","iopub.execute_input":"2022-12-02T18:52:55.864320Z","iopub.status.idle":"2022-12-02T18:52:55.901747Z","shell.execute_reply.started":"2022-12-02T18:52:55.864287Z","shell.execute_reply":"2022-12-02T18:52:55.900576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Helper Functions","metadata":{}},{"cell_type":"code","source":"def get_image_embedding(model, img_path):\n    \"\"\"\n    Returns image embedding as a pandas dataframe\n    \n    Args:\n        model: a desired model for prediction\n        img_path: a str that specifies location of image file\n    \"\"\"\n    image = load_img(img_path, target_size=(224, 224))\n    image = img_to_array(image)\n    image = np.expand_dims(image, axis=0) # (no.samples, h, w, c)\n    image = preprocess_input(image)\n    pred = model.predict(image)\n    pred_df = pd.DataFrame(pred[0]).T\n    return pred_df","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:06:43.255057Z","iopub.execute_input":"2022-12-02T18:06:43.255517Z","iopub.status.idle":"2022-12-02T18:06:43.262256Z","shell.execute_reply.started":"2022-12-02T18:06:43.255479Z","shell.execute_reply":"2022-12-02T18:06:43.261451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_img(img_path, title):\n    im = cv2.imread(img_path)\n    im = cv2.resize(im, (224, 224))\n    plt.axis('off')\n    plt.imshow(im[:,:,::-1])\n    plt.title(title)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:06:50.470637Z","iopub.execute_input":"2022-12-02T18:06:50.471154Z","iopub.status.idle":"2022-12-02T18:06:50.478961Z","shell.execute_reply.started":"2022-12-02T18:06:50.471112Z","shell.execute_reply":"2022-12-02T18:06:50.477520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fetch_most_similar_images(image_name, n_similar=7):\n    \"\"\"\n    Plots specified n similar images\n    \n    Args:\n        image_name: a str that is path to image\n        n_similar: an int that is the number of plotted similar images\n    \"\"\"\n    cell_typee = train_df[train_df[\"id\"] == image_name.split(\"/\")[-1].split(\".\")[0]].iloc[0].cell_type\n    print(\"=\"*42*2)\n    print(\"Original Cell:\")\n    show_img(image_name, cell_typee)\n    print(\"=\"*42*2)\n    curr_idx = img_embedding_df[img_embedding_df[\"image\"] == image_name].index[0]\n    closest_img = pd.DataFrame(cos_sim_df.iloc[curr_idx].nlargest(n_similar+1)[1:])\n    print(\"Similar Cells\")\n    for idx, image in (closest_img).iterrows():\n        sim_img_path = img_embedding_df.iloc[idx][\"image\"]\n        sim_img_name = train_df[train_df[\"id\"] == sim_img_path.split(\"/\")[-1].split(\".\")[0]].iloc[0].cell_type\n        sim = np.round(image.iloc[0], 3)\n        title = str(\"Image: \" + sim_img_name + \"\\nSimilarity: \" + str(sim) + \"%\")\n        show_img(sim_img_path, title)\n    print(\"*\"*42*2)","metadata":{"execution":{"iopub.status.busy":"2022-12-02T19:19:48.140738Z","iopub.execute_input":"2022-12-02T19:19:48.141140Z","iopub.status.idle":"2022-12-02T19:19:48.152192Z","shell.execute_reply.started":"2022-12-02T19:19:48.141106Z","shell.execute_reply":"2022-12-02T19:19:48.151053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fetch_least_similar_images(image_name, n_similar=7):\n    \"\"\"\n    Plots specified n non-similar images\n    \n    Args:\n        image_name: a str that is path to image\n        n_similar: an int that is the number of plotted similar images\n    \"\"\"\n    cell_typee = train_df[train_df[\"id\"] == image_name.split(\"/\")[-1].split(\".\")[0]].iloc[0].cell_type\n    print(\"=\"*42*2)\n    print(\"Original Cell:\")\n    show_img(image_name, cell_typee)\n    print(\"=\"*42*2)\n    curr_idx = img_embedding_df[img_embedding_df[\"image\"] == image_name].index[0]\n    closest_img = pd.DataFrame(cos_sim_df.iloc[curr_idx].nsmallest(n_similar+1)[1:])\n    print(\"Different Cells\")\n    for idx, image in (closest_img).iterrows():\n        sim_img_path = img_embedding_df.iloc[idx][\"image\"]\n        sim_img_name = train_df[train_df[\"id\"] == sim_img_path.split(\"/\")[-1].split(\".\")[0]].iloc[0].cell_type\n        sim = np.round(image.iloc[0], 3)\n        title = str(\"Image: \" + sim_img_name + \"\\nSimilarity: \" + str(sim) + \"%\")\n        show_img(sim_img_path, title)\n    print(\"*\"*42*2)","metadata":{"execution":{"iopub.status.busy":"2022-12-02T19:17:47.044792Z","iopub.execute_input":"2022-12-02T19:17:47.045192Z","iopub.status.idle":"2022-12-02T19:17:47.055003Z","shell.execute_reply.started":"2022-12-02T19:17:47.045157Z","shell.execute_reply":"2022-12-02T19:17:47.053845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Find Image Embeddings","metadata":{}},{"cell_type":"code","source":"model = ResNet50(include_top=False, weights=\"imagenet\", pooling=\"avg\")","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:07:34.121855Z","iopub.execute_input":"2022-12-02T18:07:34.122278Z","iopub.status.idle":"2022-12-02T18:07:40.562669Z","shell.execute_reply.started":"2022-12-02T18:07:34.122239Z","shell.execute_reply":"2022-12-02T18:07:40.561373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sanity check\nsample_path = \"/kaggle/input/sartorius-cell-instance-segmentation/train/0030fd0e6378.png\"\n\n# second last layer of resnet50 has 2048 vector size\nget_image_embedding(model, sample_path)","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:07:53.244104Z","iopub.execute_input":"2022-12-02T18:07:53.244523Z","iopub.status.idle":"2022-12-02T18:07:55.006152Z","shell.execute_reply.started":"2022-12-02T18:07:53.244481Z","shell.execute_reply":"2022-12-02T18:07:55.005013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# list of image names\nTRAIN_IMG_NAMES = os.listdir(TRAIN_PATH)\n\n# empty data frame\nimg_embedding_df = pd.DataFrame()\n\n# get image embeddings and add them to empty df\nfor filename in tqdm(TRAIN_IMG_NAMES):\n    img_path = os.path.join(TRAIN_PATH, filename)\n    current_df = get_image_embedding(model, img_path)\n    current_df[\"image\"] = img_path\n    img_embedding_df = pd.concat([img_embedding_df, current_df], ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:08:46.005985Z","iopub.execute_input":"2022-12-02T18:08:46.006556Z","iopub.status.idle":"2022-12-02T18:10:46.460006Z","shell.execute_reply.started":"2022-12-02T18:08:46.006515Z","shell.execute_reply":"2022-12-02T18:10:46.459050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape of Image Embeddings: \", img_embedding_df.shape, \"\\n\")\n\nimg_embedding_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:11:50.351119Z","iopub.execute_input":"2022-12-02T18:11:50.351536Z","iopub.status.idle":"2022-12-02T18:11:50.381846Z","shell.execute_reply.started":"2022-12-02T18:11:50.351499Z","shell.execute_reply":"2022-12-02T18:11:50.380985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Calculate Cosine Similarities","metadata":{}},{"cell_type":"code","source":"# cosine similarity data frame of embeddings\ncos_sim_df = pd.DataFrame(cosine_similarity(img_embedding_df.drop(\"image\", axis=1)))","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:12:03.639400Z","iopub.execute_input":"2022-12-02T18:12:03.639797Z","iopub.status.idle":"2022-12-02T18:12:03.671371Z","shell.execute_reply.started":"2022-12-02T18:12:03.639763Z","shell.execute_reply":"2022-12-02T18:12:03.669561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape of Cosine Similarity Matrix: \", cos_sim_df.shape, \"\\n\")\n\ncos_sim_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:12:05.028654Z","iopub.execute_input":"2022-12-02T18:12:05.029115Z","iopub.status.idle":"2022-12-02T18:12:05.061160Z","shell.execute_reply.started":"2022-12-02T18:12:05.029078Z","shell.execute_reply":"2022-12-02T18:12:05.059793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Distribution of Cell Types","metadata":{}},{"cell_type":"code","source":"train_df[\"image_path\"][0]","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:12:26.513761Z","iopub.execute_input":"2022-12-02T18:12:26.514321Z","iopub.status.idle":"2022-12-02T18:12:26.523293Z","shell.execute_reply.started":"2022-12-02T18:12:26.514274Z","shell.execute_reply":"2022-12-02T18:12:26.521562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_paths = {}\n\nfor i in range(len(train_df)):\n    cell_type = train_df[\"cell_type\"][i]\n    cell_path = train_df[\"image_path\"][i]\n    if not cell_type in cell_paths:\n        cell_paths[cell_type] = []\n    else:\n        cell_paths[cell_type].append(cell_path)\n        \nprint(\"Cell types\\n\\n\", list(cell_paths.keys()))","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:14:30.055850Z","iopub.execute_input":"2022-12-02T18:14:30.056380Z","iopub.status.idle":"2022-12-02T18:14:31.073600Z","shell.execute_reply.started":"2022-12-02T18:14:30.056319Z","shell.execute_reply":"2022-12-02T18:14:31.072334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_freqs = {}\n\nfor k in cell_paths.keys():\n    cell_freqs[k] = len(cell_paths[k])\n\ncell_freqs","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:14:42.225104Z","iopub.execute_input":"2022-12-02T18:14:42.225667Z","iopub.status.idle":"2022-12-02T18:14:42.234945Z","shell.execute_reply.started":"2022-12-02T18:14:42.225614Z","shell.execute_reply":"2022-12-02T18:14:42.233682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"freq_df = pd.DataFrame([cell_freqs.keys(), cell_freqs.values()]).T\nfreq_df.rename(columns={0: \"cell_type\",\n                       1: \"count\"}, inplace=True)\n\nfreq_df","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:15:00.514759Z","iopub.execute_input":"2022-12-02T18:15:00.515286Z","iopub.status.idle":"2022-12-02T18:15:00.533467Z","shell.execute_reply.started":"2022-12-02T18:15:00.515242Z","shell.execute_reply":"2022-12-02T18:15:00.531856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,15))\nsns.barplot(freq_df[\"cell_type\"], freq_df[\"count\"])\nplt.savefig(\"train_freq.png\")","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:15:16.564583Z","iopub.execute_input":"2022-12-02T18:15:16.565235Z","iopub.status.idle":"2022-12-02T18:15:16.843111Z","shell.execute_reply.started":"2022-12-02T18:15:16.565177Z","shell.execute_reply":"2022-12-02T18:15:16.842075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize","metadata":{}},{"cell_type":"markdown","source":"## Heatmap of Cosine Similarities","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(20, 20))\nsns.heatmap(cos_sim_df, cmap=\"BuPu\")\nplt.savefig(\"train_cos_sim_heatmap.png\")","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:16:52.663092Z","iopub.execute_input":"2022-12-02T18:16:52.663816Z","iopub.status.idle":"2022-12-02T18:16:56.982254Z","shell.execute_reply.started":"2022-12-02T18:16:52.663757Z","shell.execute_reply":"2022-12-02T18:16:56.981026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_cos_sim_cells(sim_type=\"most\", rank=5):\n    if sim_type == \"most\":\n        for cell_type in cell_paths.keys():\n            fetch_most_similar_images(os.path.join(TRAIN_PATH, cell_paths[cell_type][rank]))\n    \n    elif sim_type == \"least\":\n        for cell_type in cell_paths.keys():\n            fetch_least_similar_images(os.path.join(TRAIN_PATH, cell_paths[cell_type][rank]))\n\n    else:\n        print(\"You must pass 'most' or 'least' as similarity type and rank must be an integer.\")","metadata":{"execution":{"iopub.status.busy":"2022-12-02T18:18:13.821859Z","iopub.execute_input":"2022-12-02T18:18:13.822269Z","iopub.status.idle":"2022-12-02T18:18:13.830017Z","shell.execute_reply.started":"2022-12-02T18:18:13.822234Z","shell.execute_reply":"2022-12-02T18:18:13.828821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Plot The Most Similar Cells for Each Type","metadata":{}},{"cell_type":"code","source":"plot_cos_sim_cells(sim_type=\"most\")","metadata":{"execution":{"iopub.status.busy":"2022-12-02T19:19:57.241083Z","iopub.execute_input":"2022-12-02T19:19:57.241499Z","iopub.status.idle":"2022-12-02T19:20:00.677003Z","shell.execute_reply.started":"2022-12-02T19:19:57.241466Z","shell.execute_reply":"2022-12-02T19:20:00.675549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"######################################################################\n\n### Most Similar Cells by Order\n\n* shsy5y : shsy5y, astro \n\n* astro : astro\n\n* cort : cort\n\n######################################################################\n","metadata":{}},{"cell_type":"markdown","source":"## Plot The Least Similar Cells for Each Type","metadata":{}},{"cell_type":"code","source":"plot_cos_sim_cells(sim_type=\"least\")","metadata":{"execution":{"iopub.status.busy":"2022-12-02T19:18:39.072164Z","iopub.execute_input":"2022-12-02T19:18:39.072598Z","iopub.status.idle":"2022-12-02T19:18:41.419318Z","shell.execute_reply.started":"2022-12-02T19:18:39.072564Z","shell.execute_reply":"2022-12-02T19:18:41.418038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"######################################################################\n\n### Least Similar Cells by Order\n\n* shsy5y : astro, cort \n\n* astro : cort\n\n* cort : astro, cort\n\n######################################################################","metadata":{}},{"cell_type":"markdown","source":"# Last Words\n\n* According to cosine similarity results;\n\n  + shsy5y has more unique and distinctive morphology.\n  + astro and cort cells are mostly different.\n  + some cort cells have lower similarity score compared to corts.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}