{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":14420,"databundleVersionId":868327,"sourceType":"competition"},{"sourceId":4326728,"sourceType":"datasetVersion","datasetId":2548025}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport imageio\nimport h5py\n%matplotlib inline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:44:19.555212Z","iopub.execute_input":"2025-11-20T11:44:19.555488Z","iopub.status.idle":"2025-11-20T11:44:19.560517Z","shell.execute_reply.started":"2025-11-20T11:44:19.555469Z","shell.execute_reply":"2025-11-20T11:44:19.559698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\npath = '../input/mlcoursechapter3/chapter3'\ndata_path = '../input/recursion-cellular-image-classification'\nsys.path.append(path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:44:38.839523Z","iopub.execute_input":"2025-11-20T11:44:38.839828Z","iopub.status.idle":"2025-11-20T11:44:38.843909Z","shell.execute_reply.started":"2025-11-20T11:44:38.839807Z","shell.execute_reply":"2025-11-20T11:44:38.843061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(f'{data_path}/train.csv')\nprint('Training set dimensions: {0}'.format(train.shape))\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:44:44.072019Z","iopub.execute_input":"2025-11-20T11:44:44.072293Z","iopub.status.idle":"2025-11-20T11:44:44.118483Z","shell.execute_reply.started":"2025-11-20T11:44:44.072273Z","shell.execute_reply":"2025-11-20T11:44:44.117814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['dataset'] ='train'\ntrain['well_type']='treatment'\ntrain['cell_type']=[train['experiment'][i].partition('-')[0] for i in range(train.shape[0])]\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:44:45.819445Z","iopub.execute_input":"2025-11-20T11:44:45.820133Z","iopub.status.idle":"2025-11-20T11:44:45.965647Z","shell.execute_reply.started":"2025-11-20T11:44:45.82011Z","shell.execute_reply":"2025-11-20T11:44:45.964904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_controls = pd.read_csv(f'{data_path}/train_controls.csv')\nprint('Training controls set dimensions: {0}'.format(train_controls.shape))\ntrain_controls.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:44:48.492907Z","iopub.execute_input":"2025-11-20T11:44:48.493413Z","iopub.status.idle":"2025-11-20T11:44:48.509963Z","shell.execute_reply.started":"2025-11-20T11:44:48.493392Z","shell.execute_reply":"2025-11-20T11:44:48.509317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_controls['dataset'] ='train_controls'\ntrain_controls['cell_type']=[train_controls['experiment'][i].partition('-')[0] for i in range(train_controls.shape[0])]\ntrain_controls.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:44:51.170042Z","iopub.execute_input":"2025-11-20T11:44:51.170758Z","iopub.status.idle":"2025-11-20T11:44:51.197269Z","shell.execute_reply.started":"2025-11-20T11:44:51.170714Z","shell.execute_reply":"2025-11-20T11:44:51.196527Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pd.read_csv(f'{data_path}/test.csv')\nprint('Test set dimensions: {0}'.format(test.shape))\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:44:53.284993Z","iopub.execute_input":"2025-11-20T11:44:53.285258Z","iopub.status.idle":"2025-11-20T11:44:53.308284Z","shell.execute_reply.started":"2025-11-20T11:44:53.285239Z","shell.execute_reply":"2025-11-20T11:44:53.307567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test['dataset'] = 'test'\ntest['well_type'] = 'unknow'\ntest['cell_type'] = [test['experiment'][i].partition('-')[0] for i in range(test.shape[0])]\ntest['sirna'] = 'unknow'\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:44:55.239293Z","iopub.execute_input":"2025-11-20T11:44:55.240025Z","iopub.status.idle":"2025-11-20T11:44:55.32725Z","shell.execute_reply.started":"2025-11-20T11:44:55.239999Z","shell.execute_reply":"2025-11-20T11:44:55.326565Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_controls = pd.read_csv(f'{data_path}/test_controls.csv')\nprint('Test controls set dimensions: {0}'.format(test_controls.shape))\ntest_controls.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:44:57.736372Z","iopub.execute_input":"2025-11-20T11:44:57.737166Z","iopub.status.idle":"2025-11-20T11:44:57.751969Z","shell.execute_reply.started":"2025-11-20T11:44:57.737129Z","shell.execute_reply":"2025-11-20T11:44:57.751254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_controls['dataset'] = 'test_controls'\ntest_controls['cell_type'] = [test_controls['experiment'][i].partition('-')[0] for i in range(test_controls.shape[0])]\ntest_controls.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:45:00.055743Z","iopub.execute_input":"2025-11-20T11:45:00.056566Z","iopub.status.idle":"2025-11-20T11:45:00.075338Z","shell.execute_reply.started":"2025-11-20T11:45:00.056533Z","shell.execute_reply":"2025-11-20T11:45:00.074584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"frames=[train,train_controls,test,test_controls]\ncombined = pd.concat(frames,sort=False)\ncombined.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:45:01.905875Z","iopub.execute_input":"2025-11-20T11:45:01.906415Z","iopub.status.idle":"2025-11-20T11:45:01.917533Z","shell.execute_reply.started":"2025-11-20T11:45:01.906392Z","shell.execute_reply":"2025-11-20T11:45:01.916879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"combined.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:45:04.346676Z","iopub.execute_input":"2025-11-20T11:45:04.34698Z","iopub.status.idle":"2025-11-20T11:45:04.372839Z","shell.execute_reply.started":"2025-11-20T11:45:04.346959Z","shell.execute_reply":"2025-11-20T11:45:04.372225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in ['train','test','train_controls','test_controls']:\n    x=combined[combined.dataset==col]['cell_type'].value_counts().index\n    y=combined[combined.dataset==col]['cell_type'].value_counts()\n    plt.bar(x,y,label=col,alpha=0.7)\n    plt.legend()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:45:06.552011Z","iopub.execute_input":"2025-11-20T11:45:06.55258Z","iopub.status.idle":"2025-11-20T11:45:06.773641Z","shell.execute_reply.started":"2025-11-20T11:45:06.552558Z","shell.execute_reply":"2025-11-20T11:45:06.772967Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in ['train', 'test', 'train_controls', 'test_controls']:\n    labels = combined[combined.dataset == col]['sirna'].value_counts()\n    print(\"\\n{0} Label Count: {1}, Top 5 Most Frequent Labels:\\n{2}\"\n          .format(col, len(labels), labels.head(5)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:45:09.421301Z","iopub.execute_input":"2025-11-20T11:45:09.42203Z","iopub.status.idle":"2025-11-20T11:45:09.456761Z","shell.execute_reply.started":"2025-11-20T11:45:09.422007Z","shell.execute_reply":"2025-11-20T11:45:09.455862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"set1 = set(list(combined[combined.dataset == 'train']['sirna'].unique())).intersection(\n       set(list(combined[combined.dataset == 'train_controls']['sirna'].unique())))\nprint('Number of overlapping labels: {0}'.format(len(set1)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:45:13.419016Z","iopub.execute_input":"2025-11-20T11:45:13.419643Z","iopub.status.idle":"2025-11-20T11:45:13.439054Z","shell.execute_reply.started":"2025-11-20T11:45:13.419613Z","shell.execute_reply":"2025-11-20T11:45:13.438393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"set2 = set(list(combined[combined.dataset == 'train_controls']['sirna'].unique())).intersection(\n       set(list(combined[combined.dataset == 'test_controls']['sirna'].unique())))\nprint('Number of overlapping labels: {0}'.format(len(set2)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:45:15.010051Z","iopub.execute_input":"2025-11-20T11:45:15.01062Z","iopub.status.idle":"2025-11-20T11:45:15.025927Z","shell.execute_reply.started":"2025-11-20T11:45:15.010597Z","shell.execute_reply":"2025-11-20T11:45:15.025194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_train = combined[(combined.dataset == 'train')]\nU2OS_train = all_train[all_train['experiment'].str.contains('U2OS')]\nU2OS_train = U2OS_train[['id_code', 'sirna']]\nU2OS_train = U2OS_train.reset_index(drop=True)\nprint(U2OS_train.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:45:16.73168Z","iopub.execute_input":"2025-11-20T11:45:16.732221Z","iopub.status.idle":"2025-11-20T11:45:16.756565Z","shell.execute_reply.started":"2025-11-20T11:45:16.732199Z","shell.execute_reply":"2025-11-20T11:45:16.755927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_image(basepath, id_code, site):\n    images = np.zeros(shape=(6,512,512))\n    path = id_code.partition('_')[0] + '/Plate' + id_code.partition('_')[2][0] \\\n    + '/'+ id_code.rpartition('_')[2]\n    images[0,:,:] = imageio.imread(basepath + path + site + '_w1' + '.png')\n    images[1,:,:] = imageio.imread(basepath + path + site + '_w2' + \".png\")\n    images[2,:,:] = imageio.imread(basepath + path + site + '_w3' + \".png\")\n    images[3,:,:] = imageio.imread(basepath + path + site + '_w4' + \".png\")\n    images[4,:,:] = imageio.imread(basepath + path + site + '_w5' + \".png\")\n    images[5,:,:] = imageio.imread(basepath + path + site + '_w6' + \".png\")\n    return images\nfig, ax = plt.subplots(2,3,figsize=(18,10))\nimages = load_image(f'{data_path}/train/', U2OS_train['id_code'][40], '_s1')\nax[0][0].imshow(images[0], cmap=\"Blues\")\nax[0][1].imshow(images[1], cmap=\"Greens\")\nax[0][2].imshow(images[2], cmap=\"hot\")\nax[1][0].imshow(images[3], cmap=\"viridis\")\nax[1][1].imshow(images[4], cmap=\"gist_heat\")\nax[1][2].imshow(images[5], cmap=\"pink\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:45:18.366415Z","iopub.execute_input":"2025-11-20T11:45:18.366682Z","iopub.status.idle":"2025-11-20T11:45:19.548515Z","shell.execute_reply.started":"2025-11-20T11:45:18.366662Z","shell.execute_reply":"2025-11-20T11:45:19.547383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras import Sequential\nfrom keras.layers import Dense,Activation,Conv2D,MaxPooling2D,Flatten,Dropout\n\nmodel=Sequential(name='VGG-16')\n#BLOCK1\nmodel.add(Conv2D(filters=64,kernel_size=(3,3),activation='relu',\n                padding='same',name='block1_conv1',input_shape=(224,224,3)))\nmodel.add(Conv2D(filters=64,kernel_size=(3,3),activation='relu',\n                padding='same',name='block1_conv2'))\nmodel.add(MaxPooling2D(pool_size=(2,2),strides=(2,2),name='block1_pool'))\n# BLOCK2\nmodel.add(Conv2D(filters = 128, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block2_conv1'))\nmodel.add(Conv2D(filters = 128, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block2_conv2'))\nmodel.add(MaxPooling2D(pool_size = (2, 2), strides = (2, 2), name = 'block2_pool'))\n# BLOCK3\nmodel.add(Conv2D(filters = 256, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block3_conv1'))\nmodel.add(Conv2D(filters = 256, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block3_conv2'))\nmodel.add(Conv2D(filters = 256, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block3_conv3'))\nmodel.add(MaxPooling2D(pool_size = (2, 2), strides = (2, 2), name = 'block3_pool'))\n# BLOCK4\nmodel.add(Conv2D(filters = 512, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block4_conv1'))\nmodel.add(Conv2D(filters = 512, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block4_conv2'))\nmodel.add(Conv2D(filters = 512, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block4_conv3'))\nmodel.add(MaxPooling2D(pool_size = (2, 2), strides = (2, 2), name = 'block4_pool'))\n# BLOCK5\nmodel.add(Conv2D(filters = 512, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block5_conv1'))\nmodel.add(Conv2D(filters = 512, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block5_conv2'))\nmodel.add(Conv2D(filters = 512, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block5_conv3'))\nmodel.add(MaxPooling2D(pool_size = (2, 2), strides = (2, 2), name = 'block5_pool'))\nmodel.add(Flatten())\n#FC1\nmodel.add(Dense(4096, activation = 'relu', name = 'fc1'))\nmodel.add(Dropout(0.5))\n#FC2\nmodel.add(Dense(4096, activation = 'relu', name = 'fc2'))\nmodel.add(Dropout(0.5))\n#Softmax\nmodel.add(Dense(1108, activation = 'softmax', name = 'prediction'))\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:45:22.414952Z","iopub.execute_input":"2025-11-20T11:45:22.415632Z","iopub.status.idle":"2025-11-20T11:45:22.696567Z","shell.execute_reply.started":"2025-11-20T11:45:22.41561Z","shell.execute_reply":"2025-11-20T11:45:22.696023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nimport numpy as np\nfrom skimage.io import imread\nimport pandas as pd\n\nimport tensorflow as tf\n\nDEFAULT_BASE_PATH = '.'\nDEFAULT_METADATA_BASE_PATH = os.path.join(DEFAULT_BASE_PATH, 'metadata')\nDEFAULT_IMAGES_BASE_PATH = os.path.join(DEFAULT_BASE_PATH, 'dataset')\nDEFAULT_CHANNELS = (1, 2, 3, 4, 5, 6)\nRGB_MAP = {\n    1: {\n        'rgb': np.array([19, 0, 249]),\n        'range': [0, 51]\n    },\n    2: {\n        'rgb': np.array([42, 255, 31]),\n        'range': [0, 107]\n    },\n    3: {\n        'rgb': np.array([255, 0, 25]),\n        'range': [0, 64]\n    },\n    4: {\n        'rgb': np.array([45, 255, 252]),\n        'range': [0, 191]\n    },\n    5: {\n        'rgb': np.array([250, 0, 253]),\n        'range': [0, 89]\n    },\n    6: {\n        'rgb': np.array([254, 255, 40]),\n        'range': [0, 191]\n    }\n}\n\ndef load_image(image_path):\n    with tf.io.gfile.GFile(image_path, 'rb') as f:\n        return imread(f) \n\ndef load_images_as_tensor(image_paths, dtype=np.uint8):\n    n_channels = len(image_paths)\n\n    data = np.ndarray(shape=(512, 512, n_channels), dtype=dtype)\n\n    for ix, img_path in enumerate(image_paths):\n        data[:, :, ix] = load_image(img_path)\n\n    return data\n\ndef convert_tensor_to_rgb(t, channels=DEFAULT_CHANNELS, vmax=255, rgb_map=RGB_MAP):\n    \"\"\"\n    Converts and returns the image data as RGB image\n\n    Parameters\n    ----------\n    t : np.ndarray\n        original image data\n    channels : list of int\n        channels to include\n    vmax : int\n        the max value used for scaling\n    rgb_map : dict\n        the color mapping for each channel\n        See rxrx.io.RGB_MAP to see what the defaults are.\n\n    Returns\n    -------\n    np.ndarray the image data of the site as RGB channels\n    \"\"\"\n    colored_channels = []\n    for i, channel in enumerate(channels):\n        x = (t[:, :, i] / vmax) / \\\n            ((rgb_map[channel]['range'][1] - rgb_map[channel]['range'][0]) / 255) + \\\n            rgb_map[channel]['range'][0] / 255\n        x = np.where(x > 1., 1., x)\n        x_rgb = np.array(\n            np.outer(x, rgb_map[channel]['rgb']).reshape(512, 512, 3),\n            dtype=int)\n        colored_channels.append(x_rgb)\n    im = np.array(np.array(colored_channels).sum(axis=0), dtype=int)\n    im = np.where(im > 255, 255, im)\n    return im\n\ndef image_path(dataset,\n               experiment,\n               plate,\n               address,\n               site,\n               channel,\n               base_path=DEFAULT_IMAGES_BASE_PATH):\n    \"\"\"\n    Returns the path of a channel image.\n\n    Parameters\n    ----------\n    dataset : str\n        what subset of the data: train, test\n    experiment : str\n        experiment name\n    plate : int\n        plate number\n    address : str\n        plate address\n    site : int\n        site number\n    channel : int\n        channel number\n    base_path : str\n        the base path of the raw images\n\n    Returns\n    -------\n    str the path of image\n    \"\"\"\n    return os.path.join(base_path, dataset, experiment, \"Plate{}\".format(plate),\n                        \"{}_s{}_w{}.png\".format(address, site, channel))\n\ndef load_site(dataset,\n              experiment,\n              plate,\n              well,\n              site,\n              channels=DEFAULT_CHANNELS,\n              base_path=DEFAULT_IMAGES_BASE_PATH):\n    \"\"\"\n    Returns the image data of a site\n\n    Parameters\n    ----------\n    dataset : str\n        what subset of the data: train, test\n    experiment : str\n        experiment name\n    plate : int\n        plate number\n    address : str\n        plate address\n    site : int\n        site number\n    channels : list of int\n        channels to include\n    base_path : str\n        the base path of the raw images\n\n    Returns\n    -------\n    np.ndarray the image data of the site\n    \"\"\"\n    channel_paths = [\n        image_path(\n            dataset, experiment, plate, well, site, c, base_path=base_path)\n        for c in channels\n    ]\n    return load_images_as_tensor(channel_paths)\n\ndef load_site_as_rgb(dataset,\n                     experiment,\n                     plate,\n                     well,\n                     site,\n                     channels=DEFAULT_CHANNELS,\n                     base_path=DEFAULT_IMAGES_BASE_PATH,\n                     rgb_map=RGB_MAP):\n    \"\"\"\n    Loads and returns the image data as RGB image\n\n    Parameters\n    ----------\n    dataset : str\n        what subset of the data: train, test\n    experiment : str\n        experiment name\n    plate : int\n        plate number\n    address : str\n        plate address\n    site : int\n        site number\n    channels : list of int\n        channels to include\n    base_path : str\n        the base path of the raw images\n    rgb_map : dict\n        the color mapping for each channel\n        See rxrx.io.RGB_MAP to see what the defaults are.\n\n    Returns\n    -------\n    np.ndarray the image data of the site as RGB channels\n    \"\"\"\n    x = load_site(dataset, experiment, plate, well, site, channels, base_path)\n    return convert_tensor_to_rgb(x, channels, rgb_map=rgb_map)\n\ndef _tf_read_csv(path):\n    with tf.io.gfile.GFile(path, 'rb') as f:\n        return pd.read_csv(f)\n\ndef _load_dataset(base_path, dataset, include_controls=True):\n    df = _tf_read_csv(os.path.join(base_path, dataset + '.csv'))\n    if include_controls:\n        controls = _tf_read_csv(\n            os.path.join(base_path, dataset + '_controls.csv'))\n        df['well_type'] = 'treatment'\n        df = pd.concat([controls, df], sort=True)\n    df['cell_type'] = df.experiment.str.split(\"-\").apply(lambda a: a[0])\n    df['dataset'] = dataset\n    dfs = []\n    for site in (1, 2):\n        df = df.copy()\n        df['site'] = site\n        dfs.append(df)\n    res = pd.concat(dfs).sort_values(\n        by=['id_code', 'site']).set_index('id_code')\n    return res\n\ndef combine_metadata(base_path=DEFAULT_METADATA_BASE_PATH,\n                     include_controls=True):\n    \"\"\"\n    Combines all metadata files into a single dataframe and\n    expands it to include sites, not just wells.\n\n    Note, that the dtype of sirna is a float due to the missing\n    test values but it should be treated as an int.\n\n    Parameters\n    ----------\n    base_path : str\n        where the metadata files from Kaggle live\n    include_controls : bool\n        indicate if you want the controls included in the dataframe\n\n    Returns\n    -------\n    pandas.DataFrame the combined metadata\n    \"\"\"\n    df = pd.concat(\n        [\n            _load_dataset(\n                base_path, dataset, include_controls=include_controls)\n            for dataset in ['test', 'train']\n        ],\n        sort=True)\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:45:27.128657Z","iopub.execute_input":"2025-11-20T11:45:27.129307Z","iopub.status.idle":"2025-11-20T11:45:27.144536Z","shell.execute_reply.started":"2025-11-20T11:45:27.129282Z","shell.execute_reply":"2025-11-20T11:45:27.143909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nsys.path.append('../input/mlcoursechapter3/chapter3/rxrx1utils')\nimport rxrx.io as rio \ncell_image = load_site_as_rgb('train', 'HUVEC-01', 3, 'K09', 2, base_path=data_path)\nplt.figure(figsize=(5, 5))\nplt.imshow(cell_image)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:45:38.169216Z","iopub.execute_input":"2025-11-20T11:45:38.16992Z","iopub.status.idle":"2025-11-20T11:45:38.55121Z","shell.execute_reply.started":"2025-11-20T11:45:38.169897Z","shell.execute_reply":"2025-11-20T11:45:38.550427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nresized = tf.image.resize(cell_image,(224,224))\nresized = np.asarray(resized, dtype='uint8')\nplt.imshow(resized)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:45:42.779988Z","iopub.execute_input":"2025-11-20T11:45:42.780257Z","iopub.status.idle":"2025-11-20T11:45:42.989338Z","shell.execute_reply.started":"2025-11-20T11:45:42.780238Z","shell.execute_reply":"2025-11-20T11:45:42.98855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sirna_code = pd.get_dummies(U2OS_train, columns = ['sirna'])\ntrain_data_x = U2OS_train\ntrain_data_y = sirna_code.drop(['id_code'], axis=1)\nprint(train_data_x.shape)\nprint(train_data_y.shape)\ntrain_data_y.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:45:45.384941Z","iopub.execute_input":"2025-11-20T11:45:45.385541Z","iopub.status.idle":"2025-11-20T11:45:45.408463Z","shell.execute_reply.started":"2025-11-20T11:45:45.385518Z","shell.execute_reply":"2025-11-20T11:45:45.407689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_train_id, x_val_id, Y_train, Y_val = train_test_split(\n    train_data_x,        \n    train_data_y,        \n    test_size = 0.33,     \n    random_state = 0     \n)\nprint('Training set feature dimension: {0}, label dimension: {1}'.format(x_train_id.shape, Y_train.shape))\nprint('Validation set feature dimension: {0}, label dimension: {1}'.format(x_val_id.shape, Y_val.shape))\nx_train_id.reset_index(drop=True, inplace=True)\nY_train.reset_index(drop=True, inplace=True)\nx_val_id.reset_index(drop=True, inplace=True)\nY_val.reset_index(drop=True, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:45:49.323447Z","iopub.execute_input":"2025-11-20T11:45:49.324131Z","iopub.status.idle":"2025-11-20T11:45:49.337112Z","shell.execute_reply.started":"2025-11-20T11:45:49.324106Z","shell.execute_reply":"2025-11-20T11:45:49.336379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfull_sirna_column = pd.concat([x_train_id['sirna'], x_val_id['sirna']], axis=0)\ninteger_sirna_ids = full_sirna_column.str.split('_').str[1].astype(np.int32)\nclasses = np.sort(integer_sirna_ids.unique())\nprint(f\"Classes successfully defined. Shape: {classes.shape}, First 10: {classes[:10]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:45:57.372068Z","iopub.execute_input":"2025-11-20T11:45:57.372773Z","iopub.status.idle":"2025-11-20T11:45:57.382076Z","shell.execute_reply.started":"2025-11-20T11:45:57.372724Z","shell.execute_reply":"2025-11-20T11:45:57.381202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%mkdir ./cell","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:45:59.802139Z","iopub.execute_input":"2025-11-20T11:45:59.802413Z","iopub.status.idle":"2025-11-20T11:45:59.948514Z","shell.execute_reply.started":"2025-11-20T11:45:59.802392Z","shell.execute_reply":"2025-11-20T11:45:59.947706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ndef make_dataset(x_data,y_data,x_data_id,Y_data):\n    i = 0\n    for code in x_data_id['id_code']:\n        cell_type = code.partition('_')[0] \n        plate = int(code.partition('_')[2][0]) \n        well = code.rpartition('_')[2] \n        img = load_site_as_rgb('train', cell_type, plate, well, 1, base_path=data_path)\n        resized = tf.image.resize(img,(224,224)) \n        resized = np.asarray(resized, dtype='uint8')\n        filename = './cell/'+code+'.jpg'\n\n        plt.imsave(filename,resized) \n        x_data[i] = resized \n        y_data[i]=Y_data.loc[i].to_numpy() \n        i += 1\n        \nx_train_shape = (x_train_id.shape[0],224,224,3) \nx_val_shape = (x_val_id.shape[0],224,224,3) \n\nwith h5py.File('cell.h5','w') as f:\n\n    x_train = f.create_dataset(\"x_train\", x_train_shape ,'u1') \n    y_train = f.create_dataset(\"y_train\", Y_train.shape ,'i4') \n    x_val = f.create_dataset(\"x_val\", x_val_shape ,'u1') \n    y_val = f.create_dataset(\"y_val\", Y_val.shape ,'i4') \n    classes = f.create_dataset('classes', data = classes)\n    \n    make_dataset(x_train,y_train,x_train_id,Y_train)\n    print('x_train finished')\n    make_dataset(x_val,y_val,x_val_id,Y_val)\n    print('x_val finished')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:46:01.676984Z","iopub.execute_input":"2025-11-20T11:46:01.677825Z","iopub.status.idle":"2025-11-20T11:55:40.824608Z","shell.execute_reply.started":"2025-11-20T11:46:01.677784Z","shell.execute_reply":"2025-11-20T11:55:40.823816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras.applications.vgg16 import VGG16\nfrom keras.preprocessing import image\nfrom keras.applications.vgg16 import preprocess_input\nmodel = VGG16(weights='imagenet', include_top=False)\ndef VGG16_extract_features(img):\n    x = np.expand_dims(img, axis=0)\n    features = model.predict(x)\n    return features\nfeatures = VGG16_extract_features(resized)\nfeatures.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:55:48.104551Z","iopub.execute_input":"2025-11-20T11:55:48.10531Z","iopub.status.idle":"2025-11-20T11:55:52.826506Z","shell.execute_reply.started":"2025-11-20T11:55:48.105282Z","shell.execute_reply":"2025-11-20T11:55:52.825759Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_dataset():\n    with h5py.File('cell.h5','r') as f:\n        x_train = f['x_train'][:]\n        y_train = f['y_train'][:]\n        x_val = f['x_val'][:] \n        y_val = f['y_val'][:]\n        classes = f['classes'][:]\n        return x_train, y_train, x_val, y_val, classes\nX_train, Y_train, X_val, Y_val, classes = load_dataset()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:55:57.777983Z","iopub.execute_input":"2025-11-20T11:55:57.778509Z","iopub.status.idle":"2025-11-20T11:55:58.117162Z","shell.execute_reply.started":"2025-11-20T11:55:57.778487Z","shell.execute_reply":"2025-11-20T11:55:58.116553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nm = X_train.shape[0]\nx_train = np.zeros((m,7,7,512))\nfor i in range(m):\n    x_train[i] = VGG16_extract_features(X_train[i])\nprint(x_train.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T11:56:02.762926Z","iopub.execute_input":"2025-11-20T11:56:02.763674Z","iopub.status.idle":"2025-11-20T11:58:31.198498Z","shell.execute_reply.started":"2025-11-20T11:56:02.763648Z","shell.execute_reply":"2025-11-20T11:58:31.19781Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nm = X_val.shape[0]\nx_val = np.zeros((m,7,7,512))\nfor i in range(m):\n    x_val[i] = VGG16_extract_features(X_val[i])\nprint(x_val.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T12:00:58.502227Z","iopub.execute_input":"2025-11-20T12:00:58.503058Z","iopub.status.idle":"2025-11-20T12:02:13.258286Z","shell.execute_reply.started":"2025-11-20T12:00:58.50303Z","shell.execute_reply":"2025-11-20T12:02:13.257797Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with h5py.File('cell_features.h5','w') as f:\n    x_train = f.create_dataset(\"x_train\", data=x_train)\n    y_train = f.create_dataset(\"y_train\", data=Y_train) \n    x_val = f.create_dataset(\"x_val\", data=x_val) \n    y_val = f.create_dataset(\"y_val\", data=Y_val) \n    classes = f.create_dataset('classes',data = classes) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T12:02:30.431138Z","iopub.execute_input":"2025-11-20T12:02:30.431485Z","iopub.status.idle":"2025-11-20T12:02:30.982231Z","shell.execute_reply.started":"2025-11-20T12:02:30.431463Z","shell.execute_reply":"2025-11-20T12:02:30.981641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import h5py\ndef load_VGG16_dataset():\n    with h5py.File('cell_features.h5','r') as f:\n        x_train = f['x_train'][:] \n        y_train = f['y_train'][:]\n        x_val = f['x_val'][:] \n        y_val = f['y_val'][:] \n        classes = f['classes'][:]\n        return x_train, y_train, x_val, y_val, classes\nX_train, Y_train, X_val, Y_val, classes = load_VGG16_dataset()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T12:02:32.062495Z","iopub.execute_input":"2025-11-20T12:02:32.062814Z","iopub.status.idle":"2025-11-20T12:02:32.556562Z","shell.execute_reply.started":"2025-11-20T12:02:32.062789Z","shell.execute_reply":"2025-11-20T12:02:32.555966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras import Sequential\nfrom keras.layers import Dense, Conv2D, Flatten, Dropout\nmodel = Sequential(name = 'VGG16-Cell-Transfer')\nmodel.add(Conv2D(filters = 64, kernel_size = (1, 1), activation = 'relu',\npadding = 'same', input_shape = (7, 7, 512)))\n\nmodel.add(Flatten())\nmodel.add(Dense(2048, activation = 'relu', name = 'fc1'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(2048, activation = 'relu', name = 'fc2'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(1108, activation = 'softmax', name = 'prediction'))\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T12:02:33.957188Z","iopub.execute_input":"2025-11-20T12:02:33.957472Z","iopub.status.idle":"2025-11-20T12:02:34.032157Z","shell.execute_reply.started":"2025-11-20T12:02:33.957451Z","shell.execute_reply":"2025-11-20T12:02:34.031427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.32\nmodel.compile(optimizer='RMSProp',loss='categorical_crossentropy',metrics=['accuracy'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T12:02:37.065348Z","iopub.execute_input":"2025-11-20T12:02:37.065955Z","iopub.status.idle":"2025-11-20T12:02:37.079462Z","shell.execute_reply.started":"2025-11-20T12:02:37.065934Z","shell.execute_reply":"2025-11-20T12:02:37.07879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"epochs = 20\nbatch_size = 32\nhistory = model.fit(X_train, Y_train, epochs=epochs, batch_size=batch_size,validation_data=(X_val,Y_val))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T12:02:40.741787Z","iopub.execute_input":"2025-11-20T12:02:40.742112Z","iopub.status.idle":"2025-11-20T12:02:56.790439Z","shell.execute_reply.started":"2025-11-20T12:02:40.742091Z","shell.execute_reply":"2025-11-20T12:02:56.789893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n%matplotlib inline\nx = range(1, len(history.history['accuracy'])+1)\nplt.plot(x, history.history['accuracy'])\nplt.plot(x, history.history['val_accuracy'])\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.xticks(x)\nplt.legend(['Train', 'Val'], loc='upper left')\nplt.show()\nplt.plot(x, history.history['loss'])\nplt.plot(x, history.history['val_loss'])\nplt.title('Model loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.xticks(x)\nplt.legend(['Train', 'Val'], loc='lower left')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T12:12:35.074147Z","iopub.execute_input":"2025-11-20T12:12:35.074968Z","iopub.status.idle":"2025-11-20T12:12:35.545991Z","shell.execute_reply.started":"2025-11-20T12:12:35.074943Z","shell.execute_reply":"2025-11-20T12:12:35.54517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import h5py\ndef load_dataset():\n    with h5py.File('cell.h5','r') as f:\n        x_train = f['x_train'][:] \n        y_train = f['y_train'][:]\n        x_val = f['x_val'][:] \n        y_val = f['y_val'][:] \n        classes = f['classes'][:] \n        return x_train, y_train, x_val, y_val, classes\nX_train, Y_train, X_val, Y_val, classes = load_dataset()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T12:12:44.35069Z","iopub.execute_input":"2025-11-20T12:12:44.351385Z","iopub.status.idle":"2025-11-20T12:12:44.693104Z","shell.execute_reply.started":"2025-11-20T12:12:44.351363Z","shell.execute_reply":"2025-11-20T12:12:44.692501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, Y_train, X_val, Y_val, classes = load_dataset()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T12:12:45.283074Z","iopub.execute_input":"2025-11-20T12:12:45.283317Z","iopub.status.idle":"2025-11-20T12:12:45.667525Z","shell.execute_reply.started":"2025-11-20T12:12:45.283301Z","shell.execute_reply":"2025-11-20T12:12:45.666935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras import Sequential\nfrom keras.layers import Dense, Conv2D, Flatten, Dropout","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T12:12:46.648454Z","iopub.execute_input":"2025-11-20T12:12:46.649145Z","iopub.status.idle":"2025-11-20T12:12:46.652578Z","shell.execute_reply.started":"2025-11-20T12:12:46.649122Z","shell.execute_reply":"2025-11-20T12:12:46.651795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}