{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":14420,"databundleVersionId":868327,"sourceType":"competition"},{"sourceId":4326728,"sourceType":"datasetVersion","datasetId":2548025}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#P3.1 导入库\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport imageio\nimport h5py\n%matplotlib inline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:55.424492Z","iopub.execute_input":"2024-12-03T01:14:55.425778Z","iopub.status.idle":"2024-12-03T01:14:56.516395Z","shell.execute_reply.started":"2024-12-03T01:14:55.425723Z","shell.execute_reply":"2024-12-03T01:14:56.515228Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\n\npath = '../input/mlcoursechapter3/chapter3'\n# data_path = '../input/mlcoursechapter3/chapter3/dataset'\ndata_path = '../input/recursion-cellular-image-classification'\nsys.path.append(path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:56.518357Z","iopub.execute_input":"2024-12-03T01:14:56.518874Z","iopub.status.idle":"2024-12-03T01:14:56.525113Z","shell.execute_reply.started":"2024-12-03T01:14:56.518836Z","shell.execute_reply":"2024-12-03T01:14:56.523877Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.2 训练集观察train.csv\ntrain = pd.read_csv(f'{data_path}/train.csv')\nprint('训练集的维度：{0}'.format(train.shape))\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:56.527051Z","iopub.execute_input":"2024-12-03T01:14:56.527618Z","iopub.status.idle":"2024-12-03T01:14:56.600891Z","shell.execute_reply.started":"2024-12-03T01:14:56.527543Z","shell.execute_reply":"2024-12-03T01:14:56.599634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.3 训练集train扩增三列\ntrain['dataset'] ='train'\ntrain['well_type']='treatment'\ntrain['cell_type']=[train['experiment'][i].partition('-')[0] for i in range(train.shape[0])]\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:56.603808Z","iopub.execute_input":"2024-12-03T01:14:56.604174Z","iopub.status.idle":"2024-12-03T01:14:56.843112Z","shell.execute_reply.started":"2024-12-03T01:14:56.604139Z","shell.execute_reply":"2024-12-03T01:14:56.841887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.4 对train_control.csv做观察\ntrain_controls = pd.read_csv(f'{data_path}/train_controls.csv')\nprint('训练对照集的维度：{0}'.format(train_controls.shape))\ntrain_controls.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:56.844399Z","iopub.execute_input":"2024-12-03T01:14:56.844763Z","iopub.status.idle":"2024-12-03T01:14:56.870218Z","shell.execute_reply.started":"2024-12-03T01:14:56.844728Z","shell.execute_reply":"2024-12-03T01:14:56.869121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.5 实验对照集扩增两列\ntrain_controls['dataset'] ='train_controls'\n\ntrain_controls['cell_type']=[train_controls['experiment'][i].partition('-')[0] for i in range(train_controls.shape[0])]\ntrain_controls.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:56.871700Z","iopub.execute_input":"2024-12-03T01:14:56.872102Z","iopub.status.idle":"2024-12-03T01:14:56.914939Z","shell.execute_reply.started":"2024-12-03T01:14:56.872068Z","shell.execute_reply":"2024-12-03T01:14:56.913535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# P3.6 测试集test观察\ntest = pd.read_csv(f'{data_path}/test.csv')\nprint('测试集维度：{0}'.format(test.shape))\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:56.916347Z","iopub.execute_input":"2024-12-03T01:14:56.916725Z","iopub.status.idle":"2024-12-03T01:14:56.950694Z","shell.execute_reply.started":"2024-12-03T01:14:56.916690Z","shell.execute_reply":"2024-12-03T01:14:56.949639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# P3.7 测试集test扩增四列\ntest['dataset'] = 'test'\ntest['well_type'] = 'unknow'\ntest['cell_type'] = [test['experiment'][i].partition('-')[0] for i in range(test.shape[0])]\ntest['sirna'] = 'unknow'\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:56.951976Z","iopub.execute_input":"2024-12-03T01:14:56.952299Z","iopub.status.idle":"2024-12-03T01:14:57.096121Z","shell.execute_reply.started":"2024-12-03T01:14:56.952265Z","shell.execute_reply":"2024-12-03T01:14:57.094836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# P3.8 测试对照集test_controls观察\ntest_controls = pd.read_csv(f'{data_path}/test_controls.csv')\nprint('测试对照集维度：{0}'.format(test_controls.shape))\ntest_controls.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:57.097492Z","iopub.execute_input":"2024-12-03T01:14:57.097920Z","iopub.status.idle":"2024-12-03T01:14:57.123615Z","shell.execute_reply.started":"2024-12-03T01:14:57.097872Z","shell.execute_reply":"2024-12-03T01:14:57.122551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# P3.9 测试对照集test_controls扩增两列\ntest_controls['dataset'] = 'test_controls'\ntest_controls['cell_type'] = [test_controls['experiment'][i].partition('-')[0] for i in range(test_controls.shape[0])]\ntest_controls.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:57.129275Z","iopub.execute_input":"2024-12-03T01:14:57.129829Z","iopub.status.idle":"2024-12-03T01:14:57.167119Z","shell.execute_reply.started":"2024-12-03T01:14:57.129761Z","shell.execute_reply":"2024-12-03T01:14:57.165918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.10 四个数据集文件合并\nframes=[train,train_controls,test,test_controls]\ncombined = pd.concat(frames,sort=False)\ncombined.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:57.168430Z","iopub.execute_input":"2024-12-03T01:14:57.168915Z","iopub.status.idle":"2024-12-03T01:14:57.197699Z","shell.execute_reply.started":"2024-12-03T01:14:57.168861Z","shell.execute_reply":"2024-12-03T01:14:57.196201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.11 缺失值检查\ncombined.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:57.199155Z","iopub.execute_input":"2024-12-03T01:14:57.199523Z","iopub.status.idle":"2024-12-03T01:14:57.239678Z","shell.execute_reply.started":"2024-12-03T01:14:57.199483Z","shell.execute_reply":"2024-12-03T01:14:57.238428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.12 细胞系的类型分布统计\nfor col in ['train','test','train_controls','test_controls']:\n    x=combined[combined.dataset==col]['cell_type'].value_counts().index\n    y=combined[combined.dataset==col]['cell_type'].value_counts()\n    plt.bar(x,y,label=col,alpha=0.7)\n    plt.legend()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:57.241032Z","iopub.execute_input":"2024-12-03T01:14:57.241377Z","iopub.status.idle":"2024-12-03T01:14:57.650644Z","shell.execute_reply.started":"2024-12-03T01:14:57.241344Z","shell.execute_reply":"2024-12-03T01:14:57.649341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.13 标签分布统计\nfor col in ['train','test','train_controls','test_controls']:\n    labels=combined[combined.dataset==col]['sirna'].value_counts()\n    print(\"\\n{0}标签数：{1}，重复数前五名的标签为：\\n{2}\"\n          .format(col,len(labels),labels.head(5)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:57.652821Z","iopub.execute_input":"2024-12-03T01:14:57.653399Z","iopub.status.idle":"2024-12-03T01:14:57.699720Z","shell.execute_reply.started":"2024-12-03T01:14:57.653348Z","shell.execute_reply":"2024-12-03T01:14:57.698503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.14 train标签与train_controls标签是否有重复？\nset1=set(list(combined[combined.dataset=='train']['sirna'].unique())).intersection \\\n(set(list(combined[combined.dataset=='train_controls']['sirna'].unique())))\nprint('重复标签数为：{0}'.format(len(set1)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:57.701157Z","iopub.execute_input":"2024-12-03T01:14:57.701512Z","iopub.status.idle":"2024-12-03T01:14:57.728623Z","shell.execute_reply.started":"2024-12-03T01:14:57.701477Z","shell.execute_reply":"2024-12-03T01:14:57.727466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.13 train_controls标签是否与test_controls标签重复？\nset2=set(list(combined[combined.dataset=='train_controls']['sirna'].unique())).intersection \\\n(set(list(combined[combined.dataset=='test_controls']['sirna'].unique())))\nprint('重复标签数为：{0}'.format(len(set2)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:57.730273Z","iopub.execute_input":"2024-12-03T01:14:57.730664Z","iopub.status.idle":"2024-12-03T01:14:57.751621Z","shell.execute_reply.started":"2024-12-03T01:14:57.730627Z","shell.execute_reply":"2024-12-03T01:14:57.750419Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# P3.16 筛选用于训练和验证的数据集\nall_train = combined[(combined.dataset == 'train')]\nU2OS_train = all_train[all_train['experiment'].str.contains('U2OS')]\nU2OS_train = U2OS_train[['id_code', 'sirna']]\nU2OS_train = U2OS_train.reset_index(drop=True)\nprint(U2OS_train.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:57.753489Z","iopub.execute_input":"2024-12-03T01:14:57.753882Z","iopub.status.idle":"2024-12-03T01:14:57.783721Z","shell.execute_reply.started":"2024-12-03T01:14:57.753848Z","shell.execute_reply":"2024-12-03T01:14:57.782545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# %ls /kaggle/input/recursion-cellular-image-classification/train/U2OS-01/Plate1/D05_s1_w1.png","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:57.785104Z","iopub.execute_input":"2024-12-03T01:14:57.785406Z","iopub.status.idle":"2024-12-03T01:14:57.790593Z","shell.execute_reply.started":"2024-12-03T01:14:57.785376Z","shell.execute_reply":"2024-12-03T01:14:57.789476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# P3.17 显示指定Id对应的site1位置的6通道图像\ndef load_image(basepath, id_code, site):\n    images = np.zeros(shape=(6,512,512))\n    path = id_code.partition('_')[0] + '/Plate' + id_code.partition('_')[2][0] \\\n    + '/'+ id_code.rpartition('_')[2]\n    images[0,:,:] = imageio.imread(basepath + path + site + '_w1' + '.png')\n    images[1,:,:] = imageio.imread(basepath + path + site + '_w2' + \".png\")\n    images[2,:,:] = imageio.imread(basepath + path + site + '_w3' + \".png\")\n    images[3,:,:] = imageio.imread(basepath + path + site + '_w4' + \".png\")\n    images[4,:,:] = imageio.imread(basepath + path + site + '_w5' + \".png\")\n    images[5,:,:] = imageio.imread(basepath + path + site + '_w6' + \".png\")\n    return images\nfig, ax = plt.subplots(2,3,figsize=(18,10))\nimages = load_image(f'{data_path}/train/', U2OS_train['id_code'][40], '_s1')\nax[0][0].imshow(images[0], cmap=\"Blues\")\nax[0][1].imshow(images[1], cmap=\"Greens\")\nax[0][2].imshow(images[2], cmap=\"hot\")\nax[1][0].imshow(images[3], cmap=\"viridis\")\nax[1][1].imshow(images[4], cmap=\"gist_heat\")\nax[1][2].imshow(images[5], cmap=\"pink\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:57.791873Z","iopub.execute_input":"2024-12-03T01:14:57.792184Z","iopub.status.idle":"2024-12-03T01:14:59.424805Z","shell.execute_reply.started":"2024-12-03T01:14:57.792153Z","shell.execute_reply":"2024-12-03T01:14:59.423667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.18 标准VGG-16模型的定义\nfrom keras import Sequential\nfrom keras.layers import Dense,Activation,Conv2D,MaxPooling2D,Flatten,Dropout\n\nmodel=Sequential(name='VGG-16')\n#BLOCK1\nmodel.add(Conv2D(filters=64,kernel_size=(3,3),activation='relu',\n                padding='same',name='block1_conv1',input_shape=(224,224,3)))\nmodel.add(Conv2D(filters=64,kernel_size=(3,3),activation='relu',\n                padding='same',name='block1_conv2'))\nmodel.add(MaxPooling2D(pool_size=(2,2),strides=(2,2),name='block1_pool'))\n# BLOCK2\nmodel.add(Conv2D(filters = 128, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block2_conv1'))\nmodel.add(Conv2D(filters = 128, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block2_conv2'))\nmodel.add(MaxPooling2D(pool_size = (2, 2), strides = (2, 2), name = 'block2_pool'))\n# BLOCK3\nmodel.add(Conv2D(filters = 256, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block3_conv1'))\nmodel.add(Conv2D(filters = 256, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block3_conv2'))\nmodel.add(Conv2D(filters = 256, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block3_conv3'))\nmodel.add(MaxPooling2D(pool_size = (2, 2), strides = (2, 2), name = 'block3_pool'))\n# BLOCK4\nmodel.add(Conv2D(filters = 512, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block4_conv1'))\nmodel.add(Conv2D(filters = 512, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block4_conv2'))\nmodel.add(Conv2D(filters = 512, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block4_conv3'))\nmodel.add(MaxPooling2D(pool_size = (2, 2), strides = (2, 2), name = 'block4_pool'))\n# BLOCK5\nmodel.add(Conv2D(filters = 512, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block5_conv1'))\nmodel.add(Conv2D(filters = 512, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block5_conv2'))\nmodel.add(Conv2D(filters = 512, kernel_size = (3, 3), activation = 'relu', \n                 padding = 'same', name = 'block5_conv3'))\nmodel.add(MaxPooling2D(pool_size = (2, 2), strides = (2, 2), name = 'block5_pool'))\nmodel.add(Flatten())\n#FC1\nmodel.add(Dense(4096, activation = 'relu', name = 'fc1'))\nmodel.add(Dropout(0.5))\n#FC2\nmodel.add(Dense(4096, activation = 'relu', name = 'fc2'))\nmodel.add(Dropout(0.5))\n#Softmax\nmodel.add(Dense(1108, activation = 'softmax', name = 'prediction'))\nmodel.summary()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:14:59.426389Z","iopub.execute_input":"2024-12-03T01:14:59.426826Z","iopub.status.idle":"2024-12-03T01:15:03.910078Z","shell.execute_reply.started":"2024-12-03T01:14:59.426766Z","shell.execute_reply":"2024-12-03T01:15:03.908778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# os.path.join(base_path, dataset, experiment, \"Plate{}\".format(plate),\n#                         \"{}_s{}_w{}.png\".format(address, site, channel))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:15:03.911783Z","iopub.execute_input":"2024-12-03T01:15:03.912417Z","iopub.status.idle":"2024-12-03T01:15:03.917733Z","shell.execute_reply.started":"2024-12-03T01:15:03.912367Z","shell.execute_reply":"2024-12-03T01:15:03.916343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nimport numpy as np\nfrom skimage.io import imread\nimport pandas as pd\n\nimport tensorflow as tf\n\nDEFAULT_BASE_PATH = '.'\nDEFAULT_METADATA_BASE_PATH = os.path.join(DEFAULT_BASE_PATH, 'metadata')\nDEFAULT_IMAGES_BASE_PATH = os.path.join(DEFAULT_BASE_PATH, 'dataset')\nDEFAULT_CHANNELS = (1, 2, 3, 4, 5, 6)\nRGB_MAP = {\n    1: {\n        'rgb': np.array([19, 0, 249]),\n        'range': [0, 51]\n    },\n    2: {\n        'rgb': np.array([42, 255, 31]),\n        'range': [0, 107]\n    },\n    3: {\n        'rgb': np.array([255, 0, 25]),\n        'range': [0, 64]\n    },\n    4: {\n        'rgb': np.array([45, 255, 252]),\n        'range': [0, 191]\n    },\n    5: {\n        'rgb': np.array([250, 0, 253]),\n        'range': [0, 89]\n    },\n    6: {\n        'rgb': np.array([254, 255, 40]),\n        'range': [0, 191]\n    }\n}\n\n\ndef load_image(image_path):\n    with tf.io.gfile.GFile(image_path, 'rb') as f:\n        return imread(f) \n        #, format='png'\n\n\ndef load_images_as_tensor(image_paths, dtype=np.uint8):\n    n_channels = len(image_paths)\n\n    data = np.ndarray(shape=(512, 512, n_channels), dtype=dtype)\n\n    for ix, img_path in enumerate(image_paths):\n        data[:, :, ix] = load_image(img_path)\n\n    return data\n\n\ndef convert_tensor_to_rgb(t, channels=DEFAULT_CHANNELS, vmax=255, rgb_map=RGB_MAP):\n    \"\"\"\n    Converts and returns the image data as RGB image\n\n    Parameters\n    ----------\n    t : np.ndarray\n        original image data\n    channels : list of int\n        channels to include\n    vmax : int\n        the max value used for scaling\n    rgb_map : dict\n        the color mapping for each channel\n        See rxrx.io.RGB_MAP to see what the defaults are.\n\n    Returns\n    -------\n    np.ndarray the image data of the site as RGB channels\n    \"\"\"\n    colored_channels = []\n    for i, channel in enumerate(channels):\n        x = (t[:, :, i] / vmax) / \\\n            ((rgb_map[channel]['range'][1] - rgb_map[channel]['range'][0]) / 255) + \\\n            rgb_map[channel]['range'][0] / 255\n        x = np.where(x > 1., 1., x)\n        x_rgb = np.array(\n            np.outer(x, rgb_map[channel]['rgb']).reshape(512, 512, 3),\n            dtype=int)\n        colored_channels.append(x_rgb)\n    im = np.array(np.array(colored_channels).sum(axis=0), dtype=int)\n    im = np.where(im > 255, 255, im)\n    return im\n\n\ndef image_path(dataset,\n               experiment,\n               plate,\n               address,\n               site,\n               channel,\n               base_path=DEFAULT_IMAGES_BASE_PATH):\n    \"\"\"\n    Returns the path of a channel image.\n\n    Parameters\n    ----------\n    dataset : str\n        what subset of the data: train, test\n    experiment : str\n        experiment name\n    plate : int\n        plate number\n    address : str\n        plate address\n    site : int\n        site number\n    channel : int\n        channel number\n    base_path : str\n        the base path of the raw images\n\n    Returns\n    -------\n    str the path of image\n    \"\"\"\n    return os.path.join(base_path, dataset, experiment, \"Plate{}\".format(plate),\n                        \"{}_s{}_w{}.png\".format(address, site, channel))\n\n\ndef load_site(dataset,\n              experiment,\n              plate,\n              well,\n              site,\n              channels=DEFAULT_CHANNELS,\n              base_path=DEFAULT_IMAGES_BASE_PATH):\n    \"\"\"\n    Returns the image data of a site\n\n    Parameters\n    ----------\n    dataset : str\n        what subset of the data: train, test\n    experiment : str\n        experiment name\n    plate : int\n        plate number\n    address : str\n        plate address\n    site : int\n        site number\n    channels : list of int\n        channels to include\n    base_path : str\n        the base path of the raw images\n\n    Returns\n    -------\n    np.ndarray the image data of the site\n    \"\"\"\n    channel_paths = [\n        image_path(\n            dataset, experiment, plate, well, site, c, base_path=base_path)\n        for c in channels\n    ]\n    return load_images_as_tensor(channel_paths)\n\n\ndef load_site_as_rgb(dataset,\n                     experiment,\n                     plate,\n                     well,\n                     site,\n                     channels=DEFAULT_CHANNELS,\n                     base_path=DEFAULT_IMAGES_BASE_PATH,\n                     rgb_map=RGB_MAP):\n    \"\"\"\n    Loads and returns the image data as RGB image\n\n    Parameters\n    ----------\n    dataset : str\n        what subset of the data: train, test\n    experiment : str\n        experiment name\n    plate : int\n        plate number\n    address : str\n        plate address\n    site : int\n        site number\n    channels : list of int\n        channels to include\n    base_path : str\n        the base path of the raw images\n    rgb_map : dict\n        the color mapping for each channel\n        See rxrx.io.RGB_MAP to see what the defaults are.\n\n    Returns\n    -------\n    np.ndarray the image data of the site as RGB channels\n    \"\"\"\n    x = load_site(dataset, experiment, plate, well, site, channels, base_path)\n    return convert_tensor_to_rgb(x, channels, rgb_map=rgb_map)\n\n\ndef _tf_read_csv(path):\n    with tf.io.gfile.GFile(path, 'rb') as f:\n        return pd.read_csv(f)\n\n\ndef _load_dataset(base_path, dataset, include_controls=True):\n    df = _tf_read_csv(os.path.join(base_path, dataset + '.csv'))\n    if include_controls:\n        controls = _tf_read_csv(\n            os.path.join(base_path, dataset + '_controls.csv'))\n        df['well_type'] = 'treatment'\n        df = pd.concat([controls, df], sort=True)\n    df['cell_type'] = df.experiment.str.split(\"-\").apply(lambda a: a[0])\n    df['dataset'] = dataset\n    dfs = []\n    for site in (1, 2):\n        df = df.copy()\n        df['site'] = site\n        dfs.append(df)\n    res = pd.concat(dfs).sort_values(\n        by=['id_code', 'site']).set_index('id_code')\n    return res\n\n\ndef combine_metadata(base_path=DEFAULT_METADATA_BASE_PATH,\n                     include_controls=True):\n    \"\"\"\n    Combines all metadata files into a single dataframe and\n    expands it to include sites, not just wells.\n\n    Note, that the dtype of sirna is a float due to the missing\n    test values but it should be treated as an int.\n\n    Parameters\n    ----------\n    base_path : str\n        where the metadata files from Kaggle live\n    include_controls : bool\n        indicate if you want the controls included in the dataframe\n\n    Returns\n    -------\n    pandas.DataFrame the combined metadata\n    \"\"\"\n    df = pd.concat(\n        [\n            _load_dataset(\n                base_path, dataset, include_controls=include_controls)\n            for dataset in ['test', 'train']\n        ],\n        sort=True)\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:15:03.920218Z","iopub.execute_input":"2024-12-03T01:15:03.920722Z","iopub.status.idle":"2024-12-03T01:15:04.582004Z","shell.execute_reply.started":"2024-12-03T01:15:03.920668Z","shell.execute_reply":"2024-12-03T01:15:04.580991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.19\nimport sys\nsys.path.append('../input/mlcoursechapter3/chapter3/rxrx1utils')\nimport rxrx.io as rio #导入 rxrx工具包\n#加载并合成细胞图像，指定数据集、实验批次板微孔和位置参rio.\ncell_image = load_site_as_rgb('train', 'HUVEC-01', 3, 'K09', 2, base_path=data_path)\nplt.figure(figsize=(5, 5))\nplt.imshow(cell_image)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:15:04.583197Z","iopub.execute_input":"2024-12-03T01:15:04.583806Z","iopub.status.idle":"2024-12-03T01:15:05.041943Z","shell.execute_reply.started":"2024-12-03T01:15:04.583771Z","shell.execute_reply":"2024-12-03T01:15:05.040664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.20\nimport tensorflow as tf\nresized = tf.image.resize(cell_image,(224,224))\nresized = np.asarray(resized, dtype='uint8')\nplt.imshow(resized)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:15:05.043584Z","iopub.execute_input":"2024-12-03T01:15:05.044006Z","iopub.status.idle":"2024-12-03T01:15:05.356953Z","shell.execute_reply.started":"2024-12-03T01:15:05.043961Z","shell.execute_reply":"2024-12-03T01:15:05.355757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.21\nsirna_code = pd.get_dummies(U2OS_train, columns = ['sirna'])\ntrain_data_x = U2OS_train\ntrain_data_y = sirna_code.drop(['id_code'], axis=1)\nprint(train_data_x.shape)\nprint(train_data_y.shape)\ntrain_data_y.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:15:05.358613Z","iopub.execute_input":"2024-12-03T01:15:05.358951Z","iopub.status.idle":"2024-12-03T01:15:05.388835Z","shell.execute_reply.started":"2024-12-03T01:15:05.358918Z","shell.execute_reply":"2024-12-03T01:15:05.387727Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.22 \nfrom sklearn.model_selection import train_test_split\nx_train_id, x_val_id, Y_train, Y_val = train_test_split(\ntrain_data_x, train_data_y, test_size = .33, random_state=0)\nprint('训练集的特征维度： {0}，标签维度： {1}'.format(x_train_id.shape,Y_train.shape))\nprint('验证集的特征维度： {0}，标签维度： {1}'.format(x_val_id.shape,Y_val.shape))\nx_train_id.reset_index(drop=True, inplace=True)\nY_train.reset_index(drop=True, inplace=True)\nx_val_id.reset_index(drop=True, inplace=True)\nY_val.reset_index(drop=True, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:15:05.390246Z","iopub.execute_input":"2024-12-03T01:15:05.390609Z","iopub.status.idle":"2024-12-03T01:15:05.473893Z","shell.execute_reply.started":"2024-12-03T01:15:05.390559Z","shell.execute_reply":"2024-12-03T01:15:05.472619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.23\nclasses = pd.concat([x_train_id['sirna'],x_val_id['sirna']],axis=0).to_list()\nfor i in range(len(classes)):\n    classes[i]=classes[i].encode(encoding='utf-8')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:15:05.475297Z","iopub.execute_input":"2024-12-03T01:15:05.475695Z","iopub.status.idle":"2024-12-03T01:15:05.483485Z","shell.execute_reply.started":"2024-12-03T01:15:05.475660Z","shell.execute_reply":"2024-12-03T01:15:05.482288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%mkdir ./cell","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:15:05.489809Z","iopub.execute_input":"2024-12-03T01:15:05.490175Z","iopub.status.idle":"2024-12-03T01:15:06.630333Z","shell.execute_reply.started":"2024-12-03T01:15:05.490144Z","shell.execute_reply":"2024-12-03T01:15:06.628739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n#P3.24\ndef make_dataset(x_data,y_data,x_data_id,Y_data):\n    i = 0\n    for code in x_data_id['id_code']:\n        cell_type = code.partition('_')[0] # 细胞系类型\n        plate = int(code.partition('_')[2][0]) # 实验板编号\n        well = code.rpartition('_')[2] # 微孔编号\n        #读取 s1位置的六通道图 像，合成彩位置的六通道图        像，合成彩\n\n        img = load_site_as_rgb('train', cell_type, plate, well, 1, base_path=data_path)\n\n        resized = tf.image.resize(img,(224,224)) #重定义尺寸\n        resized = np.asarray(resized, dtype='uint8')\n        filename = './cell/'+code+'.jpg'\n#         print(filename)\n        plt.imsave(filename,resized) #保存文件\n        x_data[i] = resized #图像特征\n        y_data[i]=Y_data.loc[i].ravel() #标签\n        i += 1\nx_train_shape = (x_train_id.shape[0],224,224,3) #训练集样本空间的维度\nx_val_shape = (x_val_id.shape[0],224,224,3) #验证集样本空间的维度 验证集样本空间的维度\nwith h5py.File('cell.h5','w') as f:\n\n    x_train = f.create_dataset(\"x_train\", x_train_shape ,'i1') #训练集特征 训练集特征    \n    y_train = f.create_dataset(\"y_train\", Y_train.shape ,'i1') #训练集标签 训练集标签\n    x_val = f.create_dataset(\"x_val\", x_val_shape ,'i1') #验证集特征 验证集特征\n    y_val = f.create_dataset(\"y_val\", Y_val.shape ,'i1') #验证集标签 验证集标签\n\n    classes = f.create_dataset('classes',data = classes) #数据集类名称 数据集类名称\n    make_dataset(x_train,y_train,x_train_id,Y_train) #制作训练集 制作训练集\n    print('x_train')\n    make_dataset(x_val,y_val,x_val_id,Y_val) #\n    print('x_val')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T01:15:06.632513Z","iopub.execute_input":"2024-12-03T01:15:06.633000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.25\nfrom keras.applications.vgg16 import VGG16\nfrom keras.preprocessing import image\nfrom keras.applications.vgg16 import preprocess_input\n#下载或读取预训练模型 ，首次执行时需要下载模型参数\nmodel = VGG16(weights='imagenet', include_top=False)\n#单个图像的特征提取\ndef VGG16_extract_features(img):\n    x = np.expand_dims(img, axis=0)\n    features = model.predict(x)\n    return features\n# 用前面的 resized图像测试\nfeatures = VGG16_extract_features(resized)\nfeatures.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.26\ndef load_dataset():\n    with h5py.File('cell.h5','r') as f:\n        x_train = f['x_train'][:] #读取训练集特征\n        y_train = f['y_train'][:] #读取训练集标签\n        x_val = f['x_val'][:] #读取验证集特征\n        y_val = f['y_val'][:] #读取验证集标签\n        classes = f['classes'][:] #类名称\n        return x_train, y_train, x_val, y_val, classes\nX_train, Y_train, X_val, Y_val, classes = load_dataset()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n#P3.27\nm = X_train.shape[0]\nx_train = np.zeros((m,7,7,512))\nfor i in range(m):\n    x_train[i] = VGG16_extract_features(X_train[i])\nprint(x_train.shape)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n#P3.28\nm = X_val.shape[0]\nx_val = np.zeros((m,7,7,512))\nfor i in range(m):\n    x_val[i] = VGG16_extract_features(X_val[i])\nprint(x_val.shape)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.29\nwith h5py.File('cell_features.h5','w') as f:\n    x_train = f.create_dataset(\"x_train\", data=x_train) #训练集特征\n    y_train = f.create_dataset(\"y_train\", data=Y_train) #训练集标签\n    x_val = f.create_dataset(\"x_val\", data=x_val) #验证集特征\n    y_val = f.create_dataset(\"y_val\", data=Y_val) #验证集标签\n    classes = f.create_dataset('classes',data = classes) #数据集类名称","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.30\nimport h5py\ndef load_VGG16_dataset():\n    with h5py.File('cell_features.h5','r') as f:\n        x_train = f['x_train'][:] #读取训练集特征\n        y_train = f['y_train'][:] #读取训练集标签\n        x_val = f['x_val'][:] #读取验证集特征\n        y_val = f['y_val'][:] #读取验证集标签\n        classes = f['classes'][:] #类名称\n        return x_train, y_train, x_val, y_val, classes\nX_train, Y_train, X_val, Y_val, classes = load_VGG16_dataset()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.31\nfrom keras import Sequential\nfrom keras.layers import Dense, Conv2D, Flatten, Dropout\nmodel = Sequential(name = 'VGG16-Cell-Transfer')\nmodel.add(Conv2D(filters = 64, kernel_size = (1, 1), activation = 'relu',\npadding = 'same', input_shape = (7, 7, 512)))\n\nmodel.add(Flatten())\nmodel.add(Dense(2048, activation = 'relu', name = 'fc1'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(2048, activation = 'relu', name = 'fc2'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(1108, activation = 'softmax', name = 'prediction'))\nmodel.summary()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.32\nmodel.compile(optimizer='RMSProp',loss='categorical_crossentropy',metrics=['accuracy'])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.33\nepochs = 10\nbatch_size = 32\nhistory = model.fit(X_train, Y_train, epochs=epochs, batch_size=batch_size,validation_data=(X_val,Y_val))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.34\nimport matplotlib.pyplot as plt\n%matplotlib inline\nx = range(1, len(history.history['accuracy'])+1)\nplt.plot(x, history.history['accuracy'])\nplt.plot(x, history.history['val_accuracy'])\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.xticks(x)\nplt.legend(['Train', 'Val'], loc='upper left')\nplt.show()\nplt.plot(x, history.history['loss'])\nplt.plot(x, history.history['val_loss'])\nplt.title('Model loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.xticks(x)\nplt.legend(['Train', 'Val'], loc='lower left')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from cell_utils import load_dataset\nimport h5py\ndef load_dataset():\n    with h5py.File('cell.h5','r') as f:\n        x_train = f['x_train'][:] #读取训练集特征\n        y_train = f['y_train'][:] #读取训练集标签\n        x_val = f['x_val'][:] #读取验证集特征\n        y_val = f['y_val'][:] #读取验证集标签\n        classes = f['classes'][:] #类名称\n        return x_train, y_train, x_val, y_val, classes\nX_train, Y_train, X_val, Y_val, classes = load_dataset()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.35\n\nX_train, Y_train, X_val, Y_val, classes = load_dataset()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras import Sequential\nfrom keras.layers import Dense, Conv2D, Flatten, Dropout","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.36\n# from keras.applications.resnet50 import ResNet50\nfrom tensorflow.keras.applications.resnet50 import ResNet50\n# 加载 ResNet50模型，不带 ImageNet训练权重，分类数为 1108\nResNet50_model = ResNet50(include_top=True, weights=None, classes=1108)\n#模型编译\nResNet50_model.compile(optimizer='RMSProp',\nloss='categorical_crossentropy',\nmetrics=['accuracy'])\n#开始训练\nepochs = 5\nbatch_size = 32\nhistory = ResNet50_model.fit(X_train, Y_train, epochs=epochs, batch_size=batch_size,validation_data=(X_val,Y_val))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.37\nmodel.save('VGG16_Transfer_model.h5') # 保存 VGG16迁移学习的结构和参数 迁移学习的结构和参数\nResNet50_model.save('ResNet50_model.h5') #保存 ResNet152模型结构和参数","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#P3.38\nfrom keras.models import load_model\nfrom keras.preprocessing import image\nfrom tensorflow.keras.applications.resnet50 import preprocess_input\n# from keras.applications.resnet50 import preprocess_input\nimport pandas as pd\nimport numpy as np\nResNet50_model = load_model('ResNet50_model.h5') #加载模型 加载模型\nimg_path = './cell/U2OS-01_2_B12.jpg' #随机 指定一幅测试图像 指定一幅测试图像\nimg = image.load_img(img_path, target_size=(224, 224))\nplt.imshow(img) #显示原图\ntrain = pd.read_csv(f\"{data_path}/train.csv\")\nprint('真实的标签为： {0}'.format(train[train.id_code == 'U2OS-01_2_B12']['sirna'].values))\nx = image.img_to_array(img)\nx = np.expand_dims(x, axis=0)\nx = preprocess_input(x)\npred = ResNet50_model.predict(x) #模型预测\n# 将预测得到的概 率值进行 One-Hot编码\npred = pred.ravel()\nindex = np.argmax(pred)\npred = np.zeros(1108)\npred[index] = 1\nprint('模型预测值为： {0}'.format(pred))\nfor i, y in enumerate(Y_train):\n    if (y == pred).all():\n        print('模型预测的标签为： 模型预测的标签为： {0}'.format(classes[i].decode('utf-8')))\n        break","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}