{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":10418,"databundleVersionId":862236}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Human Protein Data\n- 사람 세포 속 단백질 위치를 예측하는 모델을 만드는 것이 목적.\n- 높은 정확도, 빠른 속도, 최소한의 하드웨어 성능으로.\n## Image Data Paper\n- 이미지 데이터의 label과 이미지 자체 정보에 문제가 있다면 인간의 힘이 아직 필요함\n- 영상 데이터 추출에 사람의 노동력 필요\n- 저작권, 수집 등의 이슈가 많음\n## Human Protein Atlas Image Classification 전략\n- 개요\n  * 세포 이미지(Human Protein Atlas)를 이용하여 단백질의 위치를 분류한다\n  * 하나의 이미지에 여러 class가 존재함\n\n- 문제점\n  1. 희귀(rare) class가 많아서 class imbalance 문제가 있음\n     * 일부 class 데이터가 매우 적음\n     * rare class 예측 어려움\n  \n  2. High-Resolution Image\n     * 이미지 품질은 좋은데 계산량에 영향을 줌\n     * 정확도와 연산 효율의 균형이 필요\n\n- methodology\n\n    1. Rotate 90도, Flip, Random Crop\n       * 얻는 것: 과적합 감소, 다양한 시야에 대한 학습을 가능하게 함\n    2. DenseNet121\n       * 단순 구조에서 더 좋은 성능\n       * 이미지 자체는 복잡하지만 안정적인 구조가 현재 상황에 맞을 수 있음\n    3. Focal Loss, Lovasz Loss\n       * Rare class에 대한 강화(imbalance 해결)\n       * F1 score(recall, precision 균형)를 위해서\n\n- experiments\n  * 여러 방법의 random crop을 수행\n  * 서로 다른 seed 사용\n  * 최종적으로: max probability 선택\n  * 얻는 것: 안정성, 일반화 성능 향상","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nPATH = './'\nTRAIN = '/kaggle/input/competitions/human-protein-atlas-image-classification/train/'\nTEST = '/kaggle/input/competitions/human-protein-atlas-image-classification/test/'\nLABELS = '/kaggle/input/competitions/human-protein-atlas-image-classification/train.csv'\nSAMPLE = '/kaggle/input/competitions/human-protein-atlas-image-classification/sample_submission.csv'\n\ndf = pd.read_csv(LABELS)\n\nprint(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-15T10:57:18.34068Z","iopub.execute_input":"2026-06-15T10:57:18.341006Z","iopub.status.idle":"2026-06-15T10:57:18.381089Z","shell.execute_reply.started":"2026-06-15T10:57:18.340978Z","shell.execute_reply":"2026-06-15T10:57:18.379924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_counts = {}\n\nfor targets in df['Target']:\n    for label in targets.split():\n        label = int(label)\n\n        if label not in label_counts:\n            label_counts[label] = 0\n        \n        label_counts[label] += 1\n\nlabel_df = pd.DataFrame({\n    'Label': list(label_counts.keys()),\n    'Count': list(label_counts.values())\n})\n\nlabel_df = label_df.sort_values('Count')\n\nplt.figure(figsize=(12,8))\nsns.barplot(\n    data=label_df,\n    x='Label',\n    y='Count'\n)\n\nplt.title(\"Class Distribution\")\nplt.show\n\n# label imbalance 확인가능 -> 8,9,10,15 라벨이 매우 적음","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-15T10:57:20.882195Z","iopub.execute_input":"2026-06-15T10:57:20.882695Z","iopub.status.idle":"2026-06-15T10:57:21.292058Z","shell.execute_reply.started":"2026-06-15T10:57:20.882638Z","shell.execute_reply":"2026-06-15T10:57:21.2911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train data 에서 label 확인\n# data의 image를 확인하기 위해 5개만 추출\n# Cmap에서 사용하는 color 이름이 다르기 때문에 dict로 지정하여 관리\n\ntrain_df = pd.read_csv(LABELS)\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-15T10:57:27.451318Z","iopub.execute_input":"2026-06-15T10:57:27.451686Z","iopub.status.idle":"2026-06-15T10:57:27.490335Z","shell.execute_reply.started":"2026-06-15T10:57:27.451649Z","shell.execute_reply":"2026-06-15T10:57:27.48943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom PIL import Image\n\nsample_ids = train_df['Id'].values[:5]\ncmap_dict = {\n    'red':'Reds',\n    'green':'Greens',\n    'blue':'Blues',\n    'yellow':'YlOrBr'\n}\n\n# 색깔별 의미\n# green: 단백질의 위치, blue: 핵 위치, red: 세포 골격, yellow: 소포체 위치\n\ncolors = ['red','green','blue','yellow']\n\nfor sample_id in sample_ids:\n    fig, ax = plt.subplots(1,4,figsize=(16,4))\n    for i, color in enumerate(colors):\n        img = Image.open(\n            TRAIN + sample_id + f'_{color}.png'\n        )\n\n        img = np.array(img)\n\n        ax[i].imshow(\n            img,\n            cmap=cmap_dict[color]\n        )\n\n        ax[i].set_title(color)\n\n        ax[i].axis('off')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-15T10:57:29.918915Z","iopub.execute_input":"2026-06-15T10:57:29.919351Z","iopub.status.idle":"2026-06-15T10:57:32.547357Z","shell.execute_reply.started":"2026-06-15T10:57:29.91931Z","shell.execute_reply":"2026-06-15T10:57:32.54618Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# RGB 채널로 보여주기 위해 Yellows는 제외한 이미지 Merge\n\nsample_id = train_df.iloc[2]['Id']\n\nred = np.array(\n    Image.open(TRAIN + sample_id + '_red.png')\n)\n\ngreen = np.array(\n    Image.open(TRAIN + sample_id + '_green.png')\n)\n\nblue = np.array(\n    Image.open(TRAIN + sample_id + '_blue.png')\n)\n\nrgb = np.stack(\n    [red, green, blue],\n    axis=-1\n)\n\nplt.figure(figsize=(8,8))\nplt.imshow(rgb)\nplt.title(\"Merged RGB View\")\nplt.axis('off')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-15T10:57:36.282719Z","iopub.execute_input":"2026-06-15T10:57:36.283079Z","iopub.status.idle":"2026-06-15T10:57:36.51204Z","shell.execute_reply.started":"2026-06-15T10:57:36.283045Z","shell.execute_reply":"2026-06-15T10:57:36.511054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 28개 class에 대한 벡터를 생성\n# target이 label 1,5,6일 때 [0,1,0,0,0,1,1,0,...] 이런식\n\ndef make_label(target):\n    y = np.zeros(28)\n\n    for t in target.split():\n        y[int(t)] = 1\n\n    return y\n\nIMG_SIZE=128\n\ndef load_image(img_id):\n    # 간단한 cnn model을 사용해서 이미지 사이즈 줄이기\n    channels = []\n\n    for color in ['red','green','blue','yellow']:\n        img = Image.open(\n            TRAIN + img_id + f'_{color}.png'\n        )\n\n        img = img.resize((IMG_SIZE, IMG_SIZE))\n        img = np.array(img)\n\n        channels.append(img) # color 채널 합침\n\n    img = np.stack(\n        channels,\n        axis=-1\n    )\n\n    return img/255.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-15T11:46:01.968178Z","iopub.execute_input":"2026-06-15T11:46:01.968616Z","iopub.status.idle":"2026-06-15T11:46:01.976688Z","shell.execute_reply.started":"2026-06-15T11:46:01.968535Z","shell.execute_reply":"2026-06-15T11:46:01.975533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = []\ny = []\n\n# 모든 데이터 셋에서 1000개만 사용\nN = 1000\n\nfor i in range(N):\n    img_id = train_df.iloc[i]['Id']\n\n    image = load_image(img_id)\n\n    label = make_label(train_df.iloc[i]['Target'])\n\n    X.append(image)\n    y.append(label)\n\nX = np.array(X)\ny = np.array(y)\n\nprint(X.shape)\nprint(y.shape)\n# 첫번째 출력: (1000, 128, 128, 4)\n## 1000장의 이미지, 128X128 크기의 이미지, 4개의 채널\n# 두번째 출력: (1000, 28)\n## 1000장의 이미지, 28개의 label\n\nfrom sklearn.model_selection import train_test_split\n\n# train:test=8:2\nX_train, X_valid, y_train, y_valid = train_test_split(\n    X,y,\n    test_size=0.2,\n    random_state=42\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-15T11:46:26.992841Z","iopub.execute_input":"2026-06-15T11:46:26.993256Z","iopub.status.idle":"2026-06-15T11:46:48.33322Z","shell.execute_reply.started":"2026-06-15T11:46:26.993214Z","shell.execute_reply":"2026-06-15T11:46:48.33213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\nmodel = tf.keras.Sequential([\n    \n    tf.keras.layers.Input(shape=(128, 128, 4)),\n\n    # relu : 곡선이 있는 이미지 분석에 필수\n    # conv2D : 특징 추출\n    tf.keras.layers.Conv2D(\n        32,\n        3,\n        activation='relu',\n        # input_shape=(128,128,4)\n    ),\n\n    # MaxPooling : 크기 축소\n    tf.keras.layers.MaxPooling2D(),\n \n    tf.keras.layers.Conv2D(\n        64,\n        3,\n        activation='relu'\n    ),\n \n    tf.keras.layers.MaxPooling2D(),\n \n    tf.keras.layers.Conv2D(\n        128,\n        3,\n        activation='relu'\n    ),\n \n    tf.keras.layers.MaxPooling2D(),\n\n    # Flatten : 벡터 변환\n    tf.keras.layers.Flatten(),\n    # Dense : 최종 판단을 위하여 28개의 class에서 복수의 정답을 추출\n    tf.keras.layers.Dense(\n        256,\n        activation='relu'\n    ),\n\n    # sigmoid : 복수정답 / softmax : 정답이 하나\n    tf.keras.layers.Dense(\n        28,\n        activation='sigmoid'\n    )\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-15T11:42:54.871728Z","iopub.execute_input":"2026-06-15T11:42:54.872091Z","iopub.status.idle":"2026-06-15T11:42:54.949865Z","shell.execute_reply.started":"2026-06-15T11:42:54.872056Z","shell.execute_reply":"2026-06-15T11:42:54.94879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['accuracy']\n)\n\nhistory = model.fit(\n    X_train, y_train,\n    validation_data=(\n        X_valid, y_valid\n    ),\n    epochs=5,\n    batch_size=16\n)\n\ndef load_test_image(img_id):\n    channels = []\n\nfor color in ['red','green','blue','yellow']:\n    img = Image.open(\n        TEST + img_id + f'_{color}.png'\n    )\n\n    img = img.resize((IMG_SIZE,IMG_SIZE))\n\n    img = np.array(img)\n\n    channels.append(img)\n\nimg = np.stack(\n    channels,\n    axis=-1\n)\n\nreturn img/255.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-15T11:43:00.980808Z","iopub.execute_input":"2026-06-15T11:43:00.981203Z","iopub.status.idle":"2026-06-15T11:44:08.235554Z","shell.execute_reply.started":"2026-06-15T11:43:00.981151Z","shell.execute_reply":"2026-06-15T11:44:08.234392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred = model.predict(X_valid)\n\nidx = 10\n\ntrue_labels = np.where(\n    y_valid[idx] == 1\n)[0]\n\npred_labels = np.where(\n    pred[idx] > 0.5\n)[0]\n\nplt.figure(figsize=(6,6))\nplt.imshow(\n    X_valid[idx][:,:,:3]\n)\n\nplt.title(\n    f\"True: {list(true_labels)}\\n\"\n    f\"Pred: {list(pred_labels)}\"\n)\n\nplt.axis('off')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-15T11:45:23.943511Z","iopub.execute_input":"2026-06-15T11:45:23.943998Z","iopub.status.idle":"2026-06-15T11:45:25.174509Z","shell.execute_reply.started":"2026-06-15T11:45:23.943961Z","shell.execute_reply":"2026-06-15T11:45:25.173556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}