{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":10418,"databundleVersionId":862236,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Conv2D, GlobalAveragePooling2D, Dense\nfrom tensorflow.keras.callbacks import ModelCheckpoint","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T14:12:33.568730Z","iopub.execute_input":"2025-06-22T14:12:33.569040Z","iopub.status.idle":"2025-06-22T14:12:43.320906Z","shell.execute_reply.started":"2025-06-22T14:12:33.569017Z","shell.execute_reply":"2025-06-22T14:12:43.320291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = 512\nNUM_CLASSES = 28\nBATCH_SIZE = 16\nEPOCHS = 10\n\nTRAIN_DIR = '/kaggle/input/human-protein-atlas-image-classification/train/'  # Update if path is different\nCSV_PATH = '/kaggle/input/human-protein-atlas-image-classification/train.csv'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T14:12:46.528549Z","iopub.execute_input":"2025-06-22T14:12:46.529742Z","iopub.status.idle":"2025-06-22T14:12:46.533814Z","shell.execute_reply.started":"2025-06-22T14:12:46.529703Z","shell.execute_reply":"2025-06-22T14:12:46.532898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def encode_labels(label_str):\n    label = np.zeros(NUM_CLASSES, dtype=np.float32)\n    for l in label_str.split():\n        label[int(l)] = 1.0\n    return label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T14:12:49.946860Z","iopub.execute_input":"2025-06-22T14:12:49.947202Z","iopub.status.idle":"2025-06-22T14:12:49.952148Z","shell.execute_reply.started":"2025-06-22T14:12:49.947178Z","shell.execute_reply":"2025-06-22T14:12:49.951248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_image_and_label(image_id, label_str, image_dir=TRAIN_DIR, size=(IMG_SIZE, IMG_SIZE)):\n    channels = []\n    for color in ['red', 'green', 'blue', 'yellow']:\n        path = os.path.join(image_dir, f\"{image_id}_{color}.png\")\n        img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n        img = cv2.resize(img, size)\n        channels.append(img)\n    image = np.stack(channels, axis=-1) / 255.0\n    label = encode_labels(label_str)\n    return image.astype(np.float32), label\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T14:12:53.292791Z","iopub.execute_input":"2025-06-22T14:12:53.293630Z","iopub.status.idle":"2025-06-22T14:12:53.299793Z","shell.execute_reply.started":"2025-06-22T14:12:53.293602Z","shell.execute_reply":"2025-06-22T14:12:53.299007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_tf_dataset(df, batch_size=BATCH_SIZE, shuffle=True):\n    def generator():\n        for _, row in df.iterrows():\n            yield load_image_and_label(row['Id'], row['Target'])\n\n    ds = tf.data.Dataset.from_generator(\n        generator,\n        output_types=(tf.float32, tf.float32),\n        output_shapes=((IMG_SIZE, IMG_SIZE, 4), (NUM_CLASSES,))\n    )\n    if shuffle:\n        ds = ds.shuffle(1024)\n    return ds.batch(batch_size).prefetch(tf.data.AUTOTUNE)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T14:12:56.021102Z","iopub.execute_input":"2025-06-22T14:12:56.021692Z","iopub.status.idle":"2025-06-22T14:12:56.026249Z","shell.execute_reply.started":"2025-06-22T14:12:56.021669Z","shell.execute_reply":"2025-06-22T14:12:56.025632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.layers import Lambda\n\ndef build_model():\n    input_tensor = Input(shape=(IMG_SIZE, IMG_SIZE, 4))\n    x = Conv2D(3, (1, 1), padding='same')(input_tensor)  # Convert 4 -> 3 channels\n    base = ResNet50(include_top=False, weights='imagenet', input_shape=(IMG_SIZE, IMG_SIZE, 3))\n    base_out = base(x)\n    x = GlobalAveragePooling2D()(base_out)\n    output = Dense(NUM_CLASSES, activation='sigmoid')(x)\n    model = Model(inputs=input_tensor, outputs=output)\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T14:12:58.846842Z","iopub.execute_input":"2025-06-22T14:12:58.847132Z","iopub.status.idle":"2025-06-22T14:12:58.852292Z","shell.execute_reply.started":"2025-06-22T14:12:58.847110Z","shell.execute_reply":"2025-06-22T14:12:58.851502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if __name__ == '__main__':\n    df = pd.read_csv(CSV_PATH)\n\n    # Sample only 20% for quick training\n    df_small = df.sample(frac=0.1, random_state=42).reset_index(drop=True)\n    train_df, val_df = train_test_split(df_small, test_size=0.1, random_state=42)\n\n    train_ds = create_tf_dataset(train_df)\n    val_ds = create_tf_dataset(val_df, shuffle=False)\n\n    model = build_model()\n    model.compile(optimizer='adam', loss='binary_crossentropy', metrics=[tf.keras.metrics.AUC(name='AUC')])\n\n    checkpoint = ModelCheckpoint('best_model.h5', monitor='val_AUC', save_best_only=True, mode='max', verbose=1)\n    model.fit(train_ds, validation_data=val_ds, epochs=EPOCHS, callbacks=[checkpoint])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T14:13:01.358512Z","iopub.execute_input":"2025-06-22T14:13:01.359029Z","iopub.status.idle":"2025-06-22T14:40:23.974158Z","shell.execute_reply.started":"2025-06-22T14:13:01.359007Z","shell.execute_reply":"2025-06-22T14:40:23.973477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the best model\nmodel.load_weights('best_model.h5')\n\n# Evaluate on validation dataset\nresults = model.evaluate(val_ds)\nprint(f\"\\n📊 Validation Loss: {results[0]:.4f}\")\nprint(f\"✅ Validation AUC: {results[1]:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T14:51:32.557463Z","iopub.execute_input":"2025-06-22T14:51:32.557789Z","iopub.status.idle":"2025-06-22T14:51:39.876759Z","shell.execute_reply.started":"2025-06-22T14:51:32.557767Z","shell.execute_reply":"2025-06-22T14:51:39.876090Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom tensorflow.keras.models import load_model\nfrom tqdm import tqdm\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T15:00:55.666013Z","iopub.execute_input":"2025-06-22T15:00:55.666668Z","iopub.status.idle":"2025-06-22T15:00:55.670512Z","shell.execute_reply.started":"2025-06-22T15:00:55.666643Z","shell.execute_reply":"2025-06-22T15:00:55.669778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TEST_DIR = '/kaggle/input/human-protein-atlas-image-classification/test/'\nMODEL_PATH = '/kaggle/working/best_model.h5'\nIMG_SIZE = 512\nNUM_CLASSES = 28","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T15:02:03.015232Z","iopub.execute_input":"2025-06-22T15:02:03.015773Z","iopub.status.idle":"2025-06-22T15:02:03.019259Z","shell.execute_reply.started":"2025-06-22T15:02:03.015747Z","shell.execute_reply":"2025-06-22T15:02:03.018663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = load_model(MODEL_PATH)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T15:02:09.769024Z","iopub.execute_input":"2025-06-22T15:02:09.769593Z","iopub.status.idle":"2025-06-22T15:02:11.413902Z","shell.execute_reply.started":"2025-06-22T15:02:09.769571Z","shell.execute_reply":"2025-06-22T15:02:11.413102Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_test_image(image_id, image_dir=TEST_DIR, size=(IMG_SIZE, IMG_SIZE)):\n    channels = []\n    for color in ['red', 'green', 'blue', 'yellow']:\n        path = os.path.join(image_dir, f\"{image_id}_{color}.png\")\n        img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n        if img is None:\n            raise FileNotFoundError(f\"Missing image: {path}\")\n        img = cv2.resize(img, size)\n        channels.append(img)\n    image = np.stack(channels, axis=-1) / 255.0\n    return image.astype(np.float32)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T15:02:25.179825Z","iopub.execute_input":"2025-06-22T15:02:25.180582Z","iopub.status.idle":"2025-06-22T15:02:25.185466Z","shell.execute_reply.started":"2025-06-22T15:02:25.180553Z","shell.execute_reply":"2025-06-22T15:02:25.184659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_files = os.listdir(TEST_DIR)\nimage_ids = sorted(set(f.split('_')[0] for f in test_files if f.endswith('.png')))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T15:02:36.443780Z","iopub.execute_input":"2025-06-22T15:02:36.444086Z","iopub.status.idle":"2025-06-22T15:02:37.195782Z","shell.execute_reply.started":"2025-06-22T15:02:36.444063Z","shell.execute_reply":"2025-06-22T15:02:37.194943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = []\nfor image_id in tqdm(image_ids):\n    img = load_test_image(image_id)\n    pred = model.predict(np.expand_dims(img, axis=0))[0]  # shape (28,)\n    pred_labels = [str(i) for i, p in enumerate(pred) if p > 0.5]\n    results.append({'Id': image_id, 'Predicted': ' '.join(pred_labels)})\n\n# --- Save as submission.csv ---\nsubmission_df = pd.DataFrame(results)\nsubmission_df.to_csv('submission.csv', index=False)\n\nprint(\"\\n✅ Done! Predictions saved in submission.csv\")\nprint(submission_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T15:02:47.316118Z","iopub.execute_input":"2025-06-22T15:02:47.316709Z","iopub.status.idle":"2025-06-22T15:34:04.840079Z","shell.execute_reply.started":"2025-06-22T15:02:47.316687Z","shell.execute_reply":"2025-06-22T15:34:04.839247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}