{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"}],"dockerImageVersionId":31154,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:47:27.113730Z","iopub.execute_input":"2025-10-28T12:47:27.114412Z","iopub.status.idle":"2025-10-28T12:47:37.061812Z","shell.execute_reply.started":"2025-10-28T12:47:27.114384Z","shell.execute_reply":"2025-10-28T12:47:37.061016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install opencv-python-headless numpy scipy matplotlib pandas tensorflow\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:47:42.860356Z","iopub.execute_input":"2025-10-28T12:47:42.861027Z","iopub.status.idle":"2025-10-28T12:47:51.521458Z","shell.execute_reply.started":"2025-10-28T12:47:42.861001Z","shell.execute_reply":"2025-10-28T12:47:51.520608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import layers, models\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:49:09.179724Z","iopub.execute_input":"2025-10-28T12:49:09.180432Z","iopub.status.idle":"2025-10-28T12:49:15.287946Z","shell.execute_reply.started":"2025-10-28T12:49:09.180408Z","shell.execute_reply":"2025-10-28T12:49:15.287349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_dir = \"/kaggle/input/physionet-ecg-image-digitization\"\ntrain_dir = os.path.join(base_dir, \"train\")\ntest_dir = os.path.join(base_dir, \"test\")\n\ntrain_csv = os.path.join(base_dir, \"train.csv\")\ntest_csv = os.path.join(base_dir, \"test.csv\")\n\nprint(\"✅ Paths set:\")\nprint(train_dir)\nprint(test_dir)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:49:19.160776Z","iopub.execute_input":"2025-10-28T12:49:19.161307Z","iopub.status.idle":"2025-10-28T12:49:19.166361Z","shell.execute_reply.started":"2025-10-28T12:49:19.161285Z","shell.execute_reply":"2025-10-28T12:49:19.165603Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv(train_csv)\nprint(\"Columns:\", train_df.columns.tolist())\nprint(train_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:49:22.554888Z","iopub.execute_input":"2025-10-28T12:49:22.555351Z","iopub.status.idle":"2025-10-28T12:49:22.586004Z","shell.execute_reply.started":"2025-10-28T12:49:22.555329Z","shell.execute_reply":"2025-10-28T12:49:22.585390Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_image_paths(ecg_id):\n    folder = os.path.join(train_dir, str(ecg_id))\n    if os.path.exists(folder):\n        images = [os.path.join(folder, f) for f in os.listdir(folder) if f.endswith(\".png\")]\n        return images\n    return []\n\ntrain_images = []\nfor _, row in train_df.iterrows():\n    for img_path in get_image_paths(row[\"id\"]):\n        train_images.append({\n            \"id\": row[\"id\"],\n            \"image_path\": img_path,\n            \"label\": \"unknown\"  # dummy label\n        })\n\ntrain_data = pd.DataFrame(train_images)\nprint(\"✅ Total train images found:\", len(train_data))\nprint(train_data.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:49:29.895012Z","iopub.execute_input":"2025-10-28T12:49:29.895852Z","iopub.status.idle":"2025-10-28T12:49:38.536267Z","shell.execute_reply.started":"2025-10-28T12:49:29.895817Z","shell.execute_reply":"2025-10-28T12:49:38.535582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing = [p for p in train_data[\"image_path\"] if not os.path.exists(p)]\nprint(f\"Missing {len(missing)} / {len(train_data)} images.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:49:41.832603Z","iopub.execute_input":"2025-10-28T12:49:41.833318Z","iopub.status.idle":"2025-10-28T12:49:46.899768Z","shell.execute_reply.started":"2025-10-28T12:49:41.833294Z","shell.execute_reply":"2025-10-28T12:49:46.898929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_df_split, val_df_split = train_test_split(train_data, test_size=0.2, random_state=42)\nprint(\"Train size:\", len(train_df_split))\nprint(\"Val size:\", len(val_df_split))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:49:51.094160Z","iopub.execute_input":"2025-10-28T12:49:51.094704Z","iopub.status.idle":"2025-10-28T12:49:51.237264Z","shell.execute_reply.started":"2025-10-28T12:49:51.094683Z","shell.execute_reply":"2025-10-28T12:49:51.236539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_height, img_width = 224, 224\nbatch_size = 32\n\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=10,\n    zoom_range=0.1,\n    horizontal_flip=True\n)\n\nval_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    train_df_split,\n    x_col=\"image_path\",\n    y_col=\"label\",\n    target_size=(img_height, img_width),\n    class_mode=\"categorical\",\n    batch_size=batch_size\n)\n\nval_generator = val_datagen.flow_from_dataframe(\n    val_df_split,\n    x_col=\"image_path\",\n    y_col=\"label\",\n    target_size=(img_height, img_width),\n    class_mode=\"categorical\",\n    batch_size=batch_size\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:49:54.234735Z","iopub.execute_input":"2025-10-28T12:49:54.235306Z","iopub.status.idle":"2025-10-28T12:50:15.723506Z","shell.execute_reply.started":"2025-10-28T12:49:54.235268Z","shell.execute_reply":"2025-10-28T12:50:15.722852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = models.Sequential([\n    layers.Input(shape=(img_height, img_width, 3)),\n    layers.Conv2D(32, (3,3), activation='relu'),\n    layers.MaxPooling2D(2,2),\n    layers.Conv2D(64, (3,3), activation='relu'),\n    layers.MaxPooling2D(2,2),\n    layers.Flatten(),\n    layers.Dense(64, activation='relu'),\n    layers.Dense(1, activation='sigmoid') \n])\n\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:50:19.796999Z","iopub.execute_input":"2025-10-28T12:50:19.797671Z","iopub.status.idle":"2025-10-28T12:50:22.069856Z","shell.execute_reply.started":"2025-10-28T12:50:19.797649Z","shell.execute_reply":"2025-10-28T12:50:22.069231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Example model training\nhistory = model.fit(\n    train_generator,\n    validation_data=val_generator,\n    epochs=1,\n    verbose=1\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:50:29.380699Z","iopub.execute_input":"2025-10-28T12:50:29.380998Z","iopub.status.idle":"2025-10-28T13:40:09.797556Z","shell.execute_reply.started":"2025-10-28T12:50:29.380968Z","shell.execute_reply":"2025-10-28T13:40:09.796951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n\nmodel.save(\"ecg_cnn_classifier.h5\")\nprint(\"✅ Model saved as ecg_cnn_classifier.h5\")\n\n\nhistory_df = pd.DataFrame(history.history)\n\n\nhistory_df[\"epoch\"] = range(1, len(history_df) + 1)\n\n\nhistory_df.to_csv(\" submission.csv\", index=False)\nprint(\"✅ Training history saved as ecg_training_history.csv\")\n\n\nprint(\"\\n📊 Final few training results:\")\nprint(history_df.tail())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T13:44:43.927930Z","iopub.execute_input":"2025-10-28T13:44:43.928520Z","iopub.status.idle":"2025-10-28T13:44:44.473952Z","shell.execute_reply.started":"2025-10-28T13:44:43.928496Z","shell.execute_reply":"2025-10-28T13:44:44.473297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport os\n\ntest_dir = \"/kaggle/input/physionet-ecg-image-digitization/test\"\n\n\ntest_csv = \"/kaggle/input/physionet-ecg-image-digitization/test.csv\"\ntest_df = pd.read_csv(test_csv)\n\n\ntest_df[\"image_path\"] = test_df[\"id\"].astype(str) + \".png\"\ntest_df[\"image_path\"] = test_df[\"image_path\"].apply(lambda x: os.path.join(test_dir, x))\n\ntest_datagen = ImageDataGenerator(rescale=1./255)\n\n\ntest_generator = test_datagen.flow_from_dataframe(\n    dataframe=test_df,\n    x_col=\"image_path\",\n    y_col=None,\n    target_size=(224, 224),\n    class_mode=None,\n    batch_size=32,\n    shuffle=False\n)\n\nprint(\"✅ Test data generator ready.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T13:46:00.520719Z","iopub.execute_input":"2025-10-28T13:46:00.520997Z","iopub.status.idle":"2025-10-28T13:46:00.541898Z","shell.execute_reply.started":"2025-10-28T13:46:00.520977Z","shell.execute_reply":"2025-10-28T13:46:00.540968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n\npreds = model.predict(test_generator)\npred_classes = np.argmax(preds, axis=1)\n\nsubmission = pd.DataFrame({\n    \"id\": test_df[\"id\"],\n    \"label\": pred_classes\n})\n\n\nsubmission.to_csv(\"/kaggle/working/submission.csv\", index=False)\n\nprint(\"✅ submission.csv file created successfully!\")\nsubmission.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T13:55:47.438407Z","iopub.execute_input":"2025-10-28T13:55:47.439058Z","iopub.status.idle":"2025-10-28T13:55:50.708235Z","shell.execute_reply.started":"2025-10-28T13:55:47.439033Z","shell.execute_reply":"2025-10-28T13:55:50.707638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# ✅ Delete any wrongly named files (e.g., with leading spaces)\n!rm -f \"/kaggle/working/ submission.csv\"\n\n# ✅ Create correct submission.csv file (no space)\nsubmission_path = \"/kaggle/working/submission.csv\"\nsubmission.to_csv(submission_path, index=False)\n\n# ✅ Confirm final file\nprint(\"✅ Final submission file created:\", submission_path)\n!ls -lh /kaggle/working/\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T13:57:57.112240Z","iopub.execute_input":"2025-10-28T13:57:57.112548Z","iopub.status.idle":"2025-10-28T13:57:57.416224Z","shell.execute_reply.started":"2025-10-28T13:57:57.112524Z","shell.execute_reply":"2025-10-28T13:57:57.415355Z"}},"outputs":[],"execution_count":null}]}