{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":35273,"databundleVersionId":3351394,"sourceType":"competition"}],"dockerImageVersionId":30171,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <center> ECHO2022 - EDA & Baseline model </center> \n## <center> Predicting cardiac ejection fraction based on echocardiographic image loops</center> \nThis notebook is meant to give participants in ECHO2022 a brief introduction to the dataset, a brief exploratory data analysis and a baseline model as benchmark for their own work and ideas. For question regarding the dataset and this competition, please visit the [competition page](https://www.kaggle.com/c/echo2022).\n\n<div style=\"width:100%;text-align: center;\"> <img align=middle src=\"https://raw.githubusercontent.com/Bsingstad/DL-images/main/EchoAnimation.gif\" alt=\"echo animation\" style=\"height:500px;margin-top:3rem;\"> </div>","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport os\nimport pandas as pd\nimport cv2\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport tensorflow as tf","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.388948,"end_time":"2022-03-15T10:07:40.33147","exception":false,"start_time":"2022-03-15T10:07:39.942522","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2025-06-13T17:45:33.701342Z","iopub.execute_input":"2025-06-13T17:45:33.701572Z","iopub.status.idle":"2025-06-13T17:45:45.854859Z","shell.execute_reply.started":"2025-06-13T17:45:33.701496Z","shell.execute_reply":"2025-06-13T17:45:45.853968Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = pd.read_csv(\"../input/echo2022/train_data.csv\")","metadata":{"papermill":{"duration":0.043705,"end_time":"2022-03-15T10:07:40.406374","exception":false,"start_time":"2022-03-15T10:07:40.362669","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2025-06-13T17:46:12.991997Z","iopub.execute_input":"2025-06-13T17:46:12.992567Z","iopub.status.idle":"2025-06-13T17:46:13.012548Z","shell.execute_reply.started":"2025-06-13T17:46:12.992512Z","shell.execute_reply":"2025-06-13T17:46:13.011973Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"../input/echo2022/sample_submission.csv\")","metadata":{"papermill":{"duration":0.040178,"end_time":"2022-03-15T10:07:40.476184","exception":false,"start_time":"2022-03-15T10:07:40.436006","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2025-06-13T17:46:14.166443Z","iopub.execute_input":"2025-06-13T17:46:14.166740Z","iopub.status.idle":"2025-06-13T17:46:14.181152Z","shell.execute_reply.started":"2025-06-13T17:46:14.166711Z","shell.execute_reply":"2025-06-13T17:46:14.180444Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_2CH_dir = \"../input/echo2022/train_data/train_data/2CH/\"\ntrain_4CH_dir = \"../input/echo2022/train_data/train_data/4CH/\"\ntest_2CH_dir = \"../input/echo2022/test_data/test_data/2CH/\"\ntest_4CH_dir = \"../input/echo2022/test_data/test_data/4CH/\"","metadata":{"papermill":{"duration":0.034765,"end_time":"2022-03-15T10:07:40.541966","exception":false,"start_time":"2022-03-15T10:07:40.507201","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2025-06-13T17:46:17.914418Z","iopub.execute_input":"2025-06-13T17:46:17.914969Z","iopub.status.idle":"2025-06-13T17:46:17.919480Z","shell.execute_reply.started":"2025-06-13T17:46:17.914930Z","shell.execute_reply":"2025-06-13T17:46:17.918688Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 📊 Data exploration","metadata":{}},{"cell_type":"code","source":"file_path = os.path.join(train_4CH_dir, \"patient001_4CH_sequence.npy\")\n\n# Load the .npy file\ndata = np.load(file_path)\n\n# Check the shape of the data to decide how to plot\nprint(f\"Shape of the data: {data.shape}\")\n\n# Assuming the data is 2D or 3D (if 3D, the third dimension might represent channels or time points)\n# Plot a single slice or the first 2D array if it's 3D\nif data.ndim == 2:\n    plt.imshow(data, cmap='gray')\n    plt.title('2D Sequence')\nelif data.ndim == 3:\n    # If it's 3D, we can plot the first channel or time step\n    plt.imshow(data[0, :, :], cmap='gray')  # Adjust indexing if needed\n    plt.title('First Channel of 3D Sequence')\nelse:\n    print(\"Data has an unsupported number of dimensions for plotting.\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T17:46:59.454905Z","iopub.execute_input":"2025-06-13T17:46:59.455631Z","iopub.status.idle":"2025-06-13T17:46:59.634967Z","shell.execute_reply.started":"2025-06-13T17:46:59.455595Z","shell.execute_reply":"2025-06-13T17:46:59.634243Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-13T17:46:44.466993Z","iopub.execute_input":"2025-06-13T17:46:44.467777Z","iopub.status.idle":"2025-06-13T17:46:44.471278Z","shell.execute_reply.started":"2025-06-13T17:46:44.467739Z","shell.execute_reply":"2025-06-13T17:46:44.470510Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import metadata about the echo images\nsequence_number = []\ntrain_img_w = []\ntrain_img_h = []\nfor i in tqdm(os.listdir(train_2CH_dir)):\n    if i.endswith(\".npy\"):\n        number, width, height = np.load(train_2CH_dir + i).shape\n        sequence_number.append(number)\n        train_img_w.append(width)\n        train_img_h.append(height)","metadata":{"papermill":{"duration":124.341585,"end_time":"2022-03-15T10:09:44.975801","exception":false,"start_time":"2022-03-15T10:07:40.634216","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-03-15T20:08:08.424802Z","iopub.execute_input":"2022-03-15T20:08:08.425053Z","iopub.status.idle":"2022-03-15T20:10:18.762891Z","shell.execute_reply.started":"2022-03-15T20:08:08.425023Z","shell.execute_reply":"2022-03-15T20:10:18.761855Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.hist(np.asarray(sequence_number), bins=10)\nplt.title(\"Distribution of the number of image across all patients in the training set\")\nplt.xlabel(\"Number of images in the echo loop\")\nplt.ylabel(\"Number of patients\")\nplt.show()","metadata":{"papermill":{"duration":0.375105,"end_time":"2022-03-15T10:09:45.501857","exception":false,"start_time":"2022-03-15T10:09:45.126752","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-03-15T20:10:42.439928Z","iopub.execute_input":"2022-03-15T20:10:42.440855Z","iopub.status.idle":"2022-03-15T20:10:42.669536Z","shell.execute_reply.started":"2022-03-15T20:10:42.440801Z","shell.execute_reply":"2022-03-15T20:10:42.668666Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.hist(np.asarray(train_img_w), bins=10)\nplt.title(\"Distribution of image widths across all patients in the training set\")\nplt.xlabel(\"Image widths\")\nplt.ylabel(\"Number of patients\")\nplt.show()","metadata":{"papermill":{"duration":0.320475,"end_time":"2022-03-15T10:09:45.972123","exception":false,"start_time":"2022-03-15T10:09:45.651648","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-03-15T20:10:50.320614Z","iopub.execute_input":"2022-03-15T20:10:50.320926Z","iopub.status.idle":"2022-03-15T20:10:50.527954Z","shell.execute_reply.started":"2022-03-15T20:10:50.320894Z","shell.execute_reply":"2022-03-15T20:10:50.527115Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.hist(np.asarray(train_img_h), bins=10)\nplt.title(\"Distribution of image heights across all patients in the training set\")\nplt.xlabel(\"Image heights\")\nplt.ylabel(\"Number of patients\")\nplt.show()","metadata":{"papermill":{"duration":0.326205,"end_time":"2022-03-15T10:09:46.449315","exception":false,"start_time":"2022-03-15T10:09:46.12311","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-03-15T20:10:51.14613Z","iopub.execute_input":"2022-03-15T20:10:51.146404Z","iopub.status.idle":"2022-03-15T20:10:51.355414Z","shell.execute_reply.started":"2022-03-15T20:10:51.146376Z","shell.execute_reply":"2022-03-15T20:10:51.354689Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.hist(np.asarray(train_data.LV_ef), bins=10)\nplt.title(\"Distribution of ejection fraction across all patients in the training set\")\nplt.xlabel(\"Ejection fraction (%)\")\nplt.ylabel(\"Number of patients\")\nplt.xlim(0,100)\nplt.show()","metadata":{"papermill":{"duration":0.33822,"end_time":"2022-03-15T10:09:46.940171","exception":false,"start_time":"2022-03-15T10:09:46.601951","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-03-15T20:10:51.879096Z","iopub.execute_input":"2022-03-15T20:10:51.879984Z","iopub.status.idle":"2022-03-15T20:10:52.09512Z","shell.execute_reply.started":"2022-03-15T20:10:51.879926Z","shell.execute_reply":"2022-03-15T20:10:52.094167Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🔧Model development using 4 channel data\nFor demonstration purpose we only use 4 channel data here. You are free to use both 4 channel and/or 2 channel data for your prediction,","metadata":{"papermill":{"duration":0.152758,"end_time":"2022-03-15T10:09:47.245764","exception":false,"start_time":"2022-03-15T10:09:47.093006","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Because of large filesize we may need to import the image data into Python in batches. \n# This can be done using batch generators.\n\ndef batch_generator(batch_size, gen_x): \n    batch_features = np.zeros((batch_size,10, 256, 256))\n    batch_labels = np.zeros((batch_size,1)) \n    while True:\n        for i in range(batch_size):\n            batch_features[i] , batch_labels[i] = next(gen_x)\n        yield np.expand_dims(batch_features,4), batch_labels\n\ndef generate_data(filelist, img_path, gt_df):\n    while True:\n        for i in filelist:\n            if i.endswith(\".npy\"):\n                img = np.load(img_path + i)\n                img = img[:10]\n                resized_img = np.zeros((10,256,256))\n                for j,k in enumerate(img):\n                    resized_img[j,:,:] = cv2.resize(k, (256,256), interpolation= cv2.INTER_LINEAR )\n                y = float(gt_df.LV_ef[np.where(gt_df.Patient_number == i.split(\"_\")[0])[0]])\n\n                yield resized_img, y","metadata":{"papermill":{"duration":0.165089,"end_time":"2022-03-15T10:09:51.557146","exception":false,"start_time":"2022-03-15T10:09:51.392057","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-03-15T20:10:55.400034Z","iopub.execute_input":"2022-03-15T20:10:55.400346Z","iopub.status.idle":"2022-03-15T20:10:55.41084Z","shell.execute_reply.started":"2022-03-15T20:10:55.400311Z","shell.execute_reply":"2022-03-15T20:10:55.410202Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = batch_generator(5,generate_data(os.listdir(train_4CH_dir),train_4CH_dir,train_data))\n\nprint(\"The shape of one batch of images (batch size = 5):\")\nprint(next(test)[0].shape)","metadata":{"papermill":{"duration":1.535095,"end_time":"2022-03-15T10:09:53.882566","exception":false,"start_time":"2022-03-15T10:09:52.347471","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-03-15T20:10:59.493535Z","iopub.execute_input":"2022-03-15T20:10:59.494139Z","iopub.status.idle":"2022-03-15T20:11:01.792118Z","shell.execute_reply.started":"2022-03-15T20:10:59.494102Z","shell.execute_reply":"2022-03-15T20:11:01.791082Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## The model\nHere we use time distributed convolutional layers and a LSTM layer to give one single prediction from a sequence of images. Feel free to try completly different models (like pre-trained model, 3D convolutions, PCA, etc.)","metadata":{"execution":{"iopub.execute_input":"2022-03-15T10:09:54.200574Z","iopub.status.busy":"2022-03-15T10:09:54.200038Z","iopub.status.idle":"2022-03-15T10:09:58.299636Z","shell.execute_reply":"2022-03-15T10:09:58.299151Z","shell.execute_reply.started":"2022-03-15T08:59:24.357231Z"},"papermill":{"duration":4.25768,"end_time":"2022-03-15T10:09:58.299771","exception":false,"start_time":"2022-03-15T10:09:54.042091","status":"completed"},"tags":[]}},{"cell_type":"code","source":"model = tf.keras.models.Sequential()\nmodel.add(\n    tf.keras.layers.TimeDistributed(\n        tf.keras.layers.Conv2D(32, (3,3), activation='relu'), \n        input_shape=(10, 256, 256, 1) # 5 images...\n    )\n)\nmodel.add(\n    tf.keras.layers.TimeDistributed(\n        tf.keras.layers.GlobalAveragePooling2D() # Or Flatten()\n    )\n)\nmodel.add(\n    tf.keras.layers.LSTM(32, activation='relu', return_sequences=False)\n)\nmodel.add(tf.keras.layers.Dense(1, activation='linear'))\nmodel.compile('adam', loss='mse', metrics=[tf.keras.metrics.RootMeanSquaredError()])","metadata":{"papermill":{"duration":3.35582,"end_time":"2022-03-15T10:10:01.81162","exception":false,"start_time":"2022-03-15T10:09:58.4558","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-03-15T20:11:05.512801Z","iopub.execute_input":"2022-03-15T20:11:05.5131Z","iopub.status.idle":"2022-03-15T20:11:07.01319Z","shell.execute_reply.started":"2022-03-15T20:11:05.513069Z","shell.execute_reply":"2022-03-15T20:11:07.012286Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ⏳ Training\nTo get a better understanding about overfitting/underfitting we encourage you to validate your model using cross validation on the training set. This is not mandatory, but a common practice in machine learning projects.","metadata":{}},{"cell_type":"code","source":"batch_size = 5\nnum_epoch = 5\nsteps = len(os.listdir(train_4CH_dir))//batch_size\nhistory = model.fit(x=batch_generator(batch_size,generate_data(os.listdir(train_4CH_dir),train_4CH_dir,train_data)), epochs=num_epoch, \n                            steps_per_epoch=steps, verbose=0)\n\nfig, (ax1, ax2) = plt.subplots(1, 2)\nfig.set_figheight(10)\nfig.set_figwidth(30)\nax1.plot(history.history[\"loss\"])\nax1.set_title(\"Loss\")\nax2.plot(history.history[\"root_mean_squared_error\"])\nax2.set_title(\"RMSE\")\nplt.show()","metadata":{"papermill":{"duration":3500.646993,"end_time":"2022-03-15T11:08:22.615499","exception":false,"start_time":"2022-03-15T10:10:01.968506","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-03-15T20:11:07.4757Z","iopub.execute_input":"2022-03-15T20:11:07.476013Z","iopub.status.idle":"2022-03-15T20:12:03.726013Z","shell.execute_reply.started":"2022-03-15T20:11:07.475979Z","shell.execute_reply":"2022-03-15T20:12:03.724801Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🔮 Do final prediction on the test set and make submission file based on the sample file","metadata":{"papermill":{"duration":0.773385,"end_time":"2022-03-15T11:08:24.258916","exception":false,"start_time":"2022-03-15T11:08:23.485531","status":"completed"},"tags":[]}},{"cell_type":"code","source":"y_pred=[]\nfor i in tqdm(sorted(os.listdir(test_4CH_dir))):\n    if i.endswith(\".npy\"):\n        img = np.load(test_4CH_dir + i)\n        img = img[:10]\n        resized_img = np.zeros((10,256,256))\n        for j,k in enumerate(img):\n            resized_img[j,:,:] = cv2.resize(k, (256,256), interpolation= cv2.INTER_LINEAR )\n        y_pred.append(model.predict(np.expand_dims(np.expand_dims(resized_img,3),0)))  \n","metadata":{"papermill":{"duration":25.013436,"end_time":"2022-03-15T11:08:50.050096","exception":false,"start_time":"2022-03-15T11:08:25.03666","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-03-15T20:12:03.727423Z","iopub.status.idle":"2022-03-15T20:12:03.728407Z","shell.execute_reply.started":"2022-03-15T20:12:03.72806Z","shell.execute_reply":"2022-03-15T20:12:03.728099Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub.LV_ef = np.asarray(y_pred).ravel()\nsample_sub.head()","metadata":{"papermill":{"duration":0.803256,"end_time":"2022-03-15T11:08:51.644026","exception":false,"start_time":"2022-03-15T11:08:50.84077","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-03-15T20:12:03.730614Z","iopub.status.idle":"2022-03-15T20:12:03.731021Z","shell.execute_reply.started":"2022-03-15T20:12:03.7308Z","shell.execute_reply":"2022-03-15T20:12:03.730828Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub.to_csv(\"submission.csv\",index=False)","metadata":{"papermill":{"duration":0.804245,"end_time":"2022-03-15T11:08:54.892751","exception":false,"start_time":"2022-03-15T11:08:54.088506","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-03-15T20:12:03.732337Z","iopub.status.idle":"2022-03-15T20:12:03.732668Z","shell.execute_reply.started":"2022-03-15T20:12:03.732493Z","shell.execute_reply":"2022-03-15T20:12:03.732518Z"},"trusted":true},"outputs":[],"execution_count":null}]}