{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35273,"databundleVersionId":3351394,"sourceType":"competition"}],"dockerImageVersionId":30839,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <p style=\"background-color: #113966; font-family: 'Arial', sans-serif; font-size: 32px; text-align: center; color: #fc7f03; padding: 15px; border-radius: 25px; text-shadow: 2px 2px 6px rgba(0, 0, 0, 0.5); border: 3px solid #fc7f03;\">Cardiac Echocardiogram Model Evaluation and Testing</p>","metadata":{}},{"cell_type":"markdown","source":"![](https://cdn.prod.website-files.com/5ff99f5f7aec6fe4770f6b92/616f144e8b372e2724a33eb0_Apical%204-Chamber%20View%20(A4C).png)","metadata":{}},{"cell_type":"markdown","source":"# Reading Data","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-29T05:40:05.668731Z","iopub.execute_input":"2025-01-29T05:40:05.669044Z","iopub.status.idle":"2025-01-29T05:40:07.007375Z","shell.execute_reply.started":"2025-01-29T05:40:05.669019Z","shell.execute_reply":"2025-01-29T05:40:07.006340Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Example of a .npy file with shape (num_images, height, width, channels)\n\nUse Case in Echocardiography\nEach 4-channel image could represent:\n\n1. B-mode grayscale image\n2. Color Doppler data\n3. Tissue Doppler Imaging (TDI)\n4. Strain or velocity map\n\nThis format is useful for deep learning applications, where neural networks (like CNNs) can process multiple channels simultaneously.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport os\nimport pandas as pd\nimport cv2\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm    #  tqdm library provides a progress bar for loops in Python. It is especially useful when working with large datasets, training deep learning models, or processing large .npy files.\nimport tensorflow as tf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-29T05:50:09.625532Z","iopub.execute_input":"2025-01-29T05:50:09.625925Z","iopub.status.idle":"2025-01-29T05:50:25.784757Z","shell.execute_reply.started":"2025-01-29T05:50:09.625894Z","shell.execute_reply":"2025-01-29T05:50:25.783639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the dataset\nloaded_data_example = np.load(\"/kaggle/input/echo2022/train_data/train_data/2CH/patient076_2CH_sequence.npy\")\n\n# Print dataset shape\nprint(\"Loaded data shape:\", loaded_data_example.shape)\nloaded_data_example","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-29T08:11:41.262095Z","iopub.execute_input":"2025-01-29T08:11:41.262486Z","iopub.status.idle":"2025-01-29T08:11:41.292640Z","shell.execute_reply.started":"2025-01-29T08:11:41.262448Z","shell.execute_reply":"2025-01-29T08:11:41.291628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = pd.read_csv(\"../input/echo2022/train_data.csv\")\ntrain_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-29T05:47:59.978399Z","iopub.execute_input":"2025-01-29T05:47:59.978835Z","iopub.status.idle":"2025-01-29T05:48:00.027122Z","shell.execute_reply.started":"2025-01-29T05:47:59.978802Z","shell.execute_reply":"2025-01-29T05:48:00.025909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"../input/echo2022/sample_submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-29T05:49:08.020191Z","iopub.execute_input":"2025-01-29T05:49:08.020593Z","iopub.status.idle":"2025-01-29T05:49:08.029010Z","shell.execute_reply.started":"2025-01-29T05:49:08.020559Z","shell.execute_reply":"2025-01-29T05:49:08.028134Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_2CH_dir = \"../input/echo2022/train_data/train_data/2CH/\"\ntrain_4CH_dir = \"../input/echo2022/train_data/train_data/4CH/\"\ntest_2CH_dir = \"../input/echo2022/test_data/test_data/2CH/\"\ntest_4CH_dir = \"../input/echo2022/test_data/test_data/4CH/\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-29T05:49:12.517270Z","iopub.execute_input":"2025-01-29T05:49:12.517669Z","iopub.status.idle":"2025-01-29T05:49:12.522230Z","shell.execute_reply.started":"2025-01-29T05:49:12.517635Z","shell.execute_reply":"2025-01-29T05:49:12.521015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import metadata about the echo images\nsequence_number = []\ntrain_img_w = []\ntrain_img_h = []\nfor i in tqdm(os.listdir(train_2CH_dir)):\n    if i.endswith(\".npy\"):\n        number, width, height = np.load(train_2CH_dir + i).shape\n        sequence_number.append(number)\n        train_img_w.append(width)\n        train_img_h.append(height)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-29T05:51:22.012380Z","iopub.execute_input":"2025-01-29T05:51:22.013120Z","iopub.status.idle":"2025-01-29T05:52:55.772827Z","shell.execute_reply.started":"2025-01-29T05:51:22.013087Z","shell.execute_reply":"2025-01-29T05:52:55.771376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Because of large filesize we may need to import the image data into Python in batches. \n# This can be done using batch generators.\n\ndef batch_generator(batch_size, gen_x): \n    batch_features = np.zeros((batch_size,10, 256, 256))\n    batch_labels = np.zeros((batch_size,1)) \n    while True:\n        for i in range(batch_size):\n            batch_features[i] , batch_labels[i] = next(gen_x)\n        yield np.expand_dims(batch_features,4), batch_labels\n\ndef generate_data(filelist, img_path, gt_df):\n    while True:\n        for i in filelist:\n            if i.endswith(\".npy\"):\n                img = np.load(img_path + i)\n                img = img[:10]\n                resized_img = np.zeros((10,256,256))\n                for j,k in enumerate(img):\n                    resized_img[j,:,:] = cv2.resize(k, (256,256), interpolation= cv2.INTER_LINEAR )\n                y = float(gt_df.LV_ef[np.where(gt_df.Patient_number == i.split(\"_\")[0])[0]])\n\n                yield resized_img, y","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-29T05:58:20.542258Z","iopub.execute_input":"2025-01-29T05:58:20.542751Z","iopub.status.idle":"2025-01-29T05:58:20.551854Z","shell.execute_reply.started":"2025-01-29T05:58:20.542698Z","shell.execute_reply":"2025-01-29T05:58:20.550262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = batch_generator(5,generate_data(os.listdir(train_4CH_dir),train_4CH_dir,train_data))\n\nprint(\"The shape of one batch of images (batch size = 5):\")\nprint(next(test)[0].shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-29T05:58:40.409488Z","iopub.execute_input":"2025-01-29T05:58:40.409844Z","iopub.status.idle":"2025-01-29T05:58:41.719706Z","shell.execute_reply.started":"2025-01-29T05:58:40.409817Z","shell.execute_reply":"2025-01-29T05:58:41.718331Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Training","metadata":{}},{"cell_type":"markdown","source":"**The model is a ConvLSTM architecture where CNN extracts spatial features, and LSTM captures temporal dependencies.**","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\nmodel = tf.keras.models.Sequential()\n\n# CNN feature extraction\nmodel.add(\n    tf.keras.layers.TimeDistributed(\n        tf.keras.layers.Conv2D(32, (3,3), activation='relu', padding='same'), \n        input_shape=(10, 256, 256, 1)\n    )\n)\nmodel.add(tf.keras.layers.TimeDistributed(tf.keras.layers.BatchNormalization()))\nmodel.add(tf.keras.layers.TimeDistributed(tf.keras.layers.MaxPooling2D((2,2))))\n\nmodel.add(tf.keras.layers.TimeDistributed(tf.keras.layers.Conv2D(64, (3,3), activation='relu', padding='same')))\nmodel.add(tf.keras.layers.TimeDistributed(tf.keras.layers.BatchNormalization()))\nmodel.add(tf.keras.layers.TimeDistributed(tf.keras.layers.MaxPooling2D((2,2))))\n\nmodel.add(tf.keras.layers.TimeDistributed(tf.keras.layers.Conv2D(128, (3,3), activation='relu', padding='same')))\nmodel.add(tf.keras.layers.TimeDistributed(tf.keras.layers.BatchNormalization()))\nmodel.add(tf.keras.layers.TimeDistributed(tf.keras.layers.MaxPooling2D((2,2))))\n\n# Flatten before LSTM\nmodel.add(tf.keras.layers.TimeDistributed(tf.keras.layers.GlobalAveragePooling2D()))\n\n# LSTM for temporal dependencies\nmodel.add(tf.keras.layers.LSTM(64, return_sequences=False, dropout=0.2, recurrent_dropout=0.2))\n\n# Fully connected output\nmodel.add(tf.keras.layers.Dense(32, activation='relu'))\nmodel.add(tf.keras.layers.Dense(1, activation='linear'))\n\n# Compile with AdamW optimizer\noptimizer = tf.keras.optimizers.AdamW(learning_rate=1e-4, weight_decay=1e-5)\nmodel.compile(optimizer=optimizer, loss='mse', metrics=[tf.keras.metrics.RootMeanSquaredError()])\n\n# Model summary\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-29T06:09:54.349468Z","iopub.execute_input":"2025-01-29T06:09:54.350005Z","iopub.status.idle":"2025-01-29T06:09:54.810946Z","shell.execute_reply.started":"2025-01-29T06:09:54.349872Z","shell.execute_reply":"2025-01-29T06:09:54.809563Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"✅ Batch Normalization for faster convergence.\n\n✅ MaxPooling2D instead of GlobalPooling to retain spatial features.\n\n✅ More CNN filters to extract richer features.\n\n✅ Dropout & Recurrent Dropout to prevent overfitting.\n\n✅ AdamW optimizer for better generalization.","metadata":{}},{"cell_type":"markdown","source":"# Model Fitting","metadata":{}},{"cell_type":"code","source":"batch_size = 5\nnum_epoch = 5\nsteps = len(os.listdir(train_4CH_dir))//batch_size\nhistory = model.fit(x=batch_generator(batch_size,generate_data(os.listdir(train_4CH_dir),train_4CH_dir,train_data)), epochs=num_epoch, \n                            steps_per_epoch=steps, verbose=0)\n\nfig, (ax1, ax2) = plt.subplots(1, 2)\nfig.set_figheight(10)\nfig.set_figwidth(30)\nax1.plot(history.history[\"loss\"])\nax1.set_title(\"Loss\")\nax2.plot(history.history[\"root_mean_squared_error\"])\nax2.set_title(\"RMSE\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-29T06:10:52.801058Z","iopub.execute_input":"2025-01-29T06:10:52.801441Z","iopub.status.idle":"2025-01-29T08:11:40.906162Z","shell.execute_reply.started":"2025-01-29T06:10:52.801386Z","shell.execute_reply":"2025-01-29T08:11:40.903735Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Prediction","metadata":{}},{"cell_type":"code","source":"y_pred=[]\nfor i in tqdm(sorted(os.listdir(test_4CH_dir))):\n    if i.endswith(\".npy\"):\n        img = np.load(test_4CH_dir + i)\n        img = img[:10]\n        resized_img = np.zeros((10,256,256))\n        for j,k in enumerate(img):\n            resized_img[j,:,:] = cv2.resize(k, (256,256), interpolation= cv2.INTER_LINEAR )\n        y_pred.append(model.predict(np.expand_dims(np.expand_dims(resized_img,3),0)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-29T08:11:41.294114Z","iopub.execute_input":"2025-01-29T08:11:41.294520Z","iopub.status.idle":"2025-01-29T08:12:15.391761Z","shell.execute_reply.started":"2025-01-29T08:11:41.294487Z","shell.execute_reply":"2025-01-29T08:12:15.390701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub.LV_ef = np.asarray(y_pred).ravel()\nsample_sub","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-29T08:12:15.393086Z","iopub.execute_input":"2025-01-29T08:12:15.393653Z","iopub.status.idle":"2025-01-29T08:12:15.435726Z","shell.execute_reply.started":"2025-01-29T08:12:15.393609Z","shell.execute_reply":"2025-01-29T08:12:15.434617Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub.to_csv(\"submission.csv\",index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}