{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport glob\n\nPATH_DATASET = \"/kaggle/input/physionet-ecg-image-digitization\"\n\ndataset = {}\nfor pdir in glob.glob(os.path.join(PATH_DATASET, \"train\", \"*\")):\n    spl = os.path.basename(pdir)\n    imgs = glob.glob(os.path.join(pdir, \"*.png\"))\n    dataset[spl] = len(imgs)\n\nprint(f\"samples: {set(dataset.values())}\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-28T08:52:48.043116Z","iopub.execute_input":"2025-10-28T08:52:48.043329Z","iopub.status.idle":"2025-10-28T08:52:52.785530Z","shell.execute_reply.started":"2025-10-28T08:52:48.043309Z","shell.execute_reply":"2025-10-28T08:52:52.784641Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport random\n\n# Choose a random subfolder from the dataset\nsubfolder = random.choice(list(dataset.keys()))\nsubfolder_path = os.path.join(PATH_DATASET, \"train\", subfolder)\n\n# Get all image files in the chosen subfolder\nimage_files = glob.glob(os.path.join(subfolder_path, \"*.png\"))\n\n# Create a 3x3 grid to display the images\nfig, axes = plt.subplots(3, 3, figsize=(12, 12))\naxes = axes.ravel() # Flatten the 2D array of axes for easy iteration\n\n# Display each image\nfor i, img_path in enumerate(image_files):\n    img = mpimg.imread(img_path)\n    axes[i].imshow(img)\n    axes[i].axis('off') # Hide the axes\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T08:52:52.787495Z","iopub.execute_input":"2025-10-28T08:52:52.787834Z","iopub.status.idle":"2025-10-28T08:53:11.106117Z","shell.execute_reply.started":"2025-10-28T08:52:52.787806Z","shell.execute_reply":"2025-10-28T08:53:11.105184Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Assuming the CSV file has the same name as the subfolder with a .csv extension\ncsv_file_name = f\"{os.path.basename(subfolder_path)}.csv\"\ncsv_file_path = os.path.join(subfolder_path, csv_file_name)\n\n# Read the CSV file into a pandas DataFrame\ndf = pd.read_csv(csv_file_path)\nprint(f\"CSV file '{csv_file_name}' loaded successfully.\")\ndisplay(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T08:57:15.553509Z","iopub.execute_input":"2025-10-28T08:57:15.554418Z","iopub.status.idle":"2025-10-28T08:57:15.933975Z","shell.execute_reply.started":"2025-10-28T08:57:15.554386Z","shell.execute_reply":"2025-10-28T08:57:15.932943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# Function to find continuous non-NaN intervals in a pandas Series\ndef find_non_nan_intervals(series):\n    intervals = []\n    start_index = None\n    for i, value in series.items():\n        if pd.notna(value):\n            if start_index is None:\n                start_index = i\n        elif start_index is not None:\n            intervals.append((start_index, i - 1))\n            start_index = None\n    if start_index is not None:  # Handle case where non-NaN values extend to the end\n        intervals.append((start_index, len(series) - 1))\n    return intervals\n\n# Identify intervals for each column\ncolumn_intervals = {}\nfor col in df.columns:\n    column_intervals[col] = find_non_nan_intervals(df[col])\n\n# Display the identified intervals\nfor col, intervals in column_intervals.items():\n    print(f\"Intervals for column '{col}': {intervals}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T08:57:18.546794Z","iopub.execute_input":"2025-10-28T08:57:18.547490Z","iopub.status.idle":"2025-10-28T08:57:18.617401Z","shell.execute_reply.started":"2025-10-28T08:57:18.547457Z","shell.execute_reply":"2025-10-28T08:57:18.616509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot all non-NaN intervals in a single chart with different colors\nplt.figure(figsize=(12, 6))\n\n# Get a colormap and create a list of colors\ncmap = plt.colormaps.get_cmap('tab10')\ncolors = [cmap(i) for i in np.linspace(0, 1, len(df.columns))]\n\nfor i, (col, intervals) in enumerate(column_intervals.items()):\n    for start, end in intervals:\n        plt.plot(df.index[start:end+1], df[col].iloc[start:end+1], color=colors[i], label=f'{col} ({start}-{end})')\n\nplt.title(f\"All Non-NaN Intervals from {os.path.basename(subfolder_path)}.csv\")\nplt.xlabel(\"Index\")\nplt.ylabel(\"Value\")\n# Place the legend outside the plot\nplt.legend(loc='center left', bbox_to_anchor=(1, 0.5))\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T09:02:00.284206Z","iopub.execute_input":"2025-10-28T09:02:00.285016Z","iopub.status.idle":"2025-10-28T09:02:00.628449Z","shell.execute_reply.started":"2025-10-28T09:02:00.284988Z","shell.execute_reply":"2025-10-28T09:02:00.627598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a combined plot with interval plots and histograms side-by-side\nn_cols_grid = 2  # Two columns in the grid\nn_rows_grid = len(df.columns) # One row for each column\n\nfig, axes = plt.subplots(nrows=n_rows_grid, ncols=n_cols_grid, figsize=(16, 2 * n_rows_grid))\naxes = axes.ravel() # Flatten the axes array for easy iteration\n\n# Determine the overall x-axis limits for interval plots\nall_indices = df.index\nmin_x = all_indices.min()\nmax_x = all_indices.max()\n\nfor i, col in enumerate(df.columns):\n    # Plot intervals in the left column\n    ax_intervals = axes[i * n_cols_grid]\n    intervals = column_intervals.get(col, []) # Get intervals for the current column\n    for start, end in intervals:\n        ax_intervals.plot(df.index[start:end+1], df[col].iloc[start:end+1], label=f'{col} ({start}-{end})')\n    ax_intervals.set_title(f'Column: {col} (Intervals)')\n    ax_intervals.set_xlabel(\"Index\")\n    ax_intervals.set_ylabel(\"Value\")\n    ax_intervals.grid(True)\n    ax_intervals.legend(loc='upper right')\n    ax_intervals.set_xlim([min_x, max_x]) # Set the same x-axis limits for interval plots\n\n    # Plot histogram in the right column\n    ax_hist = axes[i * n_cols_grid + 1]\n    df[col].hist(bins=50, ax=ax_hist)\n    ax_hist.set_title(f\"Histogram of {col}\")\n    ax_hist.set_xlabel(\"Value\")\n    ax_hist.set_ylabel(\"Frequency\")\n    ax_hist.grid(True)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-10-28T09:07:16.164283Z","iopub.execute_input":"2025-10-28T09:07:16.164566Z","iopub.status.idle":"2025-10-28T09:07:20.628882Z","shell.execute_reply.started":"2025-10-28T09:07:16.164547Z","shell.execute_reply":"2025-10-28T09:07:20.628013Z"}},"outputs":[],"execution_count":null}]}