{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":12249891,"sourceType":"datasetVersion","datasetId":7718540}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\n\ncsv_dir = \"/kaggle/input/csv-files\"\ncsv_files = [f for f in os.listdir(csv_dir) if f.endswith(\".csv\")]\n\ntotal_rows = 0\nunique_images = set()\n\nfor fn in csv_files:\n    path = os.path.join(csv_dir, fn)\n    df = pd.read_csv(path)\n    # count all rows (each is one crop)\n    n = len(df)\n    total_rows += n\n    \n    # also track unique (study, series, instance) triples\n    unique_images.update(\n        zip(df['study_id'], df['series_id'], df['instance_number'])\n    )\n    \n    print(f\"{fn}: {n} crops\")\n\nprint(f\"\\nTotal crops across all conditions: {total_rows}\")\nprint(f\"Total unique image slices:       {len(unique_images)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-25T19:14:56.000430Z","iopub.execute_input":"2025-07-25T19:14:56.001090Z","iopub.status.idle":"2025-07-25T19:14:56.979402Z","shell.execute_reply.started":"2025-07-25T19:14:56.001055Z","shell.execute_reply":"2025-07-25T19:14:56.978564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# Directory containing the three condition CSVs\ncsv_dir = \"/kaggle/input/csv-files\"\ncsv_files = [os.path.join(csv_dir, f) \n             for f in os.listdir(csv_dir) if f.endswith(\".csv\")]\n\n# Read and concatenate all rows\nall_df = pd.concat((pd.read_csv(f) for f in csv_files), ignore_index=True)\n\n# Total number of crops (rows)\ntotal_crops = len(all_df)\n\n# Count by severity\nseverity_counts = all_df['score'].value_counts()\n\nprint(f\"Total crops with x,y coordinates: {total_crops}\\n\")\nprint(\"Counts by severity:\")\nprint(severity_counts.to_string())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-25T19:15:06.220051Z","iopub.execute_input":"2025-07-25T19:15:06.220355Z","iopub.status.idle":"2025-07-25T19:15:06.505808Z","shell.execute_reply.started":"2025-07-25T19:15:06.220333Z","shell.execute_reply":"2025-07-25T19:15:06.505223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\ncsv_dir = \"/kaggle/input/csv-files\"\ncsv_files = [f for f in os.listdir(csv_dir) if f.endswith(\".csv\")]\n\nprint(\"Counts per condition (rows = number of x,y crops):\\n\")\nfor fn in csv_files:\n    path = os.path.join(csv_dir, fn)\n    df = pd.read_csv(path)\n    counts = df['score'].value_counts()\n    total = len(df)\n    print(f\"{fn}: {total} total\")\n    for severity, cnt in counts.items():\n        print(f\"  {severity}: {cnt}\")\n    print()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-25T19:34:28.250975Z","iopub.execute_input":"2025-07-25T19:34:28.251348Z","iopub.status.idle":"2025-07-25T19:34:28.528834Z","shell.execute_reply.started":"2025-07-25T19:34:28.251324Z","shell.execute_reply":"2025-07-25T19:34:28.527936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# Path to CSVs (adjust to your Kaggle input folder)\ncsv_dir = \"/kaggle/input/csv-files\"\ncsv_files = [os.path.join(csv_dir, f) for f in os.listdir(csv_dir) if f.endswith(\".csv\")]\n\n# Combine all into one DataFrame\nall_df = pd.concat((pd.read_csv(f) for f in csv_files), ignore_index=True)\n\nprint(\"Total crops:\", len(all_df))\nprint(\"Unique slices:\", all_df[['study_id','series_id','instance_number']].drop_duplicates().shape[0])\nall_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-07T06:34:43.338556Z","iopub.execute_input":"2025-09-07T06:34:43.338827Z","iopub.status.idle":"2025-09-07T06:34:45.945600Z","shell.execute_reply.started":"2025-09-07T06:34:43.338801Z","shell.execute_reply":"2025-09-07T06:34:45.944817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"condition_counts = all_df.groupby(['condition', 'score']).size().unstack(fill_value=0)\nprint(condition_counts)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-07T06:34:45.947343Z","iopub.execute_input":"2025-09-07T06:34:45.947575Z","iopub.status.idle":"2025-09-07T06:34:45.995071Z","shell.execute_reply.started":"2025-09-07T06:34:45.947556Z","shell.execute_reply":"2025-09-07T06:34:45.994288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(8,5))\nsns.countplot(data=all_df, x='score', order=['Normal/Mild','Moderate','Severe'], palette=\"Set2\")\nplt.title(\"Severity Distribution Across All Conditions\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-07T06:34:45.995974Z","iopub.execute_input":"2025-09-07T06:34:45.996256Z","iopub.status.idle":"2025-09-07T06:34:47.168194Z","shell.execute_reply.started":"2025-09-07T06:34:45.996228Z","shell.execute_reply":"2025-09-07T06:34:47.167250Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"condition_counts.plot(\n    kind=\"bar\",\n    stacked=True,\n    figsize=(10,6),\n    colormap=\"viridis\"\n)\nplt.title(\"Condition × Severity Distribution\")\nplt.ylabel(\"Number of crops\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-07T06:34:47.169110Z","iopub.execute_input":"2025-09-07T06:34:47.169598Z","iopub.status.idle":"2025-09-07T06:34:47.462928Z","shell.execute_reply.started":"2025-09-07T06:34:47.169569Z","shell.execute_reply":"2025-09-07T06:34:47.461935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_df['score'].value_counts().plot(\n    kind=\"pie\",\n    autopct='%1.1f%%',\n    startangle=90,\n    colors=[\"#66c2a5\",\"#fc8d62\",\"#8da0cb\"]\n)\nplt.ylabel(\"\")\nplt.title(\"Overall Severity Class Distribution\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-07T06:34:47.466397Z","iopub.execute_input":"2025-09-07T06:34:47.466723Z","iopub.status.idle":"2025-09-07T06:34:47.595679Z","shell.execute_reply.started":"2025-09-07T06:34:47.466696Z","shell.execute_reply":"2025-09-07T06:34:47.594763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}