{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-25T13:41:24.129123Z","iopub.execute_input":"2023-04-25T13:41:24.129644Z","iopub.status.idle":"2023-04-25T13:41:24.140032Z","shell.execute_reply.started":"2023-04-25T13:41:24.129593Z","shell.execute_reply":"2023-04-25T13:41:24.138370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!unzip ../input/diabetic-retinopathy-detection/sample.zip\n!unzip ../input/diabetic-retinopathy-detection/sampleSubmission.csv.zip\n! dir sample\n!unzip ../input/diabetic-retinopathy-detection/trainLabels.csv.zip\n","metadata":{"execution":{"iopub.status.busy":"2023-04-25T13:41:28.013139Z","iopub.execute_input":"2023-04-25T13:41:28.013767Z","iopub.status.idle":"2023-04-25T13:41:32.772242Z","shell.execute_reply.started":"2023-04-25T13:41:28.013692Z","shell.execute_reply":"2023-04-25T13:41:32.770828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Check number of sample images**","metadata":{}},{"cell_type":"code","source":"data_dir = '/kaggle/working/sample'\nprint('Number of images:', len(os.listdir(data_dir)))\n","metadata":{"execution":{"iopub.status.busy":"2023-04-25T13:49:59.932750Z","iopub.execute_input":"2023-04-25T13:49:59.933391Z","iopub.status.idle":"2023-04-25T13:49:59.942506Z","shell.execute_reply.started":"2023-04-25T13:49:59.933346Z","shell.execute_reply":"2023-04-25T13:49:59.940165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Looking into images","metadata":{}},{"cell_type":"code","source":"f, axarr = plt.subplots(2,2, figsize=(10, 10))\naxarr[0,0].imshow(Image.open(\"./sample/10_right.jpeg\"))\naxarr[0,1].imshow(Image.open(\"./sample/13_right.jpeg\"))\naxarr[1,0].imshow(Image.open(\"./sample/15_right.jpeg\"))\naxarr[1,1].imshow(Image.open(\"./sample/17_right.jpeg\"))","metadata":{"execution":{"iopub.status.busy":"2023-04-25T13:46:59.518797Z","iopub.execute_input":"2023-04-25T13:46:59.520123Z","iopub.status.idle":"2023-04-25T13:47:10.877240Z","shell.execute_reply.started":"2023-04-25T13:46:59.520060Z","shell.execute_reply":"2023-04-25T13:47:10.875896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Checking data distribution","metadata":{}},{"cell_type":"code","source":"# Load the dataset\ndf = pd.read_csv('/kaggle/working/trainLabels.csv')\n\n# Step 1: Check the shape of the data\nprint('Number of rows:', df.shape[0])\nprint('Number of columns:', df.shape[1])\n\n","metadata":{"execution":{"iopub.status.busy":"2023-04-25T13:50:07.670505Z","iopub.execute_input":"2023-04-25T13:50:07.671611Z","iopub.status.idle":"2023-04-25T13:50:07.705651Z","shell.execute_reply.started":"2023-04-25T13:50:07.671561Z","shell.execute_reply":"2023-04-25T13:50:07.703894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Check sample image size**","metadata":{}},{"cell_type":"code","source":"sample_img = Image.open(os.path.join(data_dir, os.listdir(data_dir)[0]))\nprint('Image size:', sample_img.size)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T13:50:49.198177Z","iopub.execute_input":"2023-04-25T13:50:49.198778Z","iopub.status.idle":"2023-04-25T13:50:49.209518Z","shell.execute_reply.started":"2023-04-25T13:50:49.198727Z","shell.execute_reply":"2023-04-25T13:50:49.207812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Check the image mode**","metadata":{}},{"cell_type":"code","source":"print('Image mode:', sample_img.mode)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T13:51:10.038299Z","iopub.execute_input":"2023-04-25T13:51:10.038816Z","iopub.status.idle":"2023-04-25T13:51:10.047027Z","shell.execute_reply.started":"2023-04-25T13:51:10.038763Z","shell.execute_reply":"2023-04-25T13:51:10.044824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Checking Data Distribution**","metadata":{}},{"cell_type":"code","source":"widths = []\nheights = []\n\nfor img_file in os.listdir(data_dir):\n    img = Image.open(os.path.join(data_dir, img_file))\n    width, height = img.size\n    widths.append(width)\n    heights.append(height)\n\nprint('Average image size:', np.mean(widths), 'x', np.mean(heights))\n\n# Check the distribution of image sizes\nfig, axs = plt.subplots(1, 2, figsize=(15, 6))\naxs[0].hist(widths, bins=50)\naxs[0].set_xlabel('Image width')\naxs[0].set_ylabel('Frequency')\naxs[1].hist(heights, bins=50)\naxs[1].set_xlabel('Image height')\naxs[1].set_ylabel('Frequency')\nplt.show()\n\n# Check the distribution of image modes\nmodes = []\n\nfor img_file in os.listdir(data_dir):\n    img = Image.open(os.path.join(data_dir, img_file))\n    modes.append(img.mode)\n\nprint('Image modes:', set(modes))\n","metadata":{"execution":{"iopub.status.busy":"2023-04-25T13:56:29.896399Z","iopub.execute_input":"2023-04-25T13:56:29.896978Z","iopub.status.idle":"2023-04-25T13:56:30.652057Z","shell.execute_reply.started":"2023-04-25T13:56:29.896931Z","shell.execute_reply":"2023-04-25T13:56:30.650798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Check for class distribution**","metadata":{}},{"cell_type":"code","source":"labels_df = pd.read_csv('/kaggle/working/trainLabels.csv')\nlabels_df['level'].hist(bins=5)\nplt.xlabel('Class')\nplt.ylabel('Frequency')\nplt.title('Class Distribution')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-25T13:58:35.939675Z","iopub.execute_input":"2023-04-25T13:58:35.940206Z","iopub.status.idle":"2023-04-25T13:58:36.254911Z","shell.execute_reply.started":"2023-04-25T13:58:35.940154Z","shell.execute_reply":"2023-04-25T13:58:36.253160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Checking for class Imbalance**","metadata":{}},{"cell_type":"code","source":"class_counts = labels_df['level'].value_counts()\ntotal_samples = class_counts.sum()\n\nfor i in range(5):\n    count = class_counts[i]\n    percent = (count / total_samples) * 100\n    print('Level', i, ':', count, 'samples (', percent, '% of total )')\n","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:00:11.328456Z","iopub.execute_input":"2023-04-25T14:00:11.329040Z","iopub.status.idle":"2023-04-25T14:00:11.345852Z","shell.execute_reply.started":"2023-04-25T14:00:11.328994Z","shell.execute_reply":"2023-04-25T14:00:11.343435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Analysing the distribution of target variable**","metadata":{}},{"cell_type":"code","source":"labels_df['level'].value_counts().plot(kind='pie', autopct='%1.1f%%', startangle=90, colors=['#7FB3D5', '#F7CAC9', '#EEDD82', '#FFA07A', '#90EE90'])\nplt.axis('equal')\nplt.legend(labels=['No DR', 'Mild', 'Moderate', 'Severe', 'Proliferative DR'], loc='upper left', bbox_to_anchor=(-0.1, 1.))\nplt.title('Distribution of Diabetic Retinopathy Levels')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-25T14:02:48.603548Z","iopub.execute_input":"2023-04-25T14:02:48.604483Z","iopub.status.idle":"2023-04-25T14:02:48.799963Z","shell.execute_reply.started":"2023-04-25T14:02:48.604421Z","shell.execute_reply":"2023-04-25T14:02:48.798045Z"},"trusted":true},"execution_count":null,"outputs":[]}]}