{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# clear output folder\nimport shutil\nshutil.rmtree(\"/kaggle/working\")","metadata":{"execution":{"iopub.status.busy":"2023-05-22T16:33:27.154067Z","iopub.execute_input":"2023-05-22T16:33:27.154704Z","iopub.status.idle":"2023-05-22T16:33:27.232131Z","shell.execute_reply.started":"2023-05-22T16:33:27.154658Z","shell.execute_reply":"2023-05-22T16:33:27.230186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport random\nimport csv\nimport shutil\nfrom zipfile import ZipFile\n\n###base_dir = os.path.join('..', 'input', 'diabetic-retinopathy-detection') # running on the current folder\n\nout_dir = os.path.join('')\nlabel_csv = os.path.join(base_dir, 'trainLabels.csv.zip')\nwith ZipFile(label_csv, 'r') as zip:\n    zip.extractall()\nlabel_csv = os.path.join(out_dir, 'trainLabels.csv')\ntrain_dir = os.path.join(out_dir, 'train_ds')\nvalid_dir = os.path.join(out_dir, 'valid_ds')\ntest_dir = os.path.join(out_dir, 'test_ds')\n\nif not os.path.exists(train_dir):\n    os.makedirs(train_dir)\n\nif not os.path.exists(valid_dir):\n    os.makedirs(valid_dir)\n\nif not os.path.exists(test_dir):\n    os.makedirs(test_dir)\n\nsubdirs = ['label0', 'label1', 'label2', 'label3', 'label4']\n\n# open csv file and read to dict img - label\nimg_labels = {}\nwith open(label_csv, 'r') as f:\n    reader = csv.reader(f)\n    next(reader) # skip header row\n    for row in reader:\n        img_name = row[0] + '.jpeg'\n        label = int(row[1])\n        img_labels[img_name] = label\n\n# randomly select one fifth of the images and write their rows to a new CSV file\nimg_number = int(len(list(img_labels.keys()))/5) # change to /2 or /1 for bigger dataset \n\nselected_imgs = random.sample(list(img_labels.keys()), img_number)\nrandom.shuffle(selected_imgs)\nprint(selected_imgs[0])\n# split the selected imgs to train, test and validation\ntrain_ratio = 0.7 \nvalid_ratio = 0.2 \ntest_ratio = 0.1\n\ntrain_split = int(len(selected_imgs) * train_ratio)\nvalid_split = int(len(selected_imgs) * valid_ratio)\n\n# Split the images into train and test sets\ntrain_imgs = selected_imgs[:train_split]\nvalid_imgs = selected_imgs[train_split:train_split+valid_split]\ntest_imgs = selected_imgs[train_split+valid_split:]\n\nfor img_name in train_imgs:\n    label = img_labels[img_name]\n    src_path = os.path.join(base_dir, img_name)\n    dst_path = os.path.join(train_dir, subdirs[label], img_name)\n    shutil.copy(src_path, dst_path)\nfor img_name in valid_imgs:\n    label = img_labels[img_name]\n    src_path = os.path.join(base_dir, img_name)\n    dst_path = os.path.join(valid_dir, subdirs[label], img_name)\n    shutil.copy(src_path, dst_path)\nfor img_name in test_imgs:\n    label = img_labels[img_name]\n    src_path = os.path.join(base_dir, img_name)\n    dst_path = os.path.join(test_dir, subdirs[label], img_name)\n    shutil.copy(src_path, dst_path)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-22T16:36:35.948626Z","iopub.execute_input":"2023-05-22T16:36:35.949171Z","iopub.status.idle":"2023-05-22T16:36:36.135543Z","shell.execute_reply.started":"2023-05-22T16:36:35.949130Z","shell.execute_reply":"2023-05-22T16:36:36.133507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}