{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport random\nfrom shutil import copy, make_archive\nfrom tqdm.notebook import tqdm\nimport pandas as pd\n\n\ndf=pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/train.csv\")\n\n\ndf_0 = df[df.target==0]\ndf_1 = df[df.target==1]\ndf_1 = df_1.sample(500)\ndf_0 = df_0.sample(2500)\n\ndf = pd.concat([df_0, df_1])\ndata_root = '/kaggle/input/siim-isic-melanoma-classification'\n\n\n\nos.makedirs('./dataset', exist_ok=True) # create a dataset folder to hold all the files that I wanted to download\ncopy(os.path.join(data_root, 'sample_submission.csv'), 'dataset/sample_submission.csv')\ncopy(os.path.join(data_root, 'train.csv'), 'dataset/train.csv')\ncopy(os.path.join(data_root, 'test.csv'), 'dataset/test.csv')\n\ntrain_target_dir = os.path.join('dataset', 'train')\nos.makedirs(train_target_dir, exist_ok=True) \n\ncount=0\n# iterate over df and copy images to dataset folder\nfor i, row in tqdm(df.iterrows(), total=len(df)):\n    if count%200==0:\n        print(\"copy train file \", count)\n    img_id = row['image_name']\n    img_label = row['target']\n    src_file = os.path.join(data_root, 'jpeg', 'train', f'{img_id}.jpg')\n    copy(src_file, train_target_dir)\n    count+=1\n    \nprint(\"now copy test\")\ntest_path='jpeg/test'\nk=1000\ndir_path = os.path.join(data_root, test_path)\nfiles = os.listdir(dir_path)\n\n# copy images to target folder\ntarget_dir = os.path.join('dataset', 'test')\nos.makedirs(target_dir, exist_ok=True) \nfor f in tqdm(random.choices(files, k=k)): # randomly select k images and copy them to the target folder\n    src_file = os.path.join(dir_path, f)\n    copy(src_file, target_dir)\n        \nprint(\"now creating archive\")\n# zip generated files\nmake_archive(base_name='download_dataset', format='zip', root_dir='dataset')\n ","metadata":{"execution":{"iopub.status.busy":"2024-05-06T07:07:47.774082Z","iopub.execute_input":"2024-05-06T07:07:47.774511Z","iopub.status.idle":"2024-05-06T07:12:30.947826Z","shell.execute_reply.started":"2024-05-06T07:07:47.774477Z","shell.execute_reply":"2024-05-06T07:12:30.946113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}