{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport json\nimport shutil\nimport csv\nimport random\n\n# Define the folder path where the JSON files are located\njson_folder_path = '/kaggle/input/iwildcam2022-fgvc9/metadata/metadata/iwildcam2022_train_annotations.json'\n\n# Define the folder path where you want to copy the matching files\noutput_folder = '/kaggle/output/miniWild'\n\noutput_train = output_folder + '/train'\noutput_validate = output_folder + '/validate'\n\n\n\n# Define the path to the CSV file\ncsv_path = '/kaggle/input/iwildcam2022-fgvc9/metadata/metadata/train_sequence_counts.csv'\n\nimage_folder = '/kaggle/input/iwildcam2022-fgvc9/train/train'\n\n# Create the output folder if it doesn't exist\nif not os.path.exists(output_folder):\n    os.makedirs(output_folder)\n\nif not os.path.exists(output_train):\n    os.makedirs(output_train)\n\nif not os.path.exists(output_validate):\n    os.makedirs(output_validate)\n\n# Open the CSV file and read the seq_id values\nwith open(csv_path, 'r') as csvfile:\n    json_reader = json.load(open(json_folder_path, 'r'))\n    training_sequences = json_reader['images']\n    csv_reader = csv.reader(csvfile)\n    seqs = list()\n    next(csv_reader)  # Skip the header rows\n    for row in csv_reader:\n        item = list()\n        item.append(row[0])\n        item.append(row[1])\n        seqs.append(item)\n    chosen = random.sample(seqs, 200)\n    other = list()\n    for i in seqs:\n        if i not in chosen:\n            other.append(i)\n    print(len(chosen))\n    print(len(other))\n    for row in chosen:\n        seq_id = row[0]\n        sequenceFolder = output_validate+'/'+seq_id\n        os.makedirs(sequenceFolder)\n        for sequence in training_sequences:\n            if sequence['seq_id'] == seq_id:\n                # Copy the file to the output folder\n                src_path = os.path.join(image_folder, sequence['file_name'])\n                new_name = str(sequence[\"seq_frame_num\"]) + \".jpg\"\n                dst_path = os.path.join(sequenceFolder, new_name)\n                shutil.copy(src_path, dst_path)\n                f = open(sequenceFolder + \"/label.txt\", \"w\")\n                f.write(row[1])\n                f.close()\n    for row in other:\n        seq_id = row[0]\n        sequenceFolder = output_train+'/'+seq_id\n        os.makedirs(sequenceFolder)\n        for sequence in training_sequences:\n            if sequence['seq_id'] == seq_id:\n                # Copy the file to the output folder\n                src_path = os.path.join(image_folder, sequence['file_name'])\n                new_name = str(sequence[\"seq_frame_num\"]) + \".jpg\"\n                dst_path = os.path.join(sequenceFolder, new_name)\n                shutil.copy(src_path, dst_path)\n                f = open(sequenceFolder+\"/label.txt\", \"w\")\n                f.write(row[1])\n                f.close()\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-21T18:43:44.333535Z","iopub.execute_input":"2023-04-21T18:43:44.334151Z","iopub.status.idle":"2023-04-21T18:48:48.802269Z","shell.execute_reply.started":"2023-04-21T18:43:44.334097Z","shell.execute_reply":"2023-04-21T18:48:48.799706Z"},"trusted":true},"execution_count":null,"outputs":[]}]}