{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from ast import literal_eval\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nimport numpy as np\nfrom sklearn.model_selection import StratifiedKFold\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nclass_ids = [6, 7, 8, 9, 10, 25, 41, 105, 110, 115, 148, 156, 222, 228, 235, 256, \n             280, 310, 387, 392, 394, 398, 401, 402, 430, 480, 485, 673]\nname = ['大螟', '二化螟', '稻纵卷叶螟', '白背飞虱', '褐飞虱属', '地老虎', '蝼蛄', '粘虫', '草地螟', '甜菜夜蛾', '黄足猎蝽', '八点灰灯蛾', \n         '棉铃虫', '二点委夜蛾', '甘蓝夜蛾', '蟋蟀', '黄毒蛾', '稻螟蛉', '紫条尺蛾', '水螟蛾', '线委夜蛾', '甜菜白带野螟', '歧角螟', \n        '瓜绢野螟', '豆野螟', '石蛾', '大黑鳃金龟', '干纹冬夜蛾']\n\nN_FOLDS = 6\nseed = 2022\nTRAIN_PATH = '../input/tddatafinal/marge23'","metadata":{"execution":{"iopub.status.busy":"2022-04-28T06:34:48.389984Z","iopub.execute_input":"2022-04-28T06:34:48.390564Z","iopub.status.idle":"2022-04-28T06:34:48.481627Z","shell.execute_reply.started":"2022-04-28T06:34:48.390512Z","shell.execute_reply":"2022-04-28T06:34:48.480557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = pd.read_csv('../input/taidicup2022dataset/TaiDiCup2022DataSet/图片虫子位置详情表.csv',encoding='gbk', index_col=0, \n                 names=['file_path', 'class_id','Chinese_name','x','y','xmin','ymin','xmax','ymax'],header=0)\ndf = pd.read_csv('../input/td-csv/finalout.csv',encoding='gbk', index_col=0, \n                 names=['file_path', 'class_id','xmin','ymin','xmax','ymax','label'],header=0)\nempty_path = []\nfor i in range(len(df1)):\n    if df1.iloc[i,1]==0:\n        empty_path.append(df1.iloc[i,0])\ndf = df[df['label']==1].reset_index(drop = True)\n# df = df[df['class_id']!=0].reset_index(drop = True)\nfor i in range(len(class_ids)):\n    df.class_id.replace(class_ids[i], i, inplace=True)\n        \ndf['fold'] = -1\nstrat_kfold = StratifiedKFold(n_splits=N_FOLDS, random_state=seed, shuffle=True)\n\nfor i, (_, train_index) in enumerate(strat_kfold.split(df.index, df['class_id'])):\n    df.iloc[train_index, -1] = i\ndf['fold'] = df['fold'].astype('int')\ndf = df.sort_values(by='file_path', ascending=True).reset_index(drop = True)\n\nlast_file = ''\nfor i in range(len(df)):\n    if df.iloc[i,0]!=last_file:\n        fold_t = df.iloc[i,-1]\n        last_file = df.iloc[i,0]\n    else:\n        df.iloc[i,-1]=fold_t\n\nplt.figure(figsize=(8,6),dpi=300)\nsns.jointplot(x='fold', y='class_id', data=df,\n              kind='kde', fill=True, thresh=0, size=8, cmap='BuPu')#Pastel1_r\ndf.head(5)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-04-28T06:34:48.666221Z","iopub.execute_input":"2022-04-28T06:34:48.666598Z","iopub.status.idle":"2022-04-28T06:34:56.846388Z","shell.execute_reply.started":"2022-04-28T06:34:48.666556Z","shell.execute_reply":"2022-04-28T06:34:56.845490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def coco(df):\n    images = []\n    annotations = []\n    categories = []\n    exist_class = []\n    annotion_id = 0\n    last_file = ''\n    idx = -1\n    for i in range(27):\n        categories.append({\n        'id':i,\n        'name':str(i)\n    })\n    for i, row in tqdm(df.iterrows(), total = len(df)):\n        if row['file_path'] in empty_path:\n            width  = 5472/(3648/1280)\n            height = 1280\n        else:\n            width  = 5472\n            height = 3648\n            \n        if row['file_path']==last_file:\n            xmin,ymin,xmax,ymax = row['xmin'],row['ymin'],row['xmax'],row['ymax']\n            x = int(xmin)\n            y = int(ymin)\n            w = int(xmax)-int(xmin)\n            h = int(ymax)-int(ymin)\n            annotations.append({\n                \"id\": annotion_id,\n                \"image_id\": idx,\n                \"category_id\": row['class_id'],\n                \"bbox\": [x,y,w,h],\n                \"area\": w*h,\n                \"segmentation\": [],\n                \"iscrowd\": 0\n            })\n            annotion_id += 1\n        else:\n            idx += 1\n            last_file = row['file_path']\n            images.append({\n                \"id\": idx,\n                \"file_name\": f\"{row['file_path']}\",\n                \"height\": width,\n                \"width\": height,\n            })\n            xmin,ymin,xmax,ymax = row['xmin'],row['ymin'],row['xmax'],row['ymax']\n            x = int(xmin)\n            y = int(ymin)\n            w = int(xmax)-int(xmin)\n            h = int(ymax)-int(ymin)\n            annotations.append({\n                \"id\": annotion_id,\n                \"image_id\": idx,\n                \"category_id\": row['class_id'],\n                \"bbox\": [x,y,w,h],\n                \"area\": w*h,\n                \"segmentation\": [],\n                \"iscrowd\": 0\n            })\n            annotion_id += 1\n            \n\n    json_file = {'categories':categories, 'images':images, 'annotations':annotations}\n    return json_file","metadata":{"execution":{"iopub.status.busy":"2022-04-28T06:34:56.848349Z","iopub.execute_input":"2022-04-28T06:34:56.848604Z","iopub.status.idle":"2022-04-28T06:34:56.864622Z","shell.execute_reply.started":"2022-04-28T06:34:56.848566Z","shell.execute_reply":"2022-04-28T06:34:56.863703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"json_train = coco(df[df['fold']!=0])\njson_valid = coco(df[df['fold']==0])\n# json_df = coco(df)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T06:34:56.865836Z","iopub.execute_input":"2022-04-28T06:34:56.866066Z","iopub.status.idle":"2022-04-28T06:34:58.264141Z","shell.execute_reply.started":"2022-04-28T06:34:56.866039Z","shell.execute_reply":"2022-04-28T06:34:58.263542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir /kaggle/working/data","metadata":{"execution":{"iopub.status.busy":"2022-04-28T06:34:58.265890Z","iopub.execute_input":"2022-04-28T06:34:58.266116Z","iopub.status.idle":"2022-04-28T06:34:59.035803Z","shell.execute_reply.started":"2022-04-28T06:34:58.266086Z","shell.execute_reply":"2022-04-28T06:34:59.034759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\n\nwith open('/kaggle/working/data/train.json', 'w', encoding='utf-8') as f:\n    json.dump(json_train, f, ensure_ascii=True, indent=4)\n    \nwith open('/kaggle/working/data/valid.json', 'w', encoding='utf-8') as f:\n    json.dump(json_valid, f, ensure_ascii=True, indent=4)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T06:34:59.037102Z","iopub.execute_input":"2022-04-28T06:34:59.037483Z","iopub.status.idle":"2022-04-28T06:34:59.335620Z","shell.execute_reply.started":"2022-04-28T06:34:59.037453Z","shell.execute_reply":"2022-04-28T06:34:59.334886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.makedirs('/kaggle/working/data/train2017', exist_ok=True)\nos.makedirs('/kaggle/working/data/val2017', exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T06:34:59.336615Z","iopub.execute_input":"2022-04-28T06:34:59.337408Z","iopub.status.idle":"2022-04-28T06:34:59.342317Z","shell.execute_reply.started":"2022-04-28T06:34:59.337365Z","shell.execute_reply":"2022-04-28T06:34:59.341251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\nlast_file = ''\nfor i, row in tqdm(df.iterrows(), total = len(df)):\n    if row['file_path']!=last_file:\n        last_file = row['file_path']\n        base_dir = 'val2017' if row['fold']==0 else 'train2017'\n        fname = f\"{row['file_path']}\"\n        shutil.copyfile(f\"{TRAIN_PATH}/{fname}\", f\"/kaggle/working/data/{base_dir}/{fname}\")","metadata":{"execution":{"iopub.status.busy":"2022-04-28T06:34:59.343523Z","iopub.execute_input":"2022-04-28T06:34:59.343751Z","iopub.status.idle":"2022-04-28T06:35:38.199948Z","shell.execute_reply.started":"2022-04-28T06:34:59.343721Z","shell.execute_reply":"2022-04-28T06:35:38.198449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sanity check","metadata":{}},{"cell_type":"code","source":"!pip install -Uqqq 'git+https://github.com/cocodataset/cocoapi.git#subdirectory=PythonAPI'","metadata":{"execution":{"iopub.status.busy":"2022-04-28T06:35:38.202967Z","iopub.execute_input":"2022-04-28T06:35:38.203304Z","iopub.status.idle":"2022-04-28T06:36:02.182092Z","shell.execute_reply.started":"2022-04-28T06:35:38.203270Z","shell.execute_reply":"2022-04-28T06:36:02.180503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pycocotools.coco import COCO\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nfrom random import sample","metadata":{"execution":{"iopub.status.busy":"2022-04-28T06:36:02.184295Z","iopub.execute_input":"2022-04-28T06:36:02.184558Z","iopub.status.idle":"2022-04-28T06:36:02.199032Z","shell.execute_reply.started":"2022-04-28T06:36:02.184527Z","shell.execute_reply":"2022-04-28T06:36:02.198345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = '/kaggle/working/data/train2017'\nann_file = '/kaggle/working/data/train.json'\ncoco = COCO(ann_file)\nimg_ids = coco.getImgIds()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T06:36:02.203594Z","iopub.execute_input":"2022-04-28T06:36:02.203898Z","iopub.status.idle":"2022-04-28T06:36:02.268620Z","shell.execute_reply.started":"2022-04-28T06:36:02.203866Z","shell.execute_reply":"2022-04-28T06:36:02.267712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_row = 2\nn_col = 2\nimgs = coco.loadImgs(sample(img_ids, n_row * n_col))\n_, axs = plt.subplots(n_row, n_col, figsize=(12 * n_col, 8 * n_row))\naxs = axs.flatten()\nfor img, ax in zip(imgs, axs):\n    img_img = Image.open(f\"{data_dir}/{img['file_name']}\")\n    anns = coco.loadAnns(coco.getAnnIds(imgIds=[img['id']]))\n    ax.imshow(img_img)\n    plt.sca(ax)\n    coco.showAnns(anns, draw_bbox=True)\n    plt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T06:36:02.270023Z","iopub.execute_input":"2022-04-28T06:36:02.270305Z","iopub.status.idle":"2022-04-28T06:36:04.615697Z","shell.execute_reply.started":"2022-04-28T06:36:02.270260Z","shell.execute_reply":"2022-04-28T06:36:04.613415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shutil.make_archive('data', 'zip', 'data')","metadata":{"execution":{"iopub.status.busy":"2022-04-28T06:36:04.617266Z","iopub.execute_input":"2022-04-28T06:36:04.617534Z","iopub.status.idle":"2022-04-28T06:37:47.006876Z","shell.execute_reply.started":"2022-04-28T06:36:04.617503Z","shell.execute_reply":"2022-04-28T06:37:47.005902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shutil.rmtree('data') ","metadata":{"execution":{"iopub.status.busy":"2022-04-28T06:37:50.654662Z","iopub.execute_input":"2022-04-28T06:37:50.654954Z","iopub.status.idle":"2022-04-28T06:37:51.032543Z","shell.execute_reply.started":"2022-04-28T06:37:50.654926Z","shell.execute_reply":"2022-04-28T06:37:51.031558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}