{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%reload_ext autoreload\n%autoreload 2\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:44:54.539730Z","iopub.execute_input":"2022-04-21T21:44:54.540065Z","iopub.status.idle":"2022-04-21T21:44:54.596220Z","shell.execute_reply.started":"2022-04-21T21:44:54.539981Z","shell.execute_reply":"2022-04-21T21:44:54.595516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#use fastai V2\nfrom fastai.vision.all import *","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:44:56.508729Z","iopub.execute_input":"2022-04-21T21:44:56.508981Z","iopub.status.idle":"2022-04-21T21:44:59.084757Z","shell.execute_reply.started":"2022-04-21T21:44:56.508953Z","shell.execute_reply":"2022-04-21T21:44:59.084043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lbl_csv = '../input/airbus-ship-detection/train_ship_segmentations_v2.csv'\nsegmentation_df = pd.read_csv(lbl_csv)\nsegmentation_df = segmentation_df[segmentation_df['EncodedPixels'].notnull()]\nsegmentation_df","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:45:05.041139Z","iopub.execute_input":"2022-04-21T21:45:05.041446Z","iopub.status.idle":"2022-04-21T21:45:06.120077Z","shell.execute_reply.started":"2022-04-21T21:45:05.041402Z","shell.execute_reply":"2022-04-21T21:45:06.119413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#exclude_list = ['6384c3e78.jpg','13703f040.jpg', '14715c06d.jpg',  '33e0ff2d5.jpg',\n#                '4d4e09f2a.jpg', '877691df8.jpg', '8b909bb20.jpg', 'a8d99130e.jpg', \n#                'ad55c3143.jpg', 'c8260c541.jpg', 'd6c7f17c7.jpg', 'dc3e7c901.jpg',\n#                'e44dffe88.jpg', 'ef87bad36.jpg', 'f083256d8.jpg'] #corrupted images","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:45:20.750720Z","iopub.execute_input":"2022-04-21T21:45:20.750983Z","iopub.status.idle":"2022-04-21T21:45:20.789656Z","shell.execute_reply.started":"2022-04-21T21:45:20.750954Z","shell.execute_reply":"2022-04-21T21:45:20.788883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#添加一列，保存位置\nsegmentation_df['filepath']= '../input/airbus-ship-detection/train_v2/'+segmentation_df['ImageId']\n#segmentation_df = segmentation_df[:100]#小数据测试\nsegmentation_df","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:45:49.601194Z","iopub.execute_input":"2022-04-21T21:45:49.602047Z","iopub.status.idle":"2022-04-21T21:45:49.667885Z","shell.execute_reply.started":"2022-04-21T21:45:49.602008Z","shell.execute_reply":"2022-04-21T21:45:49.667185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#创建语义分割对\nos.mkdir('train')\nos.mkdir('train/images')\nos.mkdir('train/labels')","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:45:51.187237Z","iopub.execute_input":"2022-04-21T21:45:51.187664Z","iopub.status.idle":"2022-04-21T21:45:51.230319Z","shell.execute_reply.started":"2022-04-21T21:45:51.187629Z","shell.execute_reply":"2022-04-21T21:45:51.229459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 将rle格式进行解码为图片\ndef rle_decode(mask_rle, shape=(768, 768)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return\n    Returns numpy array, 1 - mask, 0 - background\n \n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape, order='F')","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:45:53.782560Z","iopub.execute_input":"2022-04-21T21:45:53.782954Z","iopub.status.idle":"2022-04-21T21:45:53.827873Z","shell.execute_reply.started":"2022-04-21T21:45:53.782902Z","shell.execute_reply":"2022-04-21T21:45:53.827137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#数据准备消耗时间约2000s\nimport cv2\nfor index in range(len(segmentation_df)):\n    ImagePath = segmentation_df.iloc[index]['filepath']\n    image = cv2.imread(ImagePath, cv2.IMREAD_UNCHANGED)\n    mask = rle_decode(segmentation_df.iloc[index]['EncodedPixels'])\n    #写入指定地方\n    imgPath = \"./train/images/\"+segmentation_df.iloc[index]['ImageId']\n    mskPath = \"./train/labels/\"+segmentation_df.iloc[index]['ImageId']+\".png\"\n    cv2.imwrite(imgPath,image)\n    cv2.imwrite(mskPath,mask)","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:45:56.870428Z","iopub.execute_input":"2022-04-21T21:45:56.871085Z","iopub.status.idle":"2022-04-21T21:45:59.850316Z","shell.execute_reply.started":"2022-04-21T21:45:56.871044Z","shell.execute_reply":"2022-04-21T21:45:59.848959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import  os\n#import  zipfile\n#startdir = \"./train\"  #要压缩的文件夹路径，这里选择将input中的所有文件压缩\n#file_news = './' +'train.zip' # 压缩后文件夹的名字，这里压缩到kaggle之中的output文件之中，名称为result.zip\n#z = zipfile.ZipFile(file_news,'w',zipfile.ZIP_DEFLATED) #参数一：文件夹名\n#for dirpath, dirnames, filenames in os.walk(startdir):\n#    fpath = dirpath.replace(startdir,'') #这一句很重要，不replace的话，就从根目录开始复制\n#    fpath = fpath and fpath + os.sep or ''#实现当前文件夹以及包含的所有文件的压缩\n#    for filename in filenames:\n#        z.write(os.path.join(dirpath, filename),fpath+filename)\n#z.close()\n#print ('压缩成功')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fnames = get_image_files('./train/images/')\nlbl_names = get_image_files('./train/labels')","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:46:05.468893Z","iopub.execute_input":"2022-04-21T21:46:05.469138Z","iopub.status.idle":"2022-04-21T21:46:05.509528Z","shell.execute_reply.started":"2022-04-21T21:46:05.469110Z","shell.execute_reply":"2022-04-21T21:46:05.508782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print (fnames[0],lbl_names[0])#查看train和mask文件细节\nget_mask = lambda o:'./train/labels/'+str(o.stem)+'.jpg.png' #路径变化 后缀名变化","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:46:06.629431Z","iopub.execute_input":"2022-04-21T21:46:06.629686Z","iopub.status.idle":"2022-04-21T21:46:06.672154Z","shell.execute_reply.started":"2022-04-21T21:46:06.629659Z","shell.execute_reply":"2022-04-21T21:46:06.670659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#显示图片对\nimg_fn = fnames[random.randint(0,len(fnames))]\nim = PILImage.create(img_fn)\nim.show(figsize=(5,5))\nprint(im.shape)","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:46:08.530023Z","iopub.execute_input":"2022-04-21T21:46:08.530285Z","iopub.status.idle":"2022-04-21T21:46:08.846481Z","shell.execute_reply.started":"2022-04-21T21:46:08.530256Z","shell.execute_reply":"2022-04-21T21:46:08.841311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#打印配对的mask\nmask_fn = get_mask(img_fn)\nmsk = PILMask.create(mask_fn)\nmsk.show(figsize=(5,5), alpha=1)\nmsk.shape","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:46:10.198887Z","iopub.execute_input":"2022-04-21T21:46:10.199155Z","iopub.status.idle":"2022-04-21T21:46:10.429519Z","shell.execute_reply.started":"2022-04-21T21:46:10.199124Z","shell.execute_reply":"2022-04-21T21:46:10.428835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#生成DataBlock\nbinary = DataBlock(blocks=(ImageBlock, MaskBlock( ['Background', 'boat'])),    \n                   get_items=get_image_files,    #x的获取方法为get_image_files\n                   splitter=RandomSplitter(),    #随机分割\n                   get_y=get_mask,              #y的获取方法 尝试get_mask\n                   item_tfms=Resize(128,ResizeMethod.Squish),       \n                   batch_tfms=[Normalize.from_stats(*imagenet_stats)])  ","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:46:12.351830Z","iopub.execute_input":"2022-04-21T21:46:12.352099Z","iopub.status.idle":"2022-04-21T21:46:15.041585Z","shell.execute_reply.started":"2022-04-21T21:46:12.352070Z","shell.execute_reply":"2022-04-21T21:46:15.040856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#读取图片,并显示样例\ndls = binary.dataloaders('./train/images',bs=9)\ndls.show_batch( vmin=0, vmax=1)","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:46:28.102417Z","iopub.execute_input":"2022-04-21T21:46:28.103086Z","iopub.status.idle":"2022-04-21T21:46:29.044798Z","shell.execute_reply.started":"2022-04-21T21:46:28.103049Z","shell.execute_reply":"2022-04-21T21:46:29.044179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn = unet_learner(dls,resnet34,metrics=Dice).to_fp16()","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:46:32.909435Z","iopub.execute_input":"2022-04-21T21:46:32.910298Z","iopub.status.idle":"2022-04-21T21:46:36.594905Z","shell.execute_reply.started":"2022-04-21T21:46:32.910252Z","shell.execute_reply":"2022-04-21T21:46:36.594076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fit_flat_cos(10)\nlearn.recorder.plot_loss()","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:46:36.596401Z","iopub.execute_input":"2022-04-21T21:46:36.596664Z","iopub.status.idle":"2022-04-21T21:47:01.223052Z","shell.execute_reply.started":"2022-04-21T21:46:36.596624Z","shell.execute_reply":"2022-04-21T21:47:01.222216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.save('model1')","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:47:17.935628Z","iopub.execute_input":"2022-04-21T21:47:17.936341Z","iopub.status.idle":"2022-04-21T21:47:18.719014Z","shell.execute_reply.started":"2022-04-21T21:47:17.936294Z","shell.execute_reply":"2022-04-21T21:47:18.718310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lbl_names = get_image_files('./train/labels')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#读取sample_submission\nsubmission = pd.read_csv('../input/airbus-ship-detection/sample_submission_v2.csv')\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:51:36.960245Z","iopub.execute_input":"2022-04-21T21:51:36.960939Z","iopub.status.idle":"2022-04-21T21:51:37.024382Z","shell.execute_reply.started":"2022-04-21T21:51:36.960901Z","shell.execute_reply":"2022-04-21T21:51:37.023713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#验证test目录和csv数据一致\n#for idx,name in (enumerate(submission['ImageId'].iloc[:])):\n#    name =  '../input/airbus-ship-detection/test_v2/'+ str(name)\n#    if(not(os.path.exists(name))):\n#        print (idx,name)\n#print('done')","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:52:55.351715Z","iopub.execute_input":"2022-04-21T21:52:55.352616Z","iopub.status.idle":"2022-04-21T21:53:42.251773Z","shell.execute_reply.started":"2022-04-21T21:52:55.352570Z","shell.execute_reply":"2022-04-21T21:53:42.251026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_csv = '../input/airbus-ship-detection/test_v2/'+submission['ImageId']\ntest_csv","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:54:08.290866Z","iopub.execute_input":"2022-04-21T21:54:08.291152Z","iopub.status.idle":"2022-04-21T21:54:08.338792Z","shell.execute_reply.started":"2022-04-21T21:54:08.291120Z","shell.execute_reply":"2022-04-21T21:54:08.338046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_encode(mask):\n    pixels = mask.flatten()\n    pixels[0] = 0\n    pixels[-1] = 0\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 2\n    runs[1::2] = runs[1::2] - runs[:-1:2]\n    return runs","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:54:23.430123Z","iopub.execute_input":"2022-04-21T21:54:23.430395Z","iopub.status.idle":"2022-04-21T21:54:23.469249Z","shell.execute_reply.started":"2022-04-21T21:54:23.430359Z","shell.execute_reply":"2022-04-21T21:54:23.468484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#输出结果\nimport time\nfrom datetime import datetime\nimport pytz\n\nfor ibatch in range(156):\n    test_csv_part = test_csv[100*ibatch:100*(ibatch+1)]\n    test_dl = learn.dls.test_dl(test_csv_part)\n    preds=[]\n    preds = learn.get_preds(dl=test_dl)\n    print(ibatch,time.time())\n    for idx in range(100):\n        submit_np = np.array(preds[0][idx][0]<0.5).astype(np.uint8) \n        msk = PILMask.create(submit_np)\n        msk = msk.resize((768,768),Image.ANTIALIAS)  #中间多倒腾了一次，是因为观察到np的resize导致图像变形\n        submit_np2 = np.array(msk)\n        rle = rle_encode(submit_np2)\n        rle = ' '.join(str(x) for x in rle)\n        submission['EncodedPixels'][100*ibatch+idx]=rle","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:56:07.617678Z","iopub.execute_input":"2022-04-21T21:56:07.617971Z","iopub.status.idle":"2022-04-21T21:56:10.589250Z","shell.execute_reply.started":"2022-04-21T21:56:07.617939Z","shell.execute_reply":"2022-04-21T21:56:10.588474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#tz = pytz.timezone('Asia/Shanghai') #东八区\ncsv_str = 'submission.csv'\nsubmission.to_csv(csv_str,index=False, header=True)","metadata":{"execution":{"iopub.status.busy":"2022-04-21T21:56:32.210895Z","iopub.execute_input":"2022-04-21T21:56:32.211150Z","iopub.status.idle":"2022-04-21T21:56:32.283049Z","shell.execute_reply.started":"2022-04-21T21:56:32.211121Z","shell.execute_reply":"2022-04-21T21:56:32.282396Z"},"trusted":true},"execution_count":null,"outputs":[]}]}