{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Version\n* `v13`: Fold4\n* `v12`: Fold3\n* `v10`: Fold2\n* `v09`: Fold1\n* `v03`: Fold0","metadata":{}},{"cell_type":"code","source":"!pip install --upgrade seaborn","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"papermill":{"duration":9.633907,"end_time":"2021-01-01T09:44:53.448657","exception":false,"start_time":"2021-01-01T09:44:43.81475","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-19T13:02:11.663602Z","iopub.execute_input":"2021-08-19T13:02:11.664096Z","iopub.status.idle":"2021-08-19T13:02:21.199739Z","shell.execute_reply.started":"2021-08-19T13:02:11.664033Z","shell.execute_reply":"2021-08-19T13:02:21.198303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np, pandas as pd\nfrom glob import glob\nimport shutil, os\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import GroupKFold\nfrom tqdm.notebook import tqdm\nimport seaborn as sns","metadata":{"papermill":{"duration":0.926929,"end_time":"2021-01-01T09:44:54.403588","exception":false,"start_time":"2021-01-01T09:44:53.476659","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-19T13:02:21.201971Z","iopub.execute_input":"2021-08-19T13:02:21.202528Z","iopub.status.idle":"2021-08-19T13:02:21.675763Z","shell.execute_reply.started":"2021-08-19T13:02:21.202473Z","shell.execute_reply":"2021-08-19T13:02:21.674675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dim = 512 #512, 256, 'original'\nfold = 4","metadata":{"execution":{"iopub.status.busy":"2021-08-19T13:02:21.678995Z","iopub.execute_input":"2021-08-19T13:02:21.679623Z","iopub.status.idle":"2021-08-19T13:02:21.688180Z","shell.execute_reply.started":"2021-08-19T13:02:21.679577Z","shell.execute_reply":"2021-08-19T13:02:21.687173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(f'../input/vinbigdata-{dim}-image-dataset/vinbigdata/train.csv')\ntrain_df.head()","metadata":{"papermill":{"duration":0.262045,"end_time":"2021-01-01T09:44:54.691965","exception":false,"start_time":"2021-01-01T09:44:54.42992","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-19T13:02:21.693887Z","iopub.execute_input":"2021-08-19T13:02:21.694401Z","iopub.status.idle":"2021-08-19T13:02:21.882387Z","shell.execute_reply.started":"2021-08-19T13:02:21.694354Z","shell.execute_reply":"2021-08-19T13:02:21.881152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['image_path'] = f'/kaggle/input/vinbigdata-{dim}-image-dataset/vinbigdata/train/'+train_df.image_id+('.png' if dim!='original' else '.jpg')\ntrain_df.head()","metadata":{"papermill":{"duration":0.086788,"end_time":"2021-01-01T09:44:54.805857","exception":false,"start_time":"2021-01-01T09:44:54.719069","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-19T13:02:21.885252Z","iopub.execute_input":"2021-08-19T13:02:21.885581Z","iopub.status.idle":"2021-08-19T13:02:21.947573Z","shell.execute_reply.started":"2021-08-19T13:02:21.885549Z","shell.execute_reply":"2021-08-19T13:02:21.946303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Only 14 Class","metadata":{"papermill":{"duration":0.027478,"end_time":"2021-01-01T09:44:54.861374","exception":false,"start_time":"2021-01-01T09:44:54.833896","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_df = train_df[train_df.class_id!=14].reset_index(drop = True)","metadata":{"papermill":{"duration":0.05543,"end_time":"2021-01-01T09:44:54.944088","exception":false,"start_time":"2021-01-01T09:44:54.888658","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-19T13:02:21.951151Z","iopub.execute_input":"2021-08-19T13:02:21.951494Z","iopub.status.idle":"2021-08-19T13:02:21.970384Z","shell.execute_reply.started":"2021-08-19T13:02:21.951462Z","shell.execute_reply":"2021-08-19T13:02:21.969441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pre-Processing","metadata":{"papermill":{"duration":0.027303,"end_time":"2021-01-01T09:44:54.999199","exception":false,"start_time":"2021-01-01T09:44:54.971896","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_df['x_min'] = train_df.apply(lambda row: (row.x_min)/row.width, axis =1)\ntrain_df['y_min'] = train_df.apply(lambda row: (row.y_min)/row.height, axis =1)\n\ntrain_df['x_max'] = train_df.apply(lambda row: (row.x_max)/row.width, axis =1)\ntrain_df['y_max'] = train_df.apply(lambda row: (row.y_max)/row.height, axis =1)\n\ntrain_df['x_mid'] = train_df.apply(lambda row: (row.x_max+row.x_min)/2, axis =1)\ntrain_df['y_mid'] = train_df.apply(lambda row: (row.y_max+row.y_min)/2, axis =1)\n\ntrain_df['w'] = train_df.apply(lambda row: (row.x_max-row.x_min), axis =1)\ntrain_df['h'] = train_df.apply(lambda row: (row.y_max-row.y_min), axis =1)\n\ntrain_df['area'] = train_df['w']*train_df['h']\ntrain_df.head()","metadata":{"papermill":{"duration":7.821668,"end_time":"2021-01-01T09:45:02.854149","exception":false,"start_time":"2021-01-01T09:44:55.032481","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-19T13:02:21.972977Z","iopub.execute_input":"2021-08-19T13:02:21.973800Z","iopub.status.idle":"2021-08-19T13:02:31.550810Z","shell.execute_reply.started":"2021-08-19T13:02:21.973755Z","shell.execute_reply":"2021-08-19T13:02:31.549660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = ['x_min', 'y_min', 'x_max', 'y_max', 'x_mid', 'y_mid', 'w', 'h', 'area']\nX = train_df[features]\ny = train_df['class_id']\nX.shape, y.shape","metadata":{"papermill":{"duration":0.040387,"end_time":"2021-01-01T09:45:02.923416","exception":false,"start_time":"2021-01-01T09:45:02.883029","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-19T13:02:31.552410Z","iopub.execute_input":"2021-08-19T13:02:31.552822Z","iopub.status.idle":"2021-08-19T13:02:31.566844Z","shell.execute_reply.started":"2021-08-19T13:02:31.552776Z","shell.execute_reply":"2021-08-19T13:02:31.565395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_ids, class_names = list(zip(*set(zip(train_df.class_id, train_df.class_name))))\nclasses = list(np.array(class_names)[np.argsort(class_ids)])\nclasses = list(map(lambda x: str(x), classes))\nclasses","metadata":{"papermill":{"duration":0.050418,"end_time":"2021-01-01T09:45:03.002944","exception":false,"start_time":"2021-01-01T09:45:02.952526","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-19T13:02:31.569404Z","iopub.execute_input":"2021-08-19T13:02:31.569970Z","iopub.status.idle":"2021-08-19T13:02:31.599239Z","shell.execute_reply.started":"2021-08-19T13:02:31.569911Z","shell.execute_reply":"2021-08-19T13:02:31.598234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# t-SNE Visualization","metadata":{"papermill":{"duration":0.028555,"end_time":"2021-01-01T09:45:03.060819","exception":false,"start_time":"2021-01-01T09:45:03.032264","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time\nfrom sklearn.manifold import TSNE\n\ntsne = TSNE(n_components = 2, perplexity = 40, random_state=1, n_iter=250) #OG: 5000\ndata_X = X\ndata_y = y.loc[data_X.index]\nembs = tsne.fit_transform(data_X)\n# Add to dataframe for convenience\nplot_x = embs[:, 0]\nplot_y = embs[:, 1]","metadata":{"_kg_hide-output":true,"papermill":{"duration":79.550362,"end_time":"2021-01-01T09:46:22.640073","exception":false,"start_time":"2021-01-01T09:45:03.089711","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-19T13:02:31.601478Z","iopub.execute_input":"2021-08-19T13:02:31.602686Z","iopub.status.idle":"2021-08-19T13:04:27.618889Z","shell.execute_reply.started":"2021-08-19T13:02:31.602639Z","shell.execute_reply":"2021-08-19T13:04:27.617751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(figsize = (15, 15))\nplt.axis('off')\nscatter = plt.scatter(plot_x, plot_y, marker = 'o',s = 50, c=data_y.tolist(), alpha= 0.5,cmap='viridis')\nplt.legend(handles=scatter.legend_elements()[0], labels=classes)","metadata":{"papermill":{"duration":0.674101,"end_time":"2021-01-01T09:46:23.353693","exception":false,"start_time":"2021-01-01T09:46:22.679592","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-19T13:04:27.623409Z","iopub.execute_input":"2021-08-19T13:04:27.625847Z","iopub.status.idle":"2021-08-19T13:04:28.970688Z","shell.execute_reply.started":"2021-08-19T13:04:27.625806Z","shell.execute_reply":"2021-08-19T13:04:28.969489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# BBox Location","metadata":{"papermill":{"duration":0.043674,"end_time":"2021-01-01T09:46:23.441245","exception":false,"start_time":"2021-01-01T09:46:23.397571","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## x_mid Vs y_mid","metadata":{"papermill":{"duration":0.041604,"end_time":"2021-01-01T09:46:23.525588","exception":false,"start_time":"2021-01-01T09:46:23.483984","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from scipy.stats import gaussian_kde\n\n\nx_val = train_df.x_mid.values\ny_val = train_df.y_mid.values\n\n# Calculate the point density\nxy = np.vstack([x_val,y_val])\nz = gaussian_kde(xy)(xy)\n\nfig, ax = plt.subplots(figsize = (10, 10))\nax.axis('off')\nax.scatter(x_val, y_val, c=z, s=100, cmap='viridis')\n# ax.set_xlabel('x_mid')\n# ax.set_ylabel('y_mid')\nplt.show()","metadata":{"papermill":{"duration":31.402628,"end_time":"2021-01-01T09:46:54.970613","exception":false,"start_time":"2021-01-01T09:46:23.567985","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-19T13:04:28.972124Z","iopub.execute_input":"2021-08-19T13:04:28.972520Z","iopub.status.idle":"2021-08-19T13:05:07.997677Z","shell.execute_reply.started":"2021-08-19T13:04:28.972479Z","shell.execute_reply":"2021-08-19T13:05:07.996410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## bbox_w Vs bbox_h","metadata":{"papermill":{"duration":0.046802,"end_time":"2021-01-01T09:46:55.064644","exception":false,"start_time":"2021-01-01T09:46:55.017842","status":"completed"},"tags":[]}},{"cell_type":"code","source":"x_val = train_df.w.values\ny_val = train_df.h.values\n\n# Calculate the point density\nxy = np.vstack([x_val,y_val])\nz = gaussian_kde(xy)(xy)\n\nfig, ax = plt.subplots(figsize = (10, 10))\nax.axis('off')\nax.scatter(x_val, y_val, c=z, s=100, cmap='viridis')\n# ax.set_xlabel('bbox_width')\n# ax.set_ylabel('bbox_height')\nplt.show()","metadata":{"papermill":{"duration":30.485279,"end_time":"2021-01-01T09:47:25.596636","exception":false,"start_time":"2021-01-01T09:46:55.111357","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-19T13:05:07.999930Z","iopub.execute_input":"2021-08-19T13:05:08.000454Z","iopub.status.idle":"2021-08-19T13:05:47.634026Z","shell.execute_reply.started":"2021-08-19T13:05:08.000403Z","shell.execute_reply":"2021-08-19T13:05:47.632893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image Aspect Ratio","metadata":{"papermill":{"duration":0.090619,"end_time":"2021-01-01T09:47:25.783368","exception":false,"start_time":"2021-01-01T09:47:25.692749","status":"completed"},"tags":[]}},{"cell_type":"code","source":"x_val = train_df.width.values\ny_val = train_df.height.values\n\n# Calculate the point density\nxy = np.vstack([x_val,y_val])\nz = gaussian_kde(xy)(xy)\n\nfig, ax = plt.subplots(figsize = (10, 10))\nax.axis('off')\nax.scatter(x_val, y_val, c=z, s=100, cmap='viridis')\n# ax.set_xlabel('image_width')\n# ax.set_ylabel('image_height')\nplt.show()","metadata":{"papermill":{"duration":30.131145,"end_time":"2021-01-01T09:47:56.005954","exception":false,"start_time":"2021-01-01T09:47:25.874809","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-19T13:05:47.635661Z","iopub.execute_input":"2021-08-19T13:05:47.636057Z","iopub.status.idle":"2021-08-19T13:06:26.761693Z","shell.execute_reply.started":"2021-08-19T13:05:47.636020Z","shell.execute_reply":"2021-08-19T13:06:26.760217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split","metadata":{"papermill":{"duration":0.052492,"end_time":"2021-01-01T09:47:56.110766","exception":false,"start_time":"2021-01-01T09:47:56.058274","status":"completed"},"tags":[]}},{"cell_type":"code","source":"gkf  = GroupKFold(n_splits = 5)\ntrain_df['fold'] = -1\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(train_df, groups = train_df.image_id.tolist())):\n    train_df.loc[val_idx, 'fold'] = fold\ntrain_df.head()","metadata":{"papermill":{"duration":0.134603,"end_time":"2021-01-01T09:47:56.297774","exception":false,"start_time":"2021-01-01T09:47:56.163171","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-19T13:06:26.763679Z","iopub.execute_input":"2021-08-19T13:06:26.764370Z","iopub.status.idle":"2021-08-19T13:06:26.851517Z","shell.execute_reply.started":"2021-08-19T13:06:26.764300Z","shell.execute_reply":"2021-08-19T13:06:26.850187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_files = []\nval_files   = []\nval_files += list(train_df[train_df.fold==fold].image_path.unique())\ntrain_files += list(train_df[train_df.fold!=fold].image_path.unique())\nlen(train_files), len(val_files)","metadata":{"papermill":{"duration":0.086817,"end_time":"2021-01-01T09:47:56.443789","exception":false,"start_time":"2021-01-01T09:47:56.356972","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-19T13:06:26.853838Z","iopub.execute_input":"2021-08-19T13:06:26.854353Z","iopub.status.idle":"2021-08-19T13:06:26.886342Z","shell.execute_reply.started":"2021-08-19T13:06:26.854304Z","shell.execute_reply":"2021-08-19T13:06:26.885462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Copying Files","metadata":{"papermill":{"duration":0.083752,"end_time":"2021-01-01T09:47:56.584924","exception":false,"start_time":"2021-01-01T09:47:56.501172","status":"completed"},"tags":[]}},{"cell_type":"code","source":"os.makedirs('/kaggle/working/vinbigdata/labels/train', exist_ok = True)\nos.makedirs('/kaggle/working/vinbigdata/labels/val', exist_ok = True)\nos.makedirs('/kaggle/working/vinbigdata/images/train', exist_ok = True)\nos.makedirs('/kaggle/working/vinbigdata/images/val', exist_ok = True)\nlabel_dir = '/kaggle/input/vinbigdata-yolo-labels-dataset/labels'\nfor file in tqdm(train_files):\n    shutil.copy(file, '/kaggle/working/vinbigdata/images/train')\n    filename = file.split('/')[-1].split('.')[0]\n    shutil.copy(os.path.join(label_dir, filename+'.txt'), '/kaggle/working/vinbigdata/labels/train')\n    \nfor file in tqdm(val_files):\n    shutil.copy(file, '/kaggle/working/vinbigdata/images/val')\n    filename = file.split('/')[-1].split('.')[0]\n    shutil.copy(os.path.join(label_dir, filename+'.txt'), '/kaggle/working/vinbigdata/labels/val')","metadata":{"papermill":{"duration":124.654777,"end_time":"2021-01-01T09:50:01.331041","exception":false,"start_time":"2021-01-01T09:47:56.676264","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-19T13:06:26.888051Z","iopub.execute_input":"2021-08-19T13:06:26.888501Z","iopub.status.idle":"2021-08-19T13:07:04.369563Z","shell.execute_reply.started":"2021-08-19T13:06:26.888457Z","shell.execute_reply":"2021-08-19T13:07:04.368632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get Class Name","metadata":{"papermill":{"duration":0.068822,"end_time":"2021-01-01T09:50:01.458337","exception":false,"start_time":"2021-01-01T09:50:01.389515","status":"completed"},"tags":[]}},{"cell_type":"code","source":"class_ids, class_names = list(zip(*set(zip(train_df.class_id, train_df.class_name))))\nclasses = list(np.array(class_names)[np.argsort(class_ids)])\nclasses = list(map(lambda x: str(x), classes))\nclasses","metadata":{"papermill":{"duration":0.082234,"end_time":"2021-01-01T09:50:01.601574","exception":false,"start_time":"2021-01-01T09:50:01.51934","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-19T13:07:04.373175Z","iopub.execute_input":"2021-08-19T13:07:04.373470Z","iopub.status.idle":"2021-08-19T13:07:04.399472Z","shell.execute_reply.started":"2021-08-19T13:07:04.373439Z","shell.execute_reply":"2021-08-19T13:07:04.398032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# [YOLOv5](https://github.com/ultralytics/yolov5)\n![](https://user-images.githubusercontent.com/26833433/98699617-a1595a00-2377-11eb-8145-fc674eb9b1a7.jpg)\n![](https://user-images.githubusercontent.com/26833433/90187293-6773ba00-dd6e-11ea-8f90-cd94afc0427f.png)","metadata":{"papermill":{"duration":0.056257,"end_time":"2021-01-01T09:50:01.716608","exception":false,"start_time":"2021-01-01T09:50:01.660351","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# YOLOv5 Stuff","metadata":{"papermill":{"duration":0.055699,"end_time":"2021-01-01T09:50:01.82747","exception":false,"start_time":"2021-01-01T09:50:01.771771","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from os import listdir\nfrom os.path import isfile, join\nimport yaml\n\ncwd = '/kaggle/working/'\n\nwith open(join( cwd , 'train.txt'), 'w') as f:\n    for path in glob('/kaggle/working/vinbigdata/images/train/*'):\n        f.write(path+'\\n')\n            \nwith open(join( cwd , 'val.txt'), 'w') as f:\n    for path in glob('/kaggle/working/vinbigdata/images/val/*'):\n        f.write(path+'\\n')\n\ndata = dict(\n    train =  join( cwd , 'train.txt') ,\n    val   =  join( cwd , 'val.txt' ),\n    nc    = 14,\n    names = classes\n    )\n\nwith open(join( cwd , 'vinbigdata.yaml'), 'w') as outfile:\n    yaml.dump(data, outfile, default_flow_style=False)\n\nf = open(join( cwd , 'vinbigdata.yaml'), 'r')\nprint('\\nyaml:')\nprint(f.read())","metadata":{"papermill":{"duration":0.113001,"end_time":"2021-01-01T09:50:01.996448","exception":false,"start_time":"2021-01-01T09:50:01.883447","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-19T13:07:04.401478Z","iopub.execute_input":"2021-08-19T13:07:04.402473Z","iopub.status.idle":"2021-08-19T13:07:04.478509Z","shell.execute_reply.started":"2021-08-19T13:07:04.402426Z","shell.execute_reply":"2021-08-19T13:07:04.477245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#shutil.rmtree('/kaggle/working/yolov5')","metadata":{"execution":{"iopub.status.busy":"2021-08-19T13:11:30.075371Z","iopub.execute_input":"2021-08-19T13:11:30.076093Z","iopub.status.idle":"2021-08-19T13:11:30.123143Z","shell.execute_reply.started":"2021-08-19T13:11:30.076021Z","shell.execute_reply":"2021-08-19T13:11:30.121654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/ultralytics/yolov5\n# !git clone https://github.com/ultralytics/yolov5  # clone repo\n# %cd yolov5\nshutil.copytree('/kaggle/input/yolov5-official-v31-dataset/yolov5', '/kaggle/working/yolov5')\nos.chdir('/kaggle/working/yolov5')\n# %pip install -qr requirements.txt # install dependencies\n\nimport torch\nfrom IPython.display import Image, clear_output  # to display images\n\nclear_output()\nprint('Setup complete. Using torch %s %s' % (torch.__version__, torch.cuda.get_device_properties(0) if torch.cuda.is_available() else 'CPU'))","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","papermill":{"duration":6.702428,"end_time":"2021-01-01T09:50:08.784153","exception":false,"start_time":"2021-01-01T09:50:02.081725","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-19T13:11:31.992669Z","iopub.execute_input":"2021-08-19T13:11:31.993075Z","iopub.status.idle":"2021-08-19T13:11:34.035726Z","shell.execute_reply.started":"2021-08-19T13:11:31.993023Z","shell.execute_reply":"2021-08-19T13:11:34.034205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python detect.py --weights yolov5s.pt --img 640 --conf 0.25 --source data/images/\nImage(filename='runs/detect/exp/zidane.jpg', width=600)","metadata":{"papermill":{"duration":10.410768,"end_time":"2021-01-01T09:50:19.303402","exception":false,"start_time":"2021-01-01T09:50:08.892634","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Pretrained Checkpoints:\n\n| Model | AP<sup>val</sup> | AP<sup>test</sup> | AP<sub>50</sub> | Speed<sub>GPU</sub> | FPS<sub>GPU</sub> || params | FLOPS |\n|---------- |------ |------ |------ | -------- | ------| ------ |------  |  :------: |\n| [YOLOv5s](https://github.com/ultralytics/yolov5/releases/tag/v3.0)    | 37.0     | 37.0     | 56.2     | **2.4ms** | **416** || 7.5M   | 13.2B\n| [YOLOv5m](https://github.com/ultralytics/yolov5/releases/tag/v3.0)    | 44.3     | 44.3     | 63.2     | 3.4ms     | 294     || 21.8M  | 39.4B\n| [YOLOv5l](https://github.com/ultralytics/yolov5/releases/tag/v3.0)    | 47.7     | 47.7     | 66.5     | 4.4ms     | 227     || 47.8M  | 88.1B\n| [YOLOv5x](https://github.com/ultralytics/yolov5/releases/tag/v3.0)    | **49.2** | **49.2** | **67.7** | 6.9ms     | 145     || 89.0M  | 166.4B\n| | | | | | || |\n| [YOLOv5x](https://github.com/ultralytics/yolov5/releases/tag/v3.0) + TTA|**50.8**| **50.8** | **68.9** | 25.5ms    | 39      || 89.0M  | 354.3B\n| | | | | | || |\n| [YOLOv3-SPP](https://github.com/ultralytics/yolov5/releases/tag/v3.0) | 45.6     | 45.5     | 65.2     | 4.5ms     | 222     || 63.0M  | 118.0B","metadata":{"papermill":{"duration":0.064911,"end_time":"2021-01-01T09:50:19.435746","exception":false,"start_time":"2021-01-01T09:50:19.370835","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Selecting Models\nIn this notebok I'm using `v5s`. To select your prefered model just replace `--cfg models/yolov5s.yaml --weights yolov5s.pt` with the following command:\n* `v5s` : `--cfg models/yolov5s.yaml --weights yolov5s.pt`\n* `v5m` : `--cfg models/yolov5m.yaml --weights yolov5m.pt`\n* `v5l` : `--cfg models/yolov5l.yaml --weights yolov5l.pt`\n* `v5x` : `--cfg models/yolov5x.yaml --weights yolov5x.pt`","metadata":{"papermill":{"duration":0.064016,"end_time":"2021-01-01T09:50:19.564859","exception":false,"start_time":"2021-01-01T09:50:19.500843","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Train","metadata":{"papermill":{"duration":0.064553,"end_time":"2021-01-01T09:50:19.6938","exception":false,"start_time":"2021-01-01T09:50:19.629247","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# !WANDB_MODE=\"dryrun\" python train.py --img 640 --batch 16 --epochs 3 --data coco128.yaml --weights yolov5s.pt --nosave --cache \n!WANDB_MODE=\"dryrun\" python train.py --img 640 --batch 16 --epochs 1 --data /kaggle/working/vinbigdata.yaml --weights yolov5x.pt --cache","metadata":{"papermill":{"duration":19916.498298,"end_time":"2021-01-01T15:22:16.289734","exception":false,"start_time":"2021-01-01T09:50:19.791436","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-19T13:11:49.718969Z","iopub.execute_input":"2021-08-19T13:11:49.719399Z","iopub.status.idle":"2021-08-19T13:18:51.577240Z","shell.execute_reply.started":"2021-08-19T13:11:49.719349Z","shell.execute_reply":"2021-08-19T13:18:51.575826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Class Distribution","metadata":{"papermill":{"duration":4.919442,"end_time":"2021-01-01T15:22:26.398681","exception":false,"start_time":"2021-01-01T15:22:21.479239","status":"completed"},"tags":[]}},{"cell_type":"code","source":"plt.figure(figsize = (20,20))\nplt.axis('off')\nplt.imshow(plt.imread('runs/train/exp/labels_correlogram.jpg'));","metadata":{"papermill":{"duration":6.511035,"end_time":"2021-01-01T15:22:37.753063","exception":false,"start_time":"2021-01-01T15:22:31.242028","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (20,20))\nplt.axis('off')\nplt.imshow(plt.imread('runs/train/exp/labels.jpg'));","metadata":{"papermill":{"duration":5.977042,"end_time":"2021-01-01T15:22:48.614609","exception":false,"start_time":"2021-01-01T15:22:42.637567","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Batch Image","metadata":{"papermill":{"duration":5.378338,"end_time":"2021-01-01T15:22:59.482837","exception":false,"start_time":"2021-01-01T15:22:54.104499","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(figsize = (15, 15))\nplt.imshow(plt.imread('runs/train/exp/train_batch0.jpg'))\n\nplt.figure(figsize = (15, 15))\nplt.imshow(plt.imread('runs/train/exp/train_batch1.jpg'))\n\nplt.figure(figsize = (15, 15))\nplt.imshow(plt.imread('runs/train/exp/train_batch2.jpg'))","metadata":{"papermill":{"duration":7.317416,"end_time":"2021-01-01T15:23:11.777544","exception":false,"start_time":"2021-01-01T15:23:04.460128","status":"completed"},"tags":[],"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GT Vs Pred","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(3, 2, figsize = (2*5,3*5), constrained_layout = True)\nfor row in range(3):\n    ax[row][0].imshow(plt.imread(f'runs/train/exp/test_batch{row}_labels.jpg'))\n    ax[row][0].set_xticks([])\n    ax[row][0].set_yticks([])\n    ax[row][0].set_title(f'runs/train/exp/test_batch{row}_labels.jpg', fontsize = 12)\n    \n    ax[row][1].imshow(plt.imread(f'runs/train/exp/test_batch{row}_pred.jpg'))\n    ax[row][1].set_xticks([])\n    ax[row][1].set_yticks([])\n    ax[row][1].set_title(f'runs/train/exp/test_batch{row}_pred.jpg', fontsize = 12)","metadata":{"papermill":{"duration":6.453975,"end_time":"2021-01-01T15:23:23.514717","exception":false,"start_time":"2021-01-01T15:23:17.060742","status":"completed"},"tags":[],"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# (Loss, Map) Vs Epoch","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(30,15))\nplt.axis('off')\nplt.imshow(plt.imread('runs/train/exp/results.png'));","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Confusion Matrix","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(30,15))\nplt.axis('off')\nplt.imshow(plt.imread('runs/train/exp/confusion_matrix.png'));","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transform COVID Data into 256x256","metadata":{"papermill":{"duration":4.941983,"end_time":"2021-01-01T15:23:33.765831","exception":false,"start_time":"2021-01-01T15:23:28.823848","status":"completed"},"tags":[]}},{"cell_type":"code","source":"#pip install pycocotools\n#pip install pylibjpeg pylibjpeg-libjpeg pylibjpeg-openjpeg\n#Run -> Restart and Clear Cell Outputs\n#pip uninstall -y numpy\n#pip uninstall -y numpy (again)\n#pip install numpy\n#Run -> Restart and Clear Cell Outputs\n#May need to uninstall and reinstall numpy several times, inconsistent","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport os\nimport numpy as np\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm.auto import tqdm","metadata":{"execution":{"iopub.status.busy":"2021-08-19T12:26:24.998411Z","iopub.execute_input":"2021-08-19T12:26:24.998875Z","iopub.status.idle":"2021-08-19T12:26:25.206862Z","shell.execute_reply.started":"2021-08-19T12:26:24.998834Z","shell.execute_reply":"2021-08-19T12:26:25.205582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_xray(path, voi_lut = True, fix_monochrome = True):\n    #Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    \n        \n    return data\n\ndef resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2021-08-19T12:26:27.525014Z","iopub.execute_input":"2021-08-19T12:26:27.525440Z","iopub.status.idle":"2021-08-19T12:26:27.537277Z","shell.execute_reply.started":"2021-08-19T12:26:27.525407Z","shell.execute_reply":"2021-08-19T12:26:27.535607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shutil.rmtree('/kaggle/working/siim-covid19')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#for split in ['train', 'test']:\nfor split in ['test']:\n    save_dir = f'/kaggle/working/siim-covid19/{split}/'\n\n    os.makedirs(save_dir, exist_ok=True)\n\n    save_dir = f'/kaggle/working/siim-covid19/{split}/study/'\n    os.makedirs(save_dir, exist_ok=True)\n\n    for dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            # set keep_ratio=True to have original aspect ratio\n            xray = read_xray(os.path.join(dirname, file))\n            im = resize(xray, size=1000)  \n            study = dirname.split('/')[-2] + '_study.png'\n            im.save(os.path.join(save_dir, study))","metadata":{"execution":{"iopub.status.busy":"2021-08-19T12:48:34.000924Z","iopub.execute_input":"2021-08-19T12:48:34.001311Z","iopub.status.idle":"2021-08-19T13:02:11.660879Z","shell.execute_reply.started":"2021-08-19T12:48:34.001278Z","shell.execute_reply":"2021-08-19T13:02:11.659574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python detect.py --weights 'runs/train/exp/weights/best.pt'\\\n--img 640\\\n--conf 0.15\\\n--iou 0.5\\\n--save-txt\\\n--source /kaggle/working/siim-covid19/test/study\\\n--exist-ok","metadata":{"_kg_hide-output":true,"papermill":{"duration":10.763143,"end_time":"2021-01-01T15:23:49.800461","exception":false,"start_time":"2021-01-01T15:23:39.037318","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-19T13:18:51.580121Z","iopub.execute_input":"2021-08-19T13:18:51.580611Z","iopub.status.idle":"2021-08-19T13:21:20.031885Z","shell.execute_reply.started":"2021-08-19T13:18:51.580561Z","shell.execute_reply":"2021-08-19T13:21:20.030549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://stackoverflow.com/questions/65381312/how-to-convert-a-yolo-darknet-format-into-csv-file\n\nimport glob\nos.chdir(r'/kaggle/working/yolov5/runs/detect/exp/labels')\nmyFiles = glob.glob('*.txt')\n\nwidth=1000\nheight=1000\nimage_id=0\nfinal_df=[]\nfor item in myFiles:\n    row=[]\n    bbox_temp=[]\n    with open(item, 'rt') as fd:\n        first_line = fd.readline()\n        splited = first_line.split();\n        \n        row.append(image_id)\n        row.append(width)\n        row.append(height)\n        try:\n            bbox_temp.append(float(splited[1])*width)\n            bbox_temp.append(float(splited[2])*height)\n            bbox_temp.append(float(splited[3])*width)\n            bbox_temp.append(float(splited[4])*height)\n            row.append(bbox_temp)\n            final_df.append(row)\n        except:\n            print(\"file is not in YOLO format!\")\ndf = pd.DataFrame(final_df,columns=['image_id', 'width', 'height','bbox'])\ndf.to_csv(\"saved.csv\",index=False)\n\nfrom IPython.display import FileLink\nFileLink(r'saved.csv')","metadata":{"execution":{"iopub.status.busy":"2021-08-19T13:35:52.032741Z","iopub.execute_input":"2021-08-19T13:35:52.033195Z","iopub.status.idle":"2021-08-19T13:35:52.077850Z","shell.execute_reply.started":"2021-08-19T13:35:52.033158Z","shell.execute_reply":"2021-08-19T13:35:52.076574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Link: <a href=\"/kaggle/working/yolov5/runs/detect/exp/labels/saved.csv\"> Download File </a>","metadata":{}},{"cell_type":"markdown","source":"# Cleaning","metadata":{"papermill":{"duration":5.225725,"end_time":"2021-01-01T15:24:00.706026","exception":false,"start_time":"2021-01-01T15:23:55.480301","status":"completed"},"tags":[]}},{"cell_type":"code","source":"shutil.rmtree('/kaggle/working/vinbigdata')\nshutil.rmtree('runs/detect')\nfor file in (glob('runs/train/exp/**/*.png', recursive = True)+glob('runs/train/exp/**/*.jpg', recursive = True)):\n    os.remove(file)","metadata":{"papermill":{"duration":5.709202,"end_time":"2021-01-01T15:24:22.413173","exception":false,"start_time":"2021-01-01T15:24:16.703971","status":"completed"},"tags":[],"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]}]}