{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#subsequence split\n\nimport os\nimport cv2\nimport subprocess\nfrom tqdm.auto import tqdm\nimport pandas as pd\nfrom IPython.display import Video, display, HTML\nimport warnings; warnings.simplefilter(\"ignore\")\n\n\nBASE_PATH = '../input/tensorflow-great-barrier-reef/train_images/'\n\ndf = pd.read_csv(\"/kaggle/input/tensorflow-great-barrier-reef/train.csv\")\ndf['annotations'] = df['annotations'].apply(eval)\ndf['n_annotations'] = df['annotations'].str.len()\ndf['has_annotations'] = df['annotations'].str.len() > 0\ndf['has_2_or_more_annotations'] = df['annotations'].str.len() >= 2\ndf['doesnt_have_annotations'] = df['annotations'].str.len() == 0\ndf['image_path'] = BASE_PATH + \"video_\" + df['video_id'].astype(str) + \"/\" + df['video_frame'].astype(str) + \".jpg\"","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:16:41.699535Z","iopub.execute_input":"2022-04-19T03:16:41.699841Z","iopub.status.idle":"2022-04-19T03:16:42.684408Z","shell.execute_reply.started":"2022-04-19T03:16:41.69976Z","shell.execute_reply":"2022-04-19T03:16:42.683491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['sequence'].unique()\ndf['sequence'].nunique()\ndf.groupby(\"sequence\")['video_id'].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:17:24.488479Z","iopub.execute_input":"2022-04-19T03:17:24.488777Z","iopub.status.idle":"2022-04-19T03:17:24.510348Z","shell.execute_reply.started":"2022-04-19T03:17:24.488732Z","shell.execute_reply":"2022-04-19T03:17:24.509419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_agg = df.groupby([\"video_id\", 'sequence']).agg({'sequence_frame': 'count', 'has_annotations': 'sum', 'doesnt_have_annotations': 'sum'})\\\n           .rename(columns={'sequence_frame': 'Total Frames', 'has_annotations': 'Frames with at least 1 object', 'doesnt_have_annotations': \"Frames with no object\"})\ndf_agg","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:17:27.106018Z","iopub.execute_input":"2022-04-19T03:17:27.10631Z","iopub.status.idle":"2022-04-19T03:17:27.137512Z","shell.execute_reply.started":"2022-04-19T03:17:27.106263Z","shell.execute_reply":"2022-04-19T03:17:27.136467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_agg.sort_values(\"Total Frames\")","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:17:30.285176Z","iopub.execute_input":"2022-04-19T03:17:30.285709Z","iopub.status.idle":"2022-04-19T03:17:30.299815Z","shell.execute_reply.started":"2022-04-19T03:17:30.285675Z","shell.execute_reply":"2022-04-19T03:17:30.298515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_agg.sort_values(\"Frames with at least 1 object\")","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:17:33.164949Z","iopub.execute_input":"2022-04-19T03:17:33.165233Z","iopub.status.idle":"2022-04-19T03:17:33.184271Z","shell.execute_reply.started":"2022-04-19T03:17:33.165205Z","shell.execute_reply":"2022-04-19T03:17:33.182182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image_id is a unique identifier for a row\ndf['image_id'].nunique() == len(df)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:17:35.733267Z","iopub.execute_input":"2022-04-19T03:17:35.733713Z","iopub.status.idle":"2022-04-19T03:17:35.759096Z","shell.execute_reply.started":"2022-04-19T03:17:35.733649Z","shell.execute_reply":"2022-04-19T03:17:35.758242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_agg.loc[[(0, 40258)]]","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:17:40.140442Z","iopub.execute_input":"2022-04-19T03:17:40.140729Z","iopub.status.idle":"2022-04-19T03:17:40.163604Z","shell.execute_reply.started":"2022-04-19T03:17:40.140698Z","shell.execute_reply":"2022-04-19T03:17:40.162462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option(\"display.max_rows\", 500)\ndf[df['sequence'] == 40258]","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:17:41.927371Z","iopub.execute_input":"2022-04-19T03:17:41.927644Z","iopub.status.idle":"2022-04-19T03:17:42.345915Z","shell.execute_reply.started":"2022-04-19T03:17:41.927615Z","shell.execute_reply":"2022-04-19T03:17:42.345065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['start_cut_here'] = df['has_annotations'] & df['doesnt_have_annotations'].shift(1)  & df['doesnt_have_annotations'].shift(2)\ndf['end_cut_here'] = df['doesnt_have_annotations'] & df['has_annotations'].shift(1)  & df['has_annotations'].shift(2)\ndf['sequence_change'] = df['sequence'] != df['sequence'].shift(1)\ndf['last_row'] =  df.index == len(df)-1\ndf['cut_here'] = df['start_cut_here'] | df['end_cut_here'] | df['sequence_change'] | df['last_row']","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:17:51.108587Z","iopub.execute_input":"2022-04-19T03:17:51.108896Z","iopub.status.idle":"2022-04-19T03:17:51.13995Z","shell.execute_reply.started":"2022-04-19T03:17:51.108865Z","shell.execute_reply":"2022-04-19T03:17:51.139058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start_idx = 0\nfor subsequence_id, end_idx in enumerate(df[df['cut_here']].index):\n    df.loc[start_idx:end_idx, 'subsequence_id'] = subsequence_id\n    start_idx = end_idx","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:17:53.66674Z","iopub.execute_input":"2022-04-19T03:17:53.667008Z","iopub.status.idle":"2022-04-19T03:17:53.730488Z","shell.execute_reply.started":"2022-04-19T03:17:53.666977Z","shell.execute_reply":"2022-04-19T03:17:53.729591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['subsequence_id'] = df['subsequence_id'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:17:58.352305Z","iopub.execute_input":"2022-04-19T03:17:58.353128Z","iopub.status.idle":"2022-04-19T03:17:58.3619Z","shell.execute_reply.started":"2022-04-19T03:17:58.353066Z","shell.execute_reply":"2022-04-19T03:17:58.360685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['subsequence_id'].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:18:00.615343Z","iopub.execute_input":"2022-04-19T03:18:00.615633Z","iopub.status.idle":"2022-04-19T03:18:00.625722Z","shell.execute_reply.started":"2022-04-19T03:18:00.615603Z","shell.execute_reply":"2022-04-19T03:18:00.623681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_cols = ['start_cut_here', 'end_cut_here', 'sequence_change', 'last_row', 'cut_here', 'has_2_or_more_annotations', 'doesnt_have_annotations']\ndf = df.drop(drop_cols, axis=1)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:18:02.449274Z","iopub.execute_input":"2022-04-19T03:18:02.449591Z","iopub.status.idle":"2022-04-19T03:18:02.475763Z","shell.execute_reply.started":"2022-04-19T03:18:02.44956Z","shell.execute_reply":"2022-04-19T03:18:02.474844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby(\"subsequence_id\")['has_annotations'].mean().round(2).sort_values().value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:18:05.209667Z","iopub.execute_input":"2022-04-19T03:18:05.209999Z","iopub.status.idle":"2022-04-19T03:18:05.225745Z","shell.execute_reply.started":"2022-04-19T03:18:05.20993Z","shell.execute_reply":"2022-04-19T03:18:05.224799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subseq_agg = df.groupby(\"subsequence_id\")['has_annotations'].mean()\ndf_subseq_agg[~df_subseq_agg.isin([0, 1])]","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:18:07.016263Z","iopub.execute_input":"2022-04-19T03:18:07.017343Z","iopub.status.idle":"2022-04-19T03:18:07.031353Z","shell.execute_reply.started":"2022-04-19T03:18:07.017269Z","shell.execute_reply":"2022-04-19T03:18:07.030314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df['subsequence_id'] == 52]","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:18:09.000026Z","iopub.execute_input":"2022-04-19T03:18:09.000338Z","iopub.status.idle":"2022-04-19T03:18:09.060019Z","shell.execute_reply.started":"2022-04-19T03:18:09.00028Z","shell.execute_reply":"2022-04-19T03:18:09.059108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df['subsequence_id'] == 54]","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:18:12.856151Z","iopub.execute_input":"2022-04-19T03:18:12.8568Z","iopub.status.idle":"2022-04-19T03:18:12.878073Z","shell.execute_reply.started":"2022-04-19T03:18:12.856749Z","shell.execute_reply":"2022-04-19T03:18:12.877057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! mkdir videos/","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:18:15.903274Z","iopub.execute_input":"2022-04-19T03:18:15.903597Z","iopub.status.idle":"2022-04-19T03:18:16.660987Z","shell.execute_reply.started":"2022-04-19T03:18:15.903566Z","shell.execute_reply":"2022-04-19T03:18:16.6598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_image(img_path):\n    assert os.path.exists(img_path), f'{img_path} does not exist.'\n    img = cv2.imread(img_path)\n    return img\n\ndef load_image_with_annotations(img_path, annotations):\n    img = load_image(img_path)\n    if len(annotations) > 0:\n        for ann in annotations:\n            cv2.rectangle(img, (ann['x'], ann['y']),\n                (ann['x'] + ann['width'], ann['y'] + ann['height']),\n                (255, 255, 0), thickness=2,)\n    return img\n\ndef make_video(df, part_id, is_subsequence=False):\n    \"\"\"\n    Args:\n        - part_id: either a sequence or a subsequence id\n    \"\"\"\n    \n    if is_subsequence:\n        part_str = \"subsequence_id\"\n    else:\n        part_str = \"sequence\"\n    \n    print(f\"Creating video for part={part_id}, is_subsequence={is_subsequence} (querying by {part_str})\")\n    # partly borrowed from https://github.com/RobMulla/helmet-assignment/blob/main/helmet_assignment/video.py\n    fps = 15 # don't know exact value\n    width = 1280\n    height = 720\n    save_path = f'videos/video_{part_str}_{part_id}.mp4'\n    tmp_path = f'videos/tmp_video_{part_str}_{part_id}.mp4'\n    \n    \n    output_video = cv2.VideoWriter(tmp_path, cv2.VideoWriter_fourcc(*\"MP4V\"), fps, (width, height))\n    \n    df_part = df.query(f'{part_str} == @part_id')\n    for _, row in tqdm(df_part.iterrows(), total=len(df_part)):\n        img = load_image_with_annotations(row.image_path, row.annotations)\n        output_video.write(img)\n    \n    output_video.release()\n    # Not all browsers support the codec, we will re-load the file at tmp_output_path\n    # and convert to a codec that is more broadly readable using ffmpeg\n    if os.path.exists(save_path):\n        os.remove(save_path)\n    subprocess.run(\n        [\"ffmpeg\", \"-i\", tmp_path, \"-crf\", \"18\", \"-preset\", \"veryfast\", \"-vcodec\", \"libx264\", save_path],\n        stdout=subprocess.DEVNULL,\n        stderr=subprocess.DEVNULL\n    )\n    os.remove(tmp_path)\n    print(f\"Finished creating video for {part_id}... saved as {save_path}\")\n    return save_path","metadata":{"execution":{"iopub.status.busy":"2022-04-19T03:18:20.003447Z","iopub.execute_input":"2022-04-19T03:18:20.004323Z","iopub.status.idle":"2022-04-19T03:18:20.01926Z","shell.execute_reply.started":"2022-04-19T03:18:20.004262Z","shell.execute_reply":"2022-04-19T03:18:20.01834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"video_path = make_video(df, 40258)","metadata":{"execution":{"iopub.status.busy":"2022-04-18T11:21:15.207529Z","iopub.execute_input":"2022-04-18T11:21:15.207878Z","iopub.status.idle":"2022-04-18T11:21:48.218165Z","shell.execute_reply.started":"2022-04-18T11:21:15.207843Z","shell.execute_reply":"2022-04-18T11:21:48.216901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Video(video_path, width= 1280/2, height= 720/2)","metadata":{"execution":{"iopub.status.busy":"2022-04-18T11:21:48.219614Z","iopub.execute_input":"2022-04-18T11:21:48.221034Z","iopub.status.idle":"2022-04-18T11:21:48.227355Z","shell.execute_reply.started":"2022-04-18T11:21:48.221Z","shell.execute_reply":"2022-04-18T11:21:48.226825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subsequences = df.loc[df['sequence'] == 40258, 'subsequence_id'].unique()\nsubsequences","metadata":{"execution":{"iopub.status.busy":"2022-04-18T11:21:48.228206Z","iopub.execute_input":"2022-04-18T11:21:48.228883Z","iopub.status.idle":"2022-04-18T11:21:48.249152Z","shell.execute_reply.started":"2022-04-18T11:21:48.228854Z","shell.execute_reply":"2022-04-18T11:21:48.248528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for subsequence in subsequences:\n    video_path = make_video(df, subsequence, is_subsequence=True)\n    display(HTML(f\"<h2>Subsequence ID: {subsequence}</h2>\"))\n    display(Video(video_path, width= 1280/2, height= 720/2))","metadata":{"execution":{"iopub.status.busy":"2022-04-18T11:21:48.250028Z","iopub.execute_input":"2022-04-18T11:21:48.250728Z","iopub.status.idle":"2022-04-18T11:22:18.088122Z","shell.execute_reply.started":"2022-04-18T11:21:48.250701Z","shell.execute_reply":"2022-04-18T11:22:18.087177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split, StratifiedKFold\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T11:22:18.089987Z","iopub.execute_input":"2022-04-18T11:22:18.090574Z","iopub.status.idle":"2022-04-18T11:22:18.998461Z","shell.execute_reply.started":"2022-04-18T11:22:18.090527Z","shell.execute_reply":"2022-04-18T11:22:18.997849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_split  = df.groupby(\"subsequence_id\").agg({'has_annotations': 'max', 'video_frame': 'count'}).astype(int).reset_index()\ndf_split.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T11:22:18.999335Z","iopub.execute_input":"2022-04-18T11:22:18.999939Z","iopub.status.idle":"2022-04-18T11:22:19.016617Z","shell.execute_reply.started":"2022-04-18T11:22:18.999899Z","shell.execute_reply":"2022-04-18T11:22:19.015969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir train-validation-split/","metadata":{"execution":{"iopub.status.busy":"2022-04-18T11:22:19.017969Z","iopub.execute_input":"2022-04-18T11:22:19.018407Z","iopub.status.idle":"2022-04-18T11:22:19.772202Z","shell.execute_reply.started":"2022-04-18T11:22:19.018378Z","shell.execute_reply":"2022-04-18T11:22:19.771148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def analize_split(df_train, df_val, df):\n     # Analize results\n    print(f\"   Train images                 : {len(df_train) / len(df):.3f}\")\n    print(f\"   Val   images                 : {len(df_val) / len(df):.3f}\")\n    print()\n    print(f\"   Train images with annotations: {len(df_train[df_train['has_annotations']]) / len(df[df['has_annotations']]):.3f}\")\n    print(f\"   Val   images with annotations: {len(df_val[df_val['has_annotations']]) / len(df[df['has_annotations']]):.3f}\")\n    print()\n    print(f\"   Train images w/no annotations: {len(df_train[~df_train['has_annotations']]) / len(df[~df['has_annotations']]):.3f}\")\n    print(f\"   Val   images w/no annotations: {len(df_val[~df_val['has_annotations']]) / len(df[~df['has_annotations']]):.3f}\")\n    print()\n    print(f\"   Train mean annotations       : {df_train['n_annotations'].mean():.3f}\")\n    print(f\"   Val   mean annotations       : {df_val['n_annotations'].mean():.3f}\")\n    \n    print()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T11:22:19.773678Z","iopub.execute_input":"2022-04-18T11:22:19.774036Z","iopub.status.idle":"2022-04-18T11:22:19.781653Z","shell.execute_reply.started":"2022-04-18T11:22:19.774006Z","shell.execute_reply":"2022-04-18T11:22:19.780501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for test_size in [0.01, 0.05, 0.1, 0.2]:\n    print(f\"Generating train-validation split with {test_size*100}% validation\")\n    df_train_idx, df_val_idx = train_test_split(df_split['subsequence_id'], stratify=df_split[\"has_annotations\"], test_size=test_size, random_state=42)\n    df['is_train'] = df['subsequence_id'].isin(df_train_idx)\n    df_train, df_val = df[df['is_train']], df[~df['is_train']]\n    \n    # Print some statistics\n    analize_split(df_train, df_val, df)\n    \n    # Save to file\n    f_name = f\"train-validation-split/train-{test_size}.csv\"\n    print(f\"Saving file to {f_name}\")\n    df.to_csv(f_name, index=False)\n    print()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T11:22:19.788237Z","iopub.execute_input":"2022-04-18T11:22:19.788645Z","iopub.status.idle":"2022-04-18T11:22:20.666387Z","shell.execute_reply.started":"2022-04-18T11:22:19.788614Z","shell.execute_reply":"2022-04-18T11:22:20.665471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls -l train-validation-split/","metadata":{"execution":{"iopub.status.busy":"2022-04-18T11:22:20.668921Z","iopub.execute_input":"2022-04-18T11:22:20.669559Z","iopub.status.idle":"2022-04-18T11:22:21.418715Z","shell.execute_reply.started":"2022-04-18T11:22:20.669526Z","shell.execute_reply":"2022-04-18T11:22:21.417421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.drop(\"is_train\", axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-04-18T11:22:21.420643Z","iopub.execute_input":"2022-04-18T11:22:21.421044Z","iopub.status.idle":"2022-04-18T11:22:21.432796Z","shell.execute_reply.started":"2022-04-18T11:22:21.420998Z","shell.execute_reply":"2022-04-18T11:22:21.431848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_splits = 5\nkf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=2021)\nfor fold_id, (_, val_idx) in enumerate(kf.split(df_split['subsequence_id'], y=df_split[\"has_annotations\"])):\n    subseq_val_idx = df_split['subsequence_id'].iloc[val_idx]\n    df.loc[df['subsequence_id'].isin(subseq_val_idx), 'fold'] = fold_id\n    \ndf['fold'] = df['fold'].astype(int)\ndf['fold'].value_counts(dropna=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-18T11:22:21.43442Z","iopub.execute_input":"2022-04-18T11:22:21.434691Z","iopub.status.idle":"2022-04-18T11:22:21.458177Z","shell.execute_reply.started":"2022-04-18T11:22:21.434663Z","shell.execute_reply":"2022-04-18T11:22:21.457458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for fold_id in df['fold'].sort_values().unique():\n    print(\"=============================\")\n    print(f\"Analyzing fold {fold_id}\")\n    df_train, df_val = df[df['fold'] != fold_id], df[df['fold'] == fold_id]\n    analize_split(df_train, df_val, df)\n    print()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T11:22:21.45933Z","iopub.execute_input":"2022-04-18T11:22:21.459548Z","iopub.status.idle":"2022-04-18T11:22:21.541733Z","shell.execute_reply.started":"2022-04-18T11:22:21.459523Z","shell.execute_reply":"2022-04-18T11:22:21.540489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir cross-validation/","metadata":{"execution":{"iopub.status.busy":"2022-04-18T11:22:21.543211Z","iopub.execute_input":"2022-04-18T11:22:21.543578Z","iopub.status.idle":"2022-04-18T11:22:22.295908Z","shell.execute_reply.started":"2022-04-18T11:22:21.543534Z","shell.execute_reply":"2022-04-18T11:22:22.29469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv(\"cross-validation/train-5folds.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-18T11:22:22.298302Z","iopub.execute_input":"2022-04-18T11:22:22.298992Z","iopub.status.idle":"2022-04-18T11:22:22.508487Z","shell.execute_reply.started":"2022-04-18T11:22:22.29892Z","shell.execute_reply":"2022-04-18T11:22:22.507569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_splits = 10\nkf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=2021)\nfor fold_id, (_, val_idx) in enumerate(kf.split(df_split['subsequence_id'], y=df_split[\"has_annotations\"])):\n    subseq_val_idx = df_split['subsequence_id'].iloc[val_idx]\n    df.loc[df['subsequence_id'].isin(subseq_val_idx), 'fold'] = fold_id\n    \ndf['fold'] = df['fold'].astype(int)\ndf['fold'].value_counts(dropna=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-18T11:22:22.509849Z","iopub.execute_input":"2022-04-18T11:22:22.510407Z","iopub.status.idle":"2022-04-18T11:22:22.533775Z","shell.execute_reply.started":"2022-04-18T11:22:22.510365Z","shell.execute_reply":"2022-04-18T11:22:22.533135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for fold_id in df['fold'].sort_values().unique():\n    print(\"=============================\")\n    print(f\"Analyzing fold {fold_id}\")\n    df_train, df_val = df[df['fold'] != fold_id], df[df['fold'] == fold_id]\n    analize_split(df_train, df_val, df)\n    print()","metadata":{"execution":{"iopub.status.busy":"2022-04-18T11:22:22.534618Z","iopub.execute_input":"2022-04-18T11:22:22.53533Z","iopub.status.idle":"2022-04-18T11:22:22.690102Z","shell.execute_reply.started":"2022-04-18T11:22:22.535297Z","shell.execute_reply":"2022-04-18T11:22:22.689278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv(\"cross-validation/train-10folds.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-18T11:22:22.691499Z","iopub.execute_input":"2022-04-18T11:22:22.691753Z","iopub.status.idle":"2022-04-18T11:22:22.891325Z","shell.execute_reply.started":"2022-04-18T11:22:22.691721Z","shell.execute_reply":"2022-04-18T11:22:22.890633Z"},"trusted":true},"execution_count":null,"outputs":[]}]}