{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport cv2\nimport matplotlib.pyplot as plt\n\n\n# Load the train and frames csv files\n\n# Competion train csv file\ntrain = pd.read_csv('../input/dfl-bundesliga-data-shootout/train.csv')\n\n# Extracted frames csv files\ndf = pd.read_csv('../input/dfl-video-frames-resized-270x495/videos/vid_frames/train_frames.csv')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-04T22:03:59.790843Z","iopub.execute_input":"2022-08-04T22:03:59.791297Z","iopub.status.idle":"2022-08-04T22:04:00.170573Z","shell.execute_reply.started":"2022-08-04T22:03:59.791204Z","shell.execute_reply":"2022-08-04T22:04:00.169663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Extract frames from the videos:\n\nThe output doesn't fit in the working directory, you can create a new directory /kaggle/frames/ to save all output and then zip it: Total 107GB\n\n[This dataset](https://www.kaggle.com/datasets/amiiiney/dfl-video-frames-resized-270x495) has already the frames extracted for all the videos:\n* The images are resized to 495, 270 for faster training and iteration.\n* We skip saving some frames to lower the number of total frames (We save 1/6 of the total frames).","metadata":{}},{"cell_type":"code","source":"class CFG:\n    EXTRACT_FRAMES = False\n    step = 6\n    \n# Create the folder to save the frames    \nframes_folder = \"/kaggle/frames\"\n \nif not os.path.exists(frames_folder):\n    os.mkdir(frames_folder)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:04:00.172093Z","iopub.execute_input":"2022-08-04T22:04:00.172593Z","iopub.status.idle":"2022-08-04T22:04:00.177357Z","shell.execute_reply.started":"2022-08-04T22:04:00.172564Z","shell.execute_reply":"2022-08-04T22:04:00.176012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if CFG.EXTRACT_FRAMES:\n    parent_folder = \"../input\"\n\n    # Get list of videos\n    vids = os.listdir(os.path.join(parent_folder,\"dfl-bundesliga-data-shootout/train\"))\n\n    # Initiate an empty dataframe and lists\n    df = pd.DataFrame()\n    vid_id = []\n    frames = []\n    image_path = []\n\n    # Loop over the videos\n    for video in vids:\n\n        vidcap = cv2.VideoCapture(os.path.join(parent_folder,f\"dfl-bundesliga-data-shootout/train/{video}\"))\n        success,image = vidcap.read()\n        count = 0\n        fps = vidcap.get(cv2.CAP_PROP_FPS)\n        save_path = f\"/kaggle/frames/{video.replace('.mp4', '')}\"\n\n        # Create the video folders\n        if not os.path.exists(save_path):\n            os.mkdir(save_path)\n\n        # Resize and save the images\n        while success:\n            # Skip saving n frames to reduce the number of frames\n            if (count % CFG.step) == 0:\n                image = cv2.resize(image, (495, 270))\n                cv2.imwrite(f\"{save_path}/{count/fps}.jpg\", image) # count/fps: Get the time in seconds \n\n                vid_id.append(video.replace('.mp4', ''))\n                frames.append(count/fps)\n                image_path.append(f\"{video.replace('.mp4', '')}/{count/fps}.jpg\")\n                image_path2.append(f\"{save_path}/{count/fps}.jpg\")\n\n            success,image = vidcap.read()\n            count += 1\n            #print(\"time stamp current frame:\",count/fps)\n\n\n    df['video_id'] = vid_id\n    df['time'] = frames\n    df['image_path'] = image_path\n    df.to_csv('train_frames.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:04:00.191451Z","iopub.execute_input":"2022-08-04T22:04:00.192233Z","iopub.status.idle":"2022-08-04T22:04:00.203330Z","shell.execute_reply.started":"2022-08-04T22:04:00.192201Z","shell.execute_reply":"2022-08-04T22:04:00.202354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Map the event to the frames based on the timestamp:\nThis part has some dirty coding, not all the events are in a triplet format <Start - Action - End> some have up to 5 actions in the scoring interval: Starts - X -X -X - X- X- End.\n* We map the event to the corresponding frames by localizing the start and end times of each scoring interval.\n    \nExample of a scoring interval with 1 event: Challenge","metadata":{}},{"cell_type":"code","source":"for i in range(3):\n    print(train.loc[i, \"event\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:04:00.204793Z","iopub.execute_input":"2022-08-04T22:04:00.205211Z","iopub.status.idle":"2022-08-04T22:04:00.224686Z","shell.execute_reply.started":"2022-08-04T22:04:00.205154Z","shell.execute_reply":"2022-08-04T22:04:00.223426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Example of a scoring interval with more than 1 event: Throwin, Play","metadata":{}},{"cell_type":"code","source":"for i in range(4):\n    print(train.loc[100+i, \"event\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:04:00.226373Z","iopub.execute_input":"2022-08-04T22:04:00.227057Z","iopub.status.idle":"2022-08-04T22:04:00.234567Z","shell.execute_reply.started":"2022-08-04T22:04:00.227015Z","shell.execute_reply":"2022-08-04T22:04:00.233322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,25))\n\n# Start a new column to map the event to the frames\ndf[\"event\"] = \"No label\"\nndf= pd.DataFrame()\nr1, r2, r3, r4, r5 = 0, 0, 0, 0, 0\n\n# Filter the dataset by video to make use of the timestamp of each video and avoid conflict\nvids = list(train.video_id.unique())\nfor t,v in enumerate(vids):\n    \n    plt.subplot(3,4,t+1)\n    plt.title(f\"Vid: {v}\")\n    \n    # Filter both video df in the competition train and the extracted frames CSVs\n    vid_df = train[train.video_id== v].reset_index(drop=True)\n    vid_df2 = df[df.video_id== v].reset_index(drop=True)\n    \n    # Get the start and end time of each video\n    vid_start = vid_df2.time.min()\n    vid_end = vid_df2.time.max()\n    \n    print(f\"Video: {v}\")\n    \n    # Comment out to fill the white blanks with cyan color\n    #vid_df.time.plot(color=\"cyan\") \n    \n    # Plot the starting and ending time\n    plt.axhline(y=vid_end, color=\"r\", linestyle=\"--\", alpha=0.4)\n    plt.axhline(y=vid_start, color=\"r\", linestyle=\"--\", alpha=0.4)\n    plt.text( (10),vid_end+120,f\"Video end\",fontsize=12, color=\"gray\", ha=\"left\",va=\"top\"  )\n    plt.text( (10),vid_start,f\"Video start\",fontsize=12, color=\"gray\", ha=\"left\",va=\"top\"  )\n \n    \n    # Loop over the events in the competition df\n    for i,n in enumerate(vid_df.event.values):\n        \n        if i >= len(vid_df.event.values)-1:\n            continue\n            \n        # Find the starting point of the event\n        if vid_df.loc[i, \"event\"] == \"start\":\n            \n            # Find the closest end of the event\n            for j in range(10):\n                if vid_df.loc[i+j, \"event\"]== \"end\":\n                    \n                    if j == 2: # Start - X - End\n                        r1 += 1\n                        \n                        # Localize the time interval of the event in the train df\n                        time_min = vid_df.loc[i, \"time\"]\n                        time_max = vid_df.loc[i+2, \"time\"]\n\n                        # Map the event to the corresponding frames in the time interval\n                        vid_df2.loc[(vid_df2['time'] >= time_min) & (vid_df2['time'] <= time_max), \"event\"] = vid_df.loc[i+1, \"event\"]\n\n                        # Plot the annotated sequence\n                        vid_df.iloc[i:i+2].time.plot(y=\"time\",color=\"black\")\n                        \n                                #  0      1  2    3\n                    elif j ==3: # Start - X -X - End\n                        r2 += 1\n\n                        time_min1 = vid_df.loc[i, \"time\"]\n                        time_max1 = vid_df.loc[i+2, \"time\"]\n\n                        time_min2 = vid_df.loc[i+2, \"time\"]\n                        time_max2 = vid_df.loc[i+3, \"time\"]\n\n                        vid_df2.loc[(vid_df2['time'] >= time_min1) & (vid_df2['time'] <= time_max1), \"event\"] = vid_df.loc[i+1, \"event\"]\n                        vid_df2.loc[(vid_df2['time'] >= time_min2) & (vid_df2['time'] <= time_max2), \"event\"] = vid_df.loc[i+2, \"event\"]\n                        vid_df.iloc[i:i+3].time.plot(y=\"time\",color=\"black\")\n\n                    elif j == 4: # Start - X -X - X - End\n                        r3 += 1\n\n                        time_min1 = vid_df.loc[i, \"time\"]\n                        time_max1 = vid_df.loc[i+2, \"time\"]\n\n                        time_min2 = vid_df.loc[i+2, \"time\"]\n                        time_max2 = vid_df.loc[i+3, \"time\"]\n\n                        time_min3 = vid_df.loc[i+3, \"time\"]\n                        time_max3 = vid_df.loc[i+4, \"time\"]\n\n                        vid_df2.loc[(vid_df2['time'] >= time_min1) & (vid_df2['time'] <= time_max1), \"event\"] = vid_df.loc[i+1, \"event\"]\n                        vid_df2.loc[(vid_df2['time'] >= time_min2) & (vid_df2['time'] <= time_max2), \"event\"] = vid_df.loc[i+2, \"event\"]\n                        vid_df2.loc[(vid_df2['time'] >= time_min3) & (vid_df2['time'] <= time_max3), \"event\"] = vid_df.loc[i+3, \"event\"]\n                        vid_df.iloc[i:i+4].time.plot(y=\"time\",color=\"black\")\n\n                    elif j == 5:  # Start - X -X - X - X - End\n                        r4 += 1\n                       \n                        time_min1 = vid_df.loc[i, \"time\"]\n                        time_max1 = vid_df.loc[i+2, \"time\"]\n\n                        time_min2 = vid_df.loc[i+2, \"time\"]\n                        time_max2 = vid_df.loc[i+3, \"time\"]\n\n                        time_min3 = vid_df.loc[i+3, \"time\"]\n                        time_max3 = vid_df.loc[i+4, \"time\"]\n                        \n                        time_min4 = vid_df.loc[i+4, \"time\"]\n                        time_max4 = vid_df.loc[i+5, \"time\"]\n\n                        vid_df2.loc[(vid_df2['time'] >= time_min1) & (vid_df2['time'] <= time_max1), \"event\"] = vid_df.loc[i+1, \"event\"]\n                        vid_df2.loc[(vid_df2['time'] >= time_min2) & (vid_df2['time'] <= time_max2), \"event\"] = vid_df.loc[i+2, \"event\"]\n                        vid_df2.loc[(vid_df2['time'] >= time_min3) & (vid_df2['time'] <= time_max3), \"event\"] = vid_df.loc[i+3, \"event\"]\n                        vid_df2.loc[(vid_df2['time'] >= time_min4) & (vid_df2['time'] <= time_max4), \"event\"] = vid_df.loc[i+4, \"event\"]\n                        vid_df.iloc[i:i+5].time.plot(y=\"time\",color=\"black\")\n                    \n                    elif j == 6: # Start - X -X - X - X - X - End / OR MORE\n                        \n                        #print(\"More than +5\")\n                        #for ni in range(j+1):\n                        #    print(j,vid_df.loc[i+ni, \"event\"])\n                        \n                        # I checked this manually, there are 8 scoring intervals with more than 5 actions\n                        # 6 of them are all: PLAY and 2 of them start with CHALLENGE and then a sequence of PLAY\n                        # You can double check by uncommenting the code snipet above\n                        r5 += 1\n                        \n                        time_min1 = vid_df.loc[i, \"time\"]\n                        time_max1 = vid_df.loc[i+2, \"time\"]\n                        \n                        time_min2 = vid_df.loc[i+2, \"time\"]\n                        time_max2 = vid_df.loc[i+j, \"time\"]\n                        \n                        vid_df2.loc[(vid_df2['time'] >= time_min1) & (vid_df2['time'] <= time_max1), \"event\"] = vid_df.loc[i+2, \"event\"]\n                        vid_df2.loc[(vid_df2['time'] >= time_min2) & (vid_df2['time'] <= time_max2), \"event\"] = vid_df.loc[i+j-1, \"event\"]\n                        vid_df.iloc[i:i+5].time.plot(y=\"time\",color=\"black\")\n                    \n                    else:\n                        time_min1 = vid_df.loc[i, \"time\"]\n                        time_max1 = vid_df.loc[i+j, \"time\"]\n                        \n                        vid_df2.loc[(vid_df2['time'] >= time_min1) & (vid_df2['time'] <= time_max1), \"event\"] = vid_df.loc[i+j-1, \"event\"]\n                        vid_df.iloc[i:i+5].time.plot(y=\"time\",color=\"black\")\n                        \n                    break # Break after finding the 1st End of the event\n                    \n    ndf = pd.concat([ndf, vid_df2])\n    plt.xlabel(\"row number\")\n    plt.ylabel('Time in seconds')\nprint(\"-----------------------\")\nprint(\" Scoring intervals:\")\nprint(f\"Start - X - End: \\n   {r1} intervals\")\nprint(f\"Start - X - X - End: \\n   {r2} intervals\")\nprint(f\"Start - X - X - X - End: \\n   {r3} intervals \")\nprint(f\"Start - X - X - X -X - End: \\n   {r4} intervals\")\nprint(f\"Start - +5-X - End: \\n   {r5}: intervals\")\nprint(\"-----------------------\")\n          ","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:04:00.236211Z","iopub.execute_input":"2022-08-04T22:04:00.236625Z","iopub.status.idle":"2022-08-04T22:04:34.041314Z","shell.execute_reply.started":"2022-08-04T22:04:00.236596Z","shell.execute_reply":"2022-08-04T22:04:34.040394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Scoring intervals:\n\n*Black line: Annotated time interval* \n\n*White blank: Not annotated time interval*\n* The discontinuities in the plots show the time intervals that are not labeled.\n* Most of the videos are not labeled from the beginning, some scoring intervals start at the 500th second\n* Half of the videos finish around 500 seconds before the end ","metadata":{}},{"cell_type":"code","source":"train.event.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:04:34.042545Z","iopub.execute_input":"2022-08-04T22:04:34.043033Z","iopub.status.idle":"2022-08-04T22:04:34.051582Z","shell.execute_reply.started":"2022-08-04T22:04:34.043002Z","shell.execute_reply":"2022-08-04T22:04:34.050512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ndf.event.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:04:34.055491Z","iopub.execute_input":"2022-08-04T22:04:34.055823Z","iopub.status.idle":"2022-08-04T22:04:34.073799Z","shell.execute_reply.started":"2022-08-04T22:04:34.055795Z","shell.execute_reply":"2022-08-04T22:04:34.072528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"The percentage of the 'No label' images: {round(131497/len(ndf),3)*100}%\")","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:04:34.077617Z","iopub.execute_input":"2022-08-04T22:04:34.078013Z","iopub.status.idle":"2022-08-04T22:04:34.083304Z","shell.execute_reply.started":"2022-08-04T22:04:34.077982Z","shell.execute_reply":"2022-08-04T22:04:34.082185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The ratio of \"challenge\" to \"play\" and \"throwin\" to \"play\" seem to be consistent between both the train internvals and the frames.\n* 131497 frames, which is roughly 75% of the total frames have no label, which doesn't correlate with the plots (!?) I might have a bug somewhere in the labels mapping, if someone finds a bug or a better way to map the labels, feel free to let me know :) ","metadata":{}},{"cell_type":"markdown","source":"### Mapped dataframe:","metadata":{}},{"cell_type":"code","source":"print(ndf.shape)\nndf.sample(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:11:27.456314Z","iopub.execute_input":"2022-08-04T22:11:27.456757Z","iopub.status.idle":"2022-08-04T22:11:27.478505Z","shell.execute_reply.started":"2022-08-04T22:11:27.456724Z","shell.execute_reply":"2022-08-04T22:11:27.477541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dataframe with annotations only:","metadata":{}},{"cell_type":"code","source":"annotations_only_df = ndf[ndf.event != \"No label\"]\nprint(annotations_only_df.shape)\nannotations_only_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:10:39.593537Z","iopub.execute_input":"2022-08-04T22:10:39.594031Z","iopub.status.idle":"2022-08-04T22:10:39.627604Z","shell.execute_reply.started":"2022-08-04T22:10:39.593986Z","shell.execute_reply":"2022-08-04T22:10:39.626461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Save the mapped dataframe:\nThe dataframe can be used for the typical image classification models or sequential models by dropping the \"No label\" frames.","metadata":{}},{"cell_type":"code","source":"ndf.to_csv('train_labeled_frames.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:04:34.109042Z","iopub.execute_input":"2022-08-04T22:04:34.109694Z","iopub.status.idle":"2022-08-04T22:04:34.645456Z","shell.execute_reply.started":"2022-08-04T22:04:34.109661Z","shell.execute_reply":"2022-08-04T22:04:34.644252Z"},"trusted":true},"execution_count":null,"outputs":[]}]}