{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport sys\nsys.path.append('/kaggle/input/timm-0-6-9/pytorch-image-models-master')\nimport glob\nimport numpy as np\nimport pandas as pd\nimport random\nimport math\nimport gc\nimport cv2\nfrom tqdm import tqdm\nimport time\nfrom functools import lru_cache\nimport torch\nfrom torch import nn\nfrom torch.nn import functional as F\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.cuda.amp import autocast, GradScaler\nimport timm\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import matthews_corrcoef","metadata":{"execution":{"iopub.status.busy":"2023-01-27T23:29:02.630207Z","iopub.execute_input":"2023-01-27T23:29:02.630754Z","iopub.status.idle":"2023-01-27T23:29:05.996241Z","shell.execute_reply.started":"2023-01-27T23:29:02.630704Z","shell.execute_reply":"2023-01-27T23:29:05.994875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install timm","metadata":{"execution":{"iopub.status.busy":"2023-01-27T23:28:25.913759Z","iopub.execute_input":"2023-01-27T23:28:25.914175Z","iopub.status.idle":"2023-01-27T23:28:40.90615Z","shell.execute_reply.started":"2023-01-27T23:28:25.914143Z","shell.execute_reply":"2023-01-27T23:28:40.904926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# timm module\n사전 학습된 model들을 pytorch에서 활용하기 위한 timm 모듈을 추천..\n\n일부 모델들은 pretrained=True를 사용할때 오류가 나기때문에 모든 모델을 활용하는 것은 불가능하지만\n\ntorchivision에서 제공하는 pretrained model보다 (https://pytorch.org/vision/stable/models.html)\n\ntimm이 더 많은 사전학습된 모델들을 제공하기 때문에 timm 모듈 사용을 추천한다.","metadata":{}},{"cell_type":"code","source":"CFG = {\n        'seed' : 42,\n    'model' : 'resnet50',\n    'img_size' : 256,\n    'epochs' : 10,\n    'train_bs' : 100,\n    'valid_bs' : 64,\n    'lr' : 1e-3,\n    'weight_decay' : 1e-6,\n    'num_workers' :2\n       }","metadata":{"execution":{"iopub.status.busy":"2023-01-27T23:29:08.594138Z","iopub.execute_input":"2023-01-27T23:29:08.594936Z","iopub.status.idle":"2023-01-27T23:29:08.602456Z","shell.execute_reply.started":"2023-01-27T23:29:08.594878Z","shell.execute_reply":"2023-01-27T23:29:08.60095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# weight decay\noverfitting을 방지하기 위해 weight decay\n\n 모델의 weight의 제곱합을 패널티 텀으로 주어 (=제약을 걸어) loss를 최소화 하는 것을 말한다. 이는 L2 regularization과 동일하며 L2 penalty라고도 부른다. \n \n [l2](https://blog.janestreet.com/l2-regularization-and-batch-norm/)","metadata":{}},{"cell_type":"code","source":"def seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    \nseed_everything(CFG['seed'])\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"execution":{"iopub.status.busy":"2023-01-27T23:30:04.018068Z","iopub.execute_input":"2023-01-27T23:30:04.018527Z","iopub.status.idle":"2023-01-27T23:30:04.030184Z","shell.execute_reply.started":"2023-01-27T23:30:04.018489Z","shell.execute_reply":"2023-01-27T23:30:04.028566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# How to fix seed \nhttps://sswwd.tistory.com/61\n\npytorch seed\n\ntensorflow seed ","metadata":{}},{"cell_type":"code","source":"def expand_contact_id(df):\n    '''\n    Splits out contact_id into seperate columns.\n    '''\n    \n    df['game_play'] = df['contact_id'].str[:12]\n    df['step'] = df['contact_id'].str.split(\"_\").str[-3]\n    df['nfl_player_id_1'] = df['contact_id'].str.split(\"_\").str[-2]\n    df['nfl_player_id_2'] = df['contact_id'].str.split(\"_\").str[-1]\n    return df\n\nlabels = expand_contact_id(pd.read_csv(\"/kaggle/input/nfl-player-contact-detection/sample_submission.csv\"))\n\ntest_tracking = pd.read_csv(\"/kaggle/input/nfl-player-contact-detection/test_player_tracking.csv\")\n\ntest_helmets = pd.read_csv(\"/kaggle/input/nfl-player-contact-detection/test_baseline_helmets.csv\")\n\ntest_video_metadata = pd.read_csv(\"/kaggle/input/nfl-player-contact-detection/test_video_metadata.csv\")\n    \n","metadata":{"execution":{"iopub.status.busy":"2023-01-27T23:33:40.100882Z","iopub.execute_input":"2023-01-27T23:33:40.101499Z","iopub.status.idle":"2023-01-27T23:33:40.869497Z","shell.execute_reply.started":"2023-01-27T23:33:40.10145Z","shell.execute_reply":"2023-01-27T23:33:40.868159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p ../work/frames\n\nfor video in tqdm(test_helmets.video.unique()):\n    if 'Endzone2' not in video:\n        !ffmpeg -i /kaggle/input/nfl-player-contact-detection/test/{video} -q:v 2 -f image2 /kaggle/work/frames/{video}_%04d.jpg -hide_banner -loglevel error","metadata":{"execution":{"iopub.status.busy":"2023-01-27T23:33:43.558679Z","iopub.execute_input":"2023-01-27T23:33:43.559768Z","iopub.status.idle":"2023-01-27T23:34:17.250927Z","shell.execute_reply.started":"2023-01-27T23:33:43.559708Z","shell.execute_reply":"2023-01-27T23:34:17.249301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Linux command\n\n- mkdir \n\nOption -p : create subdirectories\n\n- ffmpeg https://www.nuridol.net/command/ffmpeg/\n\nFFmpeg is an open-source software that is widely used for video-related processing.\n\n-i path\n\n-q:v 2 -f ?\n\n-hide_banner :  to hide the information of FFmpeg that appears when running and to output log level info and progress information. \n\n-loglevel error : the log is output only when something goes wrong, so the screen becomes a bit cleaner.\n","metadata":{}},{"cell_type":"code","source":"# labels[\"nfl_player_id_1\"].notnull() # 49588\n# len(labels[\"nfl_player_id_1\"].notnull()) # 49588\nlabels[\"nfl_player_id_2\"].value_counts() # G?","metadata":{"execution":{"iopub.status.busy":"2023-01-28T00:09:02.639975Z","iopub.execute_input":"2023-01-28T00:09:02.640512Z","iopub.status.idle":"2023-01-28T00:09:02.660533Z","shell.execute_reply.started":"2023-01-28T00:09:02.640464Z","shell.execute_reply":"2023-01-28T00:09:02.658898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_features(df, tr_tracking, merge_col=\"step\", use_cols=[\"x_position\", \"y_position\"]):\n    output_cols = []\n    df_combo = (\n        df.astype({\"nfl_player_id_1\": \"str\"})\n        .merge(\n            tr_tracking.astype({\"nfl_player_id\": \"str\"})[\n                [\"game_play\", merge_col, \"nfl_player_id\",] + use_cols\n            ],\n            left_on=[\"game_play\", merge_col, \"nfl_player_id_1\"],\n            right_on=[\"game_play\", merge_col, \"nfl_player_id\"],\n            how=\"left\",\n        )\n        .rename(columns={c: c+\"_1\" for c in use_cols})\n        .drop(\"nfl_player_id\", axis=1)\n        .merge(\n            tr_tracking.astype({\"nfl_player_id\": \"str\"})[\n                [\"game_play\", merge_col, \"nfl_player_id\"] + use_cols\n            ],\n            left_on=[\"game_play\", merge_col, \"nfl_player_id_2\"],\n            right_on=[\"game_play\", merge_col, \"nfl_player_id\"],\n            how=\"left\",\n        )\n        .drop(\"nfl_player_id\", axis=1)\n        .rename(columns={c: c+\"_2\" for c in use_cols})\n        .sort_values([\"game_play\", merge_col, \"nfl_player_id_1\", \"nfl_player_id_2\"])\n        .reset_index(drop=True)\n    )\n    output_cols += [c+\"_1\" for c in use_cols]\n    output_cols += [c+\"_2\" for c in use_cols]\n    \n    if (\"x_position\" in use_cols) & (\"y_position\" in use_cols):\n        index = df_combo['x_position_2'].notnull()\n        \n        distance_arr = np.full(len(index), np.nan)\n        tmp_distance_arr = np.sqrt(\n            np.square(df_combo.loc[index, \"x_position_1\"] - df_combo.loc[index, \"x_position_2\"])\n            + np.square(df_combo.loc[index, \"y_position_1\"]- df_combo.loc[index, \"y_position_2\"])\n        )\n        \n        distance_arr[index] = tmp_distance_arr\n        df_combo['distance'] = distance_arr\n        output_cols += [\"distance\"]\n        \n    df_combo['G_flug'] = (df_combo['nfl_player_id_2']==\"G\")\n    output_cols += [\"G_flug\"]\n    return df_combo, output_cols\n\n\nuse_cols = [\n    'x_position', 'y_position', 'speed', 'distance',\n    'direction', 'orientation', 'acceleration', 'sa'\n]\n\ntest, feature_cols = create_features(labels, test_tracking, use_cols=use_cols)\ntest","metadata":{"execution":{"iopub.status.busy":"2023-01-28T00:16:01.852015Z","iopub.execute_input":"2023-01-28T00:16:01.852535Z","iopub.status.idle":"2023-01-28T00:16:01.962621Z","shell.execute_reply.started":"2023-01-28T00:16:01.852495Z","shell.execute_reply":"2023-01-28T00:16:01.960969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# not distance>2 == distance < 2 ??\n# step/10 * 59.94   +  5 * 59.94    + 1 ??\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_filtered = test.query('not distance>2').reset_index(drop = True)\ntest_filtered['frame'] = (test_filtered[\"step\"]/10*59.94 +5*59.94).astype('int') +1\ntest_filtered","metadata":{"execution":{"iopub.status.busy":"2023-01-28T00:18:09.777587Z","iopub.execute_input":"2023-01-28T00:18:09.778044Z","iopub.status.idle":"2023-01-28T00:18:09.809103Z","shell.execute_reply.started":"2023-01-28T00:18:09.777997Z","shell.execute_reply":"2023-01-28T00:18:09.80743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"gc collect\n\nhttps://medium.com/dmsfordsm/garbage-collection-in-python-777916fd3189","metadata":{}},{"cell_type":"code","source":"gc.collect() # ?","metadata":{"execution":{"iopub.status.busy":"2023-01-28T00:19:46.019448Z","iopub.execute_input":"2023-01-28T00:19:46.019944Z","iopub.status.idle":"2023-01-28T00:19:46.392573Z","shell.execute_reply.started":"2023-01-28T00:19:46.0199Z","shell.execute_reply":"2023-01-28T00:19:46.390983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}