{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-19T10:22:42.838285Z","iopub.execute_input":"2023-02-19T10:22:42.838747Z","iopub.status.idle":"2023-02-19T10:22:42.867583Z","shell.execute_reply.started":"2023-02-19T10:22:42.838659Z","shell.execute_reply":"2023-02-19T10:22:42.865812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def split_sequence(seq):\n    result = []\n    len_list = []\n    start = 0\n    for i in range(1, len(seq)):\n        if seq[i] != seq[i-1]+1:\n            result.append(seq[start:i])\n            len_list.append(len(seq[start:i]))\n            start = i\n    result.append(seq[start:])\n    len_list.append(len(seq[start:]))\n    return result, len_list\n\n# 示例使用\nseq = [1, 2, 3, 5, 4, 6, 7, 8, 10, 9, 11]\nresult, len_list = split_sequence(seq)\nprint(result)\nprint(len_list)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T10:22:42.870113Z","iopub.execute_input":"2023-02-19T10:22:42.870504Z","iopub.status.idle":"2023-02-19T10:22:42.879957Z","shell.execute_reply.started":"2023-02-19T10:22:42.870472Z","shell.execute_reply":"2023-02-19T10:22:42.878598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df = pd.read_csv('/kaggle/input/nfl-player-contact-detection/train_labels.csv')\n#df","metadata":{"execution":{"iopub.status.busy":"2023-02-19T10:22:42.881439Z","iopub.execute_input":"2023-02-19T10:22:42.881824Z","iopub.status.idle":"2023-02-19T10:22:42.894670Z","shell.execute_reply.started":"2023-02-19T10:22:42.881792Z","shell.execute_reply":"2023-02-19T10:22:42.892712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# group = ['game_play','nfl_player_id_1','nfl_player_id_2']\n# show = ['step', 'contact']\n\n# df_gp = df[group + show].groupby(group)\n\n# print(\"Just print double contact with same person.\")\n# for name, group in df_gp:\n    \n#     group = group[show].reset_index(drop=True)\n#     contact_step = group.loc[group['contact'] == 1, 'step'].values \n    \n#     if len(contact_step) > 0:\n        \n#         result, len_list = split_sequence(contact_step)\n#         if len(result) > 1:\n#             print(name)\n#             for i in range(len(result)):\n#                 print(\"length: \" + str(len_list[i]) + \" from \" + str(result[i][0]) + \" to \" + str(result[i][-1]))\n#         # print(contact_step)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T10:22:42.898529Z","iopub.execute_input":"2023-02-19T10:22:42.898949Z","iopub.status.idle":"2023-02-19T10:22:42.905598Z","shell.execute_reply.started":"2023-02-19T10:22:42.898908Z","shell.execute_reply":"2023-02-19T10:22:42.904487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/xgb-7299-model/oof_pred.csv')\nthreshold = 0.20957\n# df['oof'] = (df['oof'].values > threshold).astype(int)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-19T10:39:57.969870Z","iopub.execute_input":"2023-02-19T10:39:57.970241Z","iopub.status.idle":"2023-02-19T10:39:59.693825Z","shell.execute_reply.started":"2023-02-19T10:39:57.970211Z","shell.execute_reply":"2023-02-19T10:39:59.692315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef plot_label_oof(df_combos, combos, threshold = 0.2, label1 = 'ct', label2 = 'of', name = 'show'):\n    sns.set_style('whitegrid')\n    plt.figure()\n    fig, ax = plt.subplots(10,10,figsize=(18,22))\n    # print(len(combos))\n    for i in range(1,len(combos) + 1):\n        combo = combos[i-1]\n        df = df_combos[i-1]\n        plt.subplot(10,10,i)\n        plt.plot(df['step'], df['contact'], label=label1)\n        plt.plot(df['step'], df['oof'], label=label2)\n        plt.plot(df['step'], np.ones(len(df['step']))*threshold, label=label2)\n        plt.xlabel((str(combo[0]) + '_' + str(combo[1]) + '_' + str(combo[2])), fontsize=6)\n        locs, labels = plt.xticks()\n        plt.tick_params(axis='x', which='major', labelsize=4, pad=-6)\n        plt.tick_params(axis='y', which='major', labelsize=4)\n    plt.savefig(name + \".png\")\n    plt.show();\n    \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2023-02-19T11:24:41.926407Z","iopub.execute_input":"2023-02-19T11:24:41.926791Z","iopub.status.idle":"2023-02-19T11:24:41.938962Z","shell.execute_reply.started":"2023-02-19T11:24:41.926760Z","shell.execute_reply":"2023-02-19T11:24:41.937246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.ones(4)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T10:43:24.393186Z","iopub.execute_input":"2023-02-19T10:43:24.393785Z","iopub.status.idle":"2023-02-19T10:43:24.400983Z","shell.execute_reply.started":"2023-02-19T10:43:24.393749Z","shell.execute_reply":"2023-02-19T10:43:24.399831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings('ignore')\n\ngroup = ['game_play','nfl_player_id_1','nfl_player_id_2']\nshow = ['step', 'oof', 'contact']\n\ndf_gp = df[group + show].groupby(group)\n\ndf_list = []\ncombo_list = []\nfor combo, df  in df_gp:\n    if (df['oof'].sum() < 1) and (df['contact'].sum() < 1):\n        continue\n    # print(df['oof'].sum(), df['contact'].sum())\n    df_list.append(df)\n    combo_list.append(combo)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-19T10:39:59.957003Z","iopub.execute_input":"2023-02-19T10:39:59.957379Z","iopub.status.idle":"2023-02-19T10:40:17.851254Z","shell.execute_reply.started":"2023-02-19T10:39:59.957348Z","shell.execute_reply":"2023-02-19T10:40:17.849926Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"figs_num = np.ceil(len(df_list) / 100).astype('int16')\n\nfor i in range(figs_num):\n    st = i * 100\n    ed = np.min([(i + 1) * 100, len(df_list)])\n    name = \"from_\" + str(st) + \"_to_\" + str(ed)\n    plot_label_oof(df_list[st:ed], combo_list[st:ed], threshold, name = name)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T11:24:51.768180Z","iopub.execute_input":"2023-02-19T11:24:51.768573Z","iopub.status.idle":"2023-02-19T11:25:29.226283Z","shell.execute_reply.started":"2023-02-19T11:24:51.768540Z","shell.execute_reply":"2023-02-19T11:25:29.225001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# group = ['game_play','nfl_player_id_1','nfl_player_id_2']\n# show = ['step', 'oof']\n\n# df_gp = df[group + show].groupby(group)\n\n# print(\"Just print double contact with same person.\")\n# for name, group in df_gp:\n    \n#     group = group[show].reset_index(drop=True)\n#     contact_step = group.loc[group['oof'] == 1, 'step'].values \n    \n#     if len(contact_step) > 0:\n        \n#         result, len_list = split_sequence(contact_step)\n#         if len(result) > 1:\n#             print(name)\n#             for i in range(len(result)):\n#                 print(\"length: \" + str(len_list[i]) + \" from \" + str(result[i][0]) + \" to \" + str(result[i][-1]))\n#             # print(contact_step)","metadata":{"execution":{"iopub.status.busy":"2023-02-18T09:05:29.319860Z","iopub.execute_input":"2023-02-18T09:05:29.320419Z","iopub.status.idle":"2023-02-18T09:06:39.280479Z","shell.execute_reply.started":"2023-02-18T09:05:29.320385Z","shell.execute_reply":"2023-02-18T09:06:39.279099Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}