{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Coordinates Features (Pathes) Engineering","metadata":{"id":"wel9fZYkzgE7"}},{"cell_type":"markdown","source":"Thank you for visitiong this notebook! 👋🙂\nThis notebook is an auxiliary notebook for https://www.kaggle.com/code/ivanisaev/catboost-with-coordinates-features-patches/\n<div style=\"background-color:#d4f1f4; padding: 20px;\">\n<p> In this notebook provided pipeline for \n<p> 📌  Pathes engineering using room_coor_x and room_coor_y. For example, you can divide a 9x9 image to 9 patches, each patches is 3x3, the patches are coded with number [1,2,3,4,5,6,7,8,9]. Here I use 50 patches.\n<p> 📌  Calculating statistics for each patch (ciunt, mean, std, etc.) and adding this statistics for each patch in session as a feature.\n<\\div>","metadata":{"id":"aFlSEE-OzgE-"}},{"cell_type":"markdown","source":"### Data Loading","metadata":{"id":"vWmwuvuBzgFA"}},{"cell_type":"code","source":"from google.colab import drive\ndrive.mount('/content/drive')","metadata":{"id":"OA0vttrsz4UQ","outputId":"bff37e50-c99f-474c-ca76-b79775212361"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd, numpy as np, gc\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import f1_score","metadata":{"id":"He9cGiAGDmAP"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd, numpy as np, gc\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import f1_score\n# seed everything\n# torch.manual_seed(101)\nnp.random.seed(101)\n\n# READ USER ID ONLY\ntmp = pd.read_csv(\"/content/drive/MyDrive/ref-predict-student-performance-from-game-play/train.csv\",usecols=[0])\ntmp = tmp.groupby('session_id').session_id.agg('count')\n\n# COMPUTE READS AND SKIPS\nPIECES = 10\nCHUNK = int( np.ceil(len(tmp)/PIECES) )\n\nreads = []\nskips = [0]\nfor k in range(PIECES):\n    a = k*CHUNK\n    b = (k+1)*CHUNK\n    if b>len(tmp): b=len(tmp)\n    r = tmp.iloc[a:b].sum()\n    reads.append(r)\n    skips.append(skips[-1]+r)\n    \nprint(f'To avoid memory error, we will read train in {PIECES} pieces of sizes:')\nprint(reads)","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:57:41.722176Z","iopub.execute_input":"2023-05-14T00:57:41.722532Z","iopub.status.idle":"2023-05-14T00:59:04.052511Z","shell.execute_reply.started":"2023-05-14T00:57:41.722495Z","shell.execute_reply":"2023-05-14T00:59:04.051069Z"},"id":"G60OfjXozgFB","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp1 = pd.read_csv(\"/content/drive/MyDrive/ref-predict-student-performance-from-game-play/train.csv\",usecols=['room_coor_x', 'room_coor_y'])","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:59:04.054082Z","iopub.execute_input":"2023-05-14T00:59:04.054607Z","iopub.status.idle":"2023-05-14T00:59:50.646117Z","shell.execute_reply.started":"2023-05-14T00:59:04.054565Z","shell.execute_reply":"2023-05-14T00:59:50.644844Z"},"id":"b5Fk3NqYzgFC","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_height = tmp1['room_coor_x'].max()\nmin_height = tmp1['room_coor_x'].min()\nmax_width = tmp1['room_coor_y'].max()\nmin_width = tmp1['room_coor_y'].min()","metadata":{"execution":{"iopub.status.busy":"2023-05-14T01:00:06.263109Z","iopub.execute_input":"2023-05-14T01:00:06.263548Z","iopub.status.idle":"2023-05-14T01:00:06.934299Z","shell.execute_reply.started":"2023-05-14T01:00:06.263506Z","shell.execute_reply":"2023-05-14T01:00:06.933158Z"},"id":"v1FzK5i9zgFD","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del tmp\ndel tmp1 \ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-14T01:00:12.918711Z","iopub.execute_input":"2023-05-14T01:00:12.920209Z","iopub.status.idle":"2023-05-14T01:00:13.069872Z","shell.execute_reply.started":"2023-05-14T01:00:12.920150Z","shell.execute_reply":"2023-05-14T01:00:13.068710Z"},"id":"1bYVoo7QzgFD","outputId":"f62c3c18-f41a-4709-a489-787572df9b82","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_px = 8\nnum_py = 6\n# Calculate patch index given a coordinate (x,y)\ndef get_patch_index(x, y, min_image_height, max_image_height, min_image_width, max_image_width, num_patches_height = 8, num_patches_width = 6):\n    if np.isnan(x):\n        return num_px * num_py + 1\n    patch_height = (max_image_height - min_image_height + 1) / num_patches_height\n    patch_width = (max_image_width - min_image_width + 1) / num_patches_width\n    patch_x = (x - min_image_height) // patch_height\n    patch_y = (y - min_image_width) // patch_width\n    patch_index = patch_x * num_patches_width + patch_y + 1\n    return patch_index","metadata":{"execution":{"iopub.status.busy":"2023-05-14T01:00:20.902718Z","iopub.execute_input":"2023-05-14T01:00:20.903241Z","iopub.status.idle":"2023-05-14T01:00:20.912972Z","shell.execute_reply.started":"2023-05-14T01:00:20.903190Z","shell.execute_reply":"2023-05-14T01:00:20.911731Z"},"id":"gGHzEWV-zgFF","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n%%time\n\n# PROCESS TRAIN DATA IN PIECES\ngps = []\nprint(f'Processing train as {PIECES} pieces to avoid memory error... ')\nfor k in range(PIECES):\n    print(k,', ',end='')\n    SKIPS = 0\n    if k>0: SKIPS = range(1,skips[k]+1)\n    train = pd.read_csv('/content/drive/MyDrive/ref-predict-student-performance-from-game-play/train.csv',\n                        usecols = ['session_id', 'level_group', 'room_coor_x', 'room_coor_y'],\n                        nrows=reads[k], skiprows=SKIPS)\n    for _, session in train.groupby('session_id'):\n        for _, gp in session.groupby('level_group'):\n            gp['patch'] = gp.apply(lambda row : get_patch_index(row['room_coor_x'], row['room_coor_y'], min_height, max_height, min_width, max_width, num_px, num_py), axis = 1)\n            gps.append(gp)\n    \n# CONCATENATE ALL PIECES\nprint('\\n')\ndel train; gc.collect()\nnum_patches = num_px * num_py + 2\ndf = pd.concat(gps)\nprint('Shape of all train data after adding patch:', df.shape )\n# df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-14T01:01:28.255933Z","iopub.execute_input":"2023-05-14T01:01:28.256428Z"},"id":"aCDPn_13zgFF","outputId":"28e2c189-d2f6-42eb-e134-153514b2558c","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('/content/drive/MyDrive/ref-predict-student-performance-from-game-play/df_patch.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:46:09.692569Z","iopub.execute_input":"2023-05-14T00:46:09.693166Z"},"id":"-cKrFP-UzgFF","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = pd.read_csv('/content/drive/MyDrive/ref-predict-student-performance-from-game-play/df_patch.csv')","metadata":{"id":"luMTVTtV4Dil"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp.head(10)","metadata":{"id":"iU--SuanDsK5","outputId":"ca32d82d-8a91-44bd-c6f9-d46fb23a39c1"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATS = ['patch']\nPATCHES = [int(a) for a in sorted(tmp.patch.unique())]\nNUMS = ['room_coor_x', 'room_coor_y']\ndef feature_engineer(train):\n    \n    dfs = []\n    for c in CATS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in CATS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('count')\n        tmp.name = tmp.name + '_count'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    \n    for c in PATCHES: \n        train[f'patch{c}'] = (train.patch == c).astype('int64')\n    for c in PATCHES:\n        tmp = train.groupby(['session_id','level_group'])[f'patch{c}'].agg('sum')\n        tmp.name = str(tmp.name) + '_sum'\n        dfs.append(tmp)\n    for c in PATCHES:\n        tmp = train.groupby(['session_id','level_group'])[f'patch{c}'].agg('mean')\n        tmp.name = str(tmp.name) + '_mean'\n        dfs.append(tmp)\n    for c in PATCHES:\n        tmp = train.groupby(['session_id','level_group'])[f'patch{c}'].agg('std')\n        tmp.name = str(tmp.name) + '_std'\n        dfs.append(tmp)\n\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n        \n    df = pd.concat(dfs,axis=1)\n    for p in PATCHES:\n        tmp1 = train.loc[train[f'patch{p}'] == 1].groupby(['session_id','level_group'], as_index = False)['room_coor_x'].agg('mean').rename(columns = {'room_coor_x':f'patch_{p}_mean'})\n        tmp2 = train.loc[train[f'patch{p}'] == 1].groupby(['session_id','level_group'], as_index = False)['room_coor_x'].agg('std').rename(columns = {'room_coor_x':f'patch_{p}_std'})\n        tmp1[f'patch_{p}_std'] = tmp2[f'patch_{p}_std']\n        df.merge(tmp1, on = ['session_id','level_group'])\n    \n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"id":"wRxB-hnbzgFG","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# PROCESS TRAIN DATA IN PIECES\nall_pieces = []\nprint(f'Processing train as {PIECES} pieces to avoid memory error... ')\nfor k in range(PIECES):\n    print(k,', ',end='')\n    SKIPS = 0\n    if k>0: SKIPS = range(1,skips[k]+1)\n    train = pd.read_csv('/content/drive/MyDrive/ref-predict-student-performance-from-game-play/df_patch.csv',\n                        nrows=reads[k], skiprows=SKIPS)\n    df = feature_engineer(train)\n    all_pieces.append(df)\n    \n# CONCATENATE ALL PIECES\nprint('\\n')\ndel train; gc.collect()\ndf = pd.concat(all_pieces, axis=0)\nprint('Shape of all train data after feature engineering:', df.shape )\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:21:25.901115Z","iopub.execute_input":"2023-05-14T00:21:25.901899Z","iopub.status.idle":"2023-05-14T00:21:25.994675Z","shell.execute_reply.started":"2023-05-14T00:21:25.901827Z","shell.execute_reply":"2023-05-14T00:21:25.993423Z"},"id":"eDZ9-y5WzgFG","outputId":"48a20d20-ec57-4a5b-f869-71d641ab9c32","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.reset_index()","metadata":{"id":"gXcokx7_MCGo"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(2)","metadata":{"id":"EjDn2lVjM3D6","outputId":"e1e61a3d-38b5-487f-de42-a1e3a4abc9c4"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('/content/drive/MyDrive/ref-predict-student-performance-from-game-play/df_patch_features_w_sid.csv', index=False)","metadata":{"id":"CiW5grYZzgFH","trusted":true},"execution_count":null,"outputs":[]}]}