{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# PTLS working submission\nHi there!\n[Here](https://www.kaggle.com/code/ivanisaev/embeds-catboost-w-embed-features/) I wrote about my experiment with Catboost embedding fetures using Pytorch lifestream. As I wrote I got promissing results in offline testing (0.695 Catboost baseline with my embedding features gave me 0.699) but models were too havy (more 8 GB for 18 questions) and I couldn't make inference in 8 Gb RAM notebook.\n\nWhen @vadimkamaev shared [this notebook](https://www.kaggle.com/code/vadimkamaev/catboost) I desided to built lightweight embeddings only on coordinates patches (not 'patch + event type + elapsed time diff' as before) and only per level groups (not per level as before). \n\nAnd I got quite lightweight models (les than 0.5 Gb for all 18 models). Then I used inference template which I shared before to make inference.\n\nThen I as before understood that I need all required packages for Pytorch lifestream (about 20) without Internet and imported them locally. And it works fine for sumbission data.\n\nSo the inference is working and I got results thats are slightly differ from notebook without embedding fetures. I believe that they could be good but my notebook has Notebook Threw Exception. If you find the way to fix it then you will possibly will have a good solution to submit. I am ready to share embedding training and model training code if you need.","metadata":{}},{"cell_type":"code","source":"import pandas as pd, numpy as np\nfrom catboost import CatBoostClassifier\nimport pickle\nimport sys","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:48:44.095079Z","iopub.execute_input":"2023-06-18T20:48:44.095999Z","iopub.status.idle":"2023-06-18T20:48:45.417318Z","shell.execute_reply.started":"2023-06-18T20:48:44.095955Z","shell.execute_reply":"2023-06-18T20:48:45.416199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Train Data and Labels","metadata":{"papermill":{"duration":0.004542,"end_time":"2023-02-07T00:59:59.189777","exception":false,"start_time":"2023-02-07T00:59:59.185235","status":"completed"},"tags":[]}},{"cell_type":"code","source":"dtypes = {\"session_id\": 'int64',\n          \"index\": np.int16,\n          \"elapsed_time\": np.int32,\n          \"event_name\": 'category',\n          \"name\": 'category',\n          \"level\": np.int8,\n          \"page\": np.float16,\n          \"room_coor_x\": np.float16,\n          \"room_coor_y\": np.float16,\n          \"screen_coor_x\": np.float16,\n          \"screen_coor_y\": np.float16,\n          \"hover_duration\": np.float32,\n          \"text\": 'category',\n          \"fqid\": 'category',\n          \"room_fqid\": 'category',\n          \"text_fqid\": 'category',\n          \"fullscreen\": np.int8,\n          \"hq\": np.int8,\n          \"music\": np.int8,\n          \"level_group\": 'category'\n          }\nuse_col = ['session_id', 'index', 'elapsed_time', 'event_name', 'name', 'level', 'page',\n           'room_coor_x', 'room_coor_y', 'hover_duration', 'text', 'fqid', 'room_fqid', 'text_fqid', 'level_group']","metadata":{"papermill":{"duration":59.284316,"end_time":"2023-02-07T01:00:58.478743","exception":false,"start_time":"2023-02-07T00:59:59.194427","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T20:48:45.421973Z","iopub.execute_input":"2023-06-18T20:48:45.423861Z","iopub.status.idle":"2023-06-18T20:48:45.431261Z","shell.execute_reply.started":"2023-06-18T20:48:45.423816Z","shell.execute_reply":"2023-06-18T20:48:45.430628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]) )\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\n# print( targets.shape )\n# targets.head()","metadata":{"papermill":{"duration":0.598155,"end_time":"2023-02-07T01:00:59.082015","exception":false,"start_time":"2023-02-07T01:00:58.48386","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T20:48:45.434314Z","iopub.execute_input":"2023-06-18T20:48:45.435343Z","iopub.status.idle":"2023-06-18T20:48:46.387058Z","shell.execute_reply.started":"2023-06-18T20:48:45.435282Z","shell.execute_reply":"2023-06-18T20:48:46.385950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_df = pd.read_csv('/kaggle/input/featur/feature_sort.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:48:46.389044Z","iopub.execute_input":"2023-06-18T20:48:46.389304Z","iopub.status.idle":"2023-06-18T20:48:46.502296Z","shell.execute_reply.started":"2023-06-18T20:48:46.389278Z","shell.execute_reply":"2023-06-18T20:48:46.501203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineer","metadata":{"papermill":{"duration":0.005196,"end_time":"2023-02-07T01:00:59.092865","exception":false,"start_time":"2023-02-07T01:00:59.087669","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def delt_time_def(df):\n    df.sort_values(by=['session_id', 'elapsed_time'], inplace=True)\n    df['d_time'] = df['elapsed_time'].diff(1)\n    df['d_time'].fillna(0, inplace=True)\n    df['delt_time'] = df['d_time'].clip(0, 103000)\n    df['delt_time_next'] = df['delt_time'].shift(-1)\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:48:46.503408Z","iopub.execute_input":"2023-06-18T20:48:46.503645Z","iopub.status.idle":"2023-06-18T20:48:46.510044Z","shell.execute_reply.started":"2023-06-18T20:48:46.503621Z","shell.execute_reply":"2023-06-18T20:48:46.508896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train, kol_f):\n    global kol_col, kol_col_max\n    kol_col = 9\n    kol_col_max = 11+kol_f*2\n    col = [i for i in range(0,kol_col_max)]\n    new_train = pd.DataFrame(index=train['session_id'].unique(), columns=col, dtype=np.float16)  \n    new_train[10] = new_train.index # \"session_id\"    \n\n    new_train[0] = train.groupby(['session_id'])['d_time'].quantile(q=0.3)\n    new_train[1] = train.groupby(['session_id'])['d_time'].quantile(q=0.8)\n    new_train[2] = train.groupby(['session_id'])['d_time'].quantile(q=0.5)\n    new_train[3] = train.groupby(['session_id'])['d_time'].quantile(q=0.65)\n    new_train[4] = train.groupby(['session_id'])['hover_duration'].agg('mean')\n    new_train[5] = train.groupby(['session_id'])['hover_duration'].agg('std')    \n    new_train[6] = new_train[10].apply(lambda x: int(str(x)[:2])).astype(np.uint8) # \"year\"\n    new_train[7] = new_train[10].apply(lambda x: int(str(x)[2:4])+1).astype(np.uint8) # \"month\"\n    new_train[8] = new_train[10].apply(lambda x: int(str(x)[4:6])).astype(np.uint8) # \"day\"\n    new_train[9] = new_train[10].apply(lambda x: int(str(x)[6:8])).astype(np.uint8) + new_train[10].apply(lambda x: int(str(x)[8:10])).astype(np.uint8)/60\n    new_train[10] = 0\n    new_train = new_train.fillna(-1)\n    \n    return new_train","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:48:46.512577Z","iopub.execute_input":"2023-06-18T20:48:46.513108Z","iopub.status.idle":"2023-06-18T20:48:46.534073Z","shell.execute_reply.started":"2023-06-18T20:48:46.513079Z","shell.execute_reply":"2023-06-18T20:48:46.532580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_next_t(row_f, new_train, train, gran_1, gran_2, i):\n    global kol_col\n    kol_col +=1\n    col1 = row_f['col1']\n    val1 = row_f['val1']\n    maska = (train[col1] == val1)\n    if row_f['kol_col'] == 1:       \n        new_train[kol_col] = train[maska].groupby(['session_id'])['delt_time_next'].sum()\n        if gran_1:\n            kol_col +=1\n            new_train[kol_col] = train[maska].groupby(['session_id'])['delt_time'].mean()\n        if gran_2:\n            kol_col +=1\n            new_train[kol_col] = train[maska].groupby(['session_id'])['index'].count()          \n    elif row_f['kol_col'] == 2: \n        col2 = row_f['col2']\n        val2 = row_f['val2']\n        maska = maska & (train[col2] == val2)        \n        new_train[kol_col] = train[maska].groupby(['session_id'])['delt_time_next'].sum()\n        if gran_1:\n            kol_col +=1\n            new_train[kol_col] = train[maska].groupby(['session_id'])['delt_time'].mean()\n        if gran_2:\n            kol_col +=1\n            new_train[kol_col] = train[maska].groupby(['session_id'])['index'].count()\n    return new_train","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:48:46.536832Z","iopub.execute_input":"2023-06-18T20:48:46.537137Z","iopub.status.idle":"2023-06-18T20:48:46.554918Z","shell.execute_reply.started":"2023-06-18T20:48:46.537109Z","shell.execute_reply":"2023-06-18T20:48:46.553675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_next_t_otvet(row_f, new_train, train, gran_1, gran_2, i):\n    global kol_col\n    kol_col +=1\n    col1 = row_f['col1']\n    val1 = row_f['val1']\n    maska = (train[col1] == val1)\n    if row_f['kol_col'] == 1:      \n        new_train[kol_col] = train[maska]['delt_time_next'].sum()\n        if gran_1:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['delt_time'].mean()\n        if gran_2:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['index'].count()          \n    elif row_f['kol_col'] == 2: \n        col2 = row_f['col2']\n        val2 = row_f['val2']\n        maska = maska & (train[col2] == val2)        \n        new_train[kol_col] = train[maska]['delt_time_next'].sum()\n        if gran_1:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['delt_time'].mean()\n        if gran_2:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['index'].count()\n    return new_train","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:48:46.558559Z","iopub.execute_input":"2023-06-18T20:48:46.558885Z","iopub.status.idle":"2023-06-18T20:48:46.579105Z","shell.execute_reply.started":"2023-06-18T20:48:46.558852Z","shell.execute_reply":"2023-06-18T20:48:46.577924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def experiment_feature_next_t_otvet(row_f, new_train, train, gran_1, gran_2, i):\n    global kol_col\n    kol_col +=1\n    if row_f['kol_col'] == 1: \n        maska = train[row_f['col1']] == row_f['val1']\n        new_train[kol_col] = train[maska]['delt_time_next'].sum()\n        if gran_1:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['delt_time'].mean()\n        if gran_2:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['index'].count()          \n    elif row_f['kol_col'] == 2: \n        col2 = row_f['col2']\n        val2 = row_f['val2']\n        maska = (train[col1] == val1) & (train[col2] == val2)        \n        new_train[kol_col] = train[maska]['delt_time_next'].sum()\n        if gran_1:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['delt_time'].mean()\n        if gran_2:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['index'].count()\n    return new_train","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:48:46.583000Z","iopub.execute_input":"2023-06-18T20:48:46.583350Z","iopub.status.idle":"2023-06-18T20:48:46.604721Z","shell.execute_reply.started":"2023-06-18T20:48:46.583317Z","shell.execute_reply":"2023-06-18T20:48:46.603832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_quest_otvet(new_train, train, quest, kol_f):\n    global kol_col\n    kol_col = 9\n    g1 = 0.7 \n    g2 = 0.3 \n\n    feature_q = feature_df[feature_df['quest'] == quest].copy()\n    feature_q.reset_index(drop=True, inplace=True)\n    \n    gran1 = round(kol_f * g1)\n    gran2 = round(kol_f * g2)    \n    for i in range(0, kol_f):         \n        row_f = feature_q.loc[i]\n        new_train = feature_next_t_otvet(row_f, new_train, train, i < gran1, i <  gran2, i) \n    col = [i for i in range(0,kol_col+1)]\n    return new_train[col]","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:48:46.611031Z","iopub.execute_input":"2023-06-18T20:48:46.612906Z","iopub.status.idle":"2023-06-18T20:48:46.622831Z","shell.execute_reply.started":"2023-06-18T20:48:46.612869Z","shell.execute_reply":"2023-06-18T20:48:46.621981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer_new(new_train, train, feature_q, kol_f):\n    g1 = 0.7 \n    g2 = 0.3 \n    gran1 = round(kol_f * g1)\n    gran2 = round(kol_f * g2)    \n    for i in range(0, kol_f): \n        row_f = feature_q.loc[i]       \n        new_train = feature_next_t(row_f, new_train, train, i < gran1, i <  gran2, i)         \n    return new_train","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:48:46.625737Z","iopub.execute_input":"2023-06-18T20:48:46.626825Z","iopub.status.idle":"2023-06-18T20:48:46.643174Z","shell.execute_reply.started":"2023-06-18T20:48:46.626787Z","shell.execute_reply":"2023-06-18T20:48:46.642041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_quest(new_train, train, quest, kol_f):\n    global kol_col\n    kol_col = 9\n    feature_q = feature_df[feature_df['quest'] == quest].copy()\n    feature_q.reset_index(drop=True, inplace=True)\n    new_train = feature_engineer_new(new_train, train, feature_q, kol_f)\n    col = [i for i in range(0,kol_col+1)]\n    return new_train[col]","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:48:46.644439Z","iopub.execute_input":"2023-06-18T20:48:46.644949Z","iopub.status.idle":"2023-06-18T20:48:46.659515Z","shell.execute_reply.started":"2023-06-18T20:48:46.644915Z","shell.execute_reply":"2023-06-18T20:48:46.658730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model(old_train, quests, models, list_kol_f):\n    \n    kol_quest = len(quests)\n    # ITERATE THRU QUESTIONS\n    for q in quests:\n        print('### quest ', q, end='')\n        new_train = feature_engineer(old_train, list_kol_f[q])\n        train_x = feature_quest(new_train, old_train, q, list_kol_f[q])\n        print (' ---- ', 'train_q.shape = ', train_x.shape)\n           \n        # TRAIN DATA\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q==q].set_index('session').loc[train_users]\n\n        # TRAIN MODEL \n\n        model = CatBoostClassifier(\n            n_estimators = ctb_params[q]['n_estimators'],\n            learning_rate= ctb_params[q]['learning_rate'],\n            depth = 6\n        )\n        \n        model.fit(train_x.astype('float32'), train_y['correct'], verbose=False)\n\n        # SAVE MODEL, PREDICT VALID OOF\n        models[f'{q}'] = model\n    print('***')\n    \n    return models","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:48:46.660781Z","iopub.execute_input":"2023-06-18T20:48:46.661212Z","iopub.status.idle":"2023-06-18T20:48:46.672104Z","shell.execute_reply.started":"2023-06-18T20:48:46.661181Z","shell.execute_reply":"2023-06-18T20:48:46.671188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = {}\nbest_threshold = 0.63","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:48:46.673686Z","iopub.execute_input":"2023-06-18T20:48:46.674480Z","iopub.status.idle":"2023-06-18T20:48:46.688408Z","shell.execute_reply.started":"2023-06-18T20:48:46.674447Z","shell.execute_reply":"2023-06-18T20:48:46.687288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_kol_f = {\n    1:140,3:110,\n    4:110, 5:220, 6:120, 7:110, 8:110, 9:100, 10:120, 11:120,\n    14: 110, 15:160, 16:105, 17:140             \n             }","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:48:46.689700Z","iopub.execute_input":"2023-06-18T20:48:46.690437Z","iopub.status.idle":"2023-06-18T20:48:46.699487Z","shell.execute_reply.started":"2023-06-18T20:48:46.690407Z","shell.execute_reply":"2023-06-18T20:48:46.698708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df0_4 = pd.read_csv('/kaggle/input/featur/train_0_4t.csv', dtype=dtypes) \n# kol_lvl = (df0_4 .groupby(['session_id'])['level'].agg('nunique') < 5)\n# list_session = kol_lvl[kol_lvl].index\n# df0_4  = df0_4 [~df0_4 ['session_id'].isin(list_session)]\n# df0_4 = delt_time_def(df0_4)\n\n# quests_0_4 = [1, 3] \n# # list_kol_f = {1:140,3:110}\n\n# models = create_model(df0_4, quests_0_4, models, list_kol_f)\n# del df0_4","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:48:46.700683Z","iopub.execute_input":"2023-06-18T20:48:46.701123Z","iopub.status.idle":"2023-06-18T20:48:46.712542Z","shell.execute_reply.started":"2023-06-18T20:48:46.701095Z","shell.execute_reply":"2023-06-18T20:48:46.711719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df5_12 = pd.read_csv('/kaggle/input/featur/train_5_12t.csv', dtype=dtypes)\n# kol_lvl = (df5_12.groupby(['session_id'])['level'].agg('nunique') < 8)\n# list_session = kol_lvl[kol_lvl].index\n# df5_12 = df5_12[~df5_12['session_id'].isin(list_session)]\n# df5_12 = delt_time_def(df5_12)\n# quests_5_12 = [4, 5, 6, 7, 8, 9, 10, 11] \n\n# # list_kol_f = {4:110, 5:220, 6:120, 7:110, 8:110, 9:100, 10:120, 11:120}\n\n# models = create_model(df5_12, quests_5_12, models, list_kol_f)\n# del df5_12","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:48:46.713917Z","iopub.execute_input":"2023-06-18T20:48:46.714432Z","iopub.status.idle":"2023-06-18T20:48:46.725949Z","shell.execute_reply.started":"2023-06-18T20:48:46.714400Z","shell.execute_reply":"2023-06-18T20:48:46.725122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df13_22 = pd.read_csv('/kaggle/input/featur/train_13_22t.csv', dtype=dtypes) \n# kol_lvl = (df13_22 .groupby(['session_id'])['level'].agg('nunique') < 10)\n# list_session = kol_lvl[kol_lvl].index\n# df13_22  = df13_22 [~df13_22 ['session_id'].isin(list_session)]\n# df13_22 = delt_time_def(df13_22)\n\n# quests_13_22 = [14, 15, 16, 17] \n# # list_kol_f = {14: 110, 15:160, 16:105, 17:140}\n\n# models = create_model(df13_22, quests_13_22, models, list_kol_f)\n# del df13_22","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:48:46.727334Z","iopub.execute_input":"2023-06-18T20:48:46.727863Z","iopub.status.idle":"2023-06-18T20:48:46.742750Z","shell.execute_reply.started":"2023-06-18T20:48:46.727828Z","shell.execute_reply":"2023-06-18T20:48:46.741675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Packages import for Pytorch lifestream to run w/o Internet and embedding inference","metadata":{}},{"cell_type":"code","source":"# !pip install -U pytorch-lightning --no-index --find-links=file:///kaggle/input/pytorch-lightning-195\n!pip install -U nvidia-cuda-runtime-cu11 --no-index --find-links=file:///kaggle/input/nvidia-cuda-runtime-cu11-11799\n!pip install -U nvidia-cuda-nvrtc-cu11 --no-index --find-links=file:///kaggle/input/nvidia-cuda-nvrtc-cu11-11799\n!pip install -U nvidia-cublas-cu11 --no-index --find-links=file:///kaggle/input/nvidia-cublas-cu11\n!pip install -U nvidia-cudnn-cu11 --no-index --find-links=file:///kaggle/input/nvidia-cudnn-cu11-85096\n!pip install -U torch --no-index --find-links=file:///kaggle/input/torch-1131\n!pip install -U torchvision --no-index --find-links=file:///kaggle/input/torchvision-0141","metadata":{"_kg_hide-output":true,"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-06-18T20:48:46.744154Z","iopub.execute_input":"2023-06-18T20:48:46.744695Z","iopub.status.idle":"2023-06-18T20:50:49.184132Z","shell.execute_reply.started":"2023-06-18T20:48:46.744638Z","shell.execute_reply":"2023-06-18T20:50:49.183329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -U antlr4_python3_runtime --no-index --find-links=file:///kaggle/input/antlr4-python3-runtime-4932","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-18T20:50:49.186640Z","iopub.execute_input":"2023-06-18T20:50:49.186979Z","iopub.status.idle":"2023-06-18T20:50:58.232110Z","shell.execute_reply.started":"2023-06-18T20:50:49.186946Z","shell.execute_reply":"2023-06-18T20:50:58.231361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install -U pyarrow --no-index --find-links=file:///kaggle/input/pyarrow-1200-cp37","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-18T20:50:58.233058Z","iopub.execute_input":"2023-06-18T20:50:58.233320Z","iopub.status.idle":"2023-06-18T20:50:58.236889Z","shell.execute_reply.started":"2023-06-18T20:50:58.233289Z","shell.execute_reply":"2023-06-18T20:50:58.236328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -U omegaconf --no-index --find-links=file:///kaggle/input/omegaconf-2301","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-18T20:50:58.237610Z","iopub.execute_input":"2023-06-18T20:50:58.237866Z","iopub.status.idle":"2023-06-18T20:51:07.447872Z","shell.execute_reply.started":"2023-06-18T20:50:58.237837Z","shell.execute_reply":"2023-06-18T20:51:07.447131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -U wheel --no-index --find-links=file:///kaggle/input/wheel-0400/","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-18T20:51:07.448807Z","iopub.execute_input":"2023-06-18T20:51:07.449049Z","iopub.status.idle":"2023-06-18T20:51:26.559889Z","shell.execute_reply.started":"2023-06-18T20:51:07.449021Z","shell.execute_reply":"2023-06-18T20:51:26.559178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %cd /kaggle/working\n# !mkdir hydra-core-132\n# !cp -r /kaggle/input/hydra-core-132/hydra-core-1.3.2 hydra-core-132\n# %cd /kaggle/working/hydra-core-132/hydra-core-1.3.2\n# !python3 setup.py build && python3 setup.py install\n!pip install -U hydra-core --no-index --find-links=file:///kaggle/input/hydra-core-132whl/","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-18T20:51:26.561078Z","iopub.execute_input":"2023-06-18T20:51:26.561333Z","iopub.status.idle":"2023-06-18T20:51:35.628641Z","shell.execute_reply.started":"2023-06-18T20:51:26.561296Z","shell.execute_reply":"2023-06-18T20:51:35.627930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %cd /kaggle/working\n# !mkdir protobuf-3201\n# !cp -r /kaggle/input/protobuf-3201/protobuf-3.20.1 protobuf-3201\n# %cd /kaggle/working/protobuf-3201/protobuf-3.20.1\n# !python3 setup.py build && python3 setup.py install\n!pip install -U protobuf --no-index --find-links=file:///kaggle/input/protobuf-3201whl/","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-18T20:51:35.629580Z","iopub.execute_input":"2023-06-18T20:51:35.629840Z","iopub.status.idle":"2023-06-18T20:51:44.477019Z","shell.execute_reply.started":"2023-06-18T20:51:35.629813Z","shell.execute_reply":"2023-06-18T20:51:44.476213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %cd /kaggle/working\n# !mkdir pydeprecate-032\n# !cp -r /kaggle/input/pydeprecate-032/pyDeprecate-0.3.2 pydeprecate-032\n# %cd /kaggle/working/pydeprecate-032/pyDeprecate-0.3.2\n# !python3 setup.py build && python3 setup.py install\n!pip install -U pydeprecate --no-index --find-links=file:///kaggle/input/pydeprecate-032whl/","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-18T20:51:44.478044Z","iopub.execute_input":"2023-06-18T20:51:44.478330Z","iopub.status.idle":"2023-06-18T20:51:53.372571Z","shell.execute_reply.started":"2023-06-18T20:51:44.478286Z","shell.execute_reply":"2023-06-18T20:51:53.371358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %cd /kaggle/working\n# !mkdir certifi-202357\n# !cp -r /kaggle/input/certifi-202357/certifi-2023.5.7 certifi-202357\n# %cd /kaggle/working/certifi-202357/certifi-2023.5.7\n# !python3 setup.py build && python3 setup.py install\n!pip install -U certifi --no-index --find-links=file:///kaggle/input/certifi-202357whl/","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-18T20:51:53.373704Z","iopub.execute_input":"2023-06-18T20:51:53.373998Z","iopub.status.idle":"2023-06-18T20:52:03.421864Z","shell.execute_reply.started":"2023-06-18T20:51:53.373966Z","shell.execute_reply":"2023-06-18T20:52:03.420993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip uninstall pytorch-lightning -y\n!pip install -U pytorch-lightning --no-index --find-links=file:///kaggle/input/pytorch-lightning-160","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-18T20:52:03.426360Z","iopub.execute_input":"2023-06-18T20:52:03.427284Z","iopub.status.idle":"2023-06-18T20:52:14.614568Z","shell.execute_reply.started":"2023-06-18T20:52:03.427246Z","shell.execute_reply":"2023-06-18T20:52:14.613825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # %cd /kaggle/working\n# # !mkdir pytorch-lifestream-0.5.2\n# !cp -r /kaggle/input/pytorch-lifestream-052/pytorch-lifestream-0.5.2/* /kaggle/working\n# # %cd /kaggle/working/pytorch-lifestream-0.5.2/pytorch-lifestream-0.5.2\n# !python3 setup.py build && python3 setup.py install","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:52:14.615517Z","iopub.execute_input":"2023-06-18T20:52:14.615796Z","iopub.status.idle":"2023-06-18T20:52:14.620567Z","shell.execute_reply.started":"2023-06-18T20:52:14.615764Z","shell.execute_reply":"2023-06-18T20:52:14.619752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\nimport os\nimport sys\n\nstandard_dir = os.getcwd()\n\nshutil.copytree('/kaggle/input/pytorch-lifestream-052/pytorch-lifestream-0.5.2', '/kaggle/working/tmp')\n\nmodule_dir = '/kaggle/working/tmp/'  # Update the module directory accordingly\n\n# Check if the module directory exists\nif os.path.isdir(module_dir):\n    module_path = os.path.join(module_dir)  # Path to the module directory\n\n    # Add the module directory to sys.path temporarily\n    sys.path.insert(0, module_path)\n\n    # Import the package and perform any necessary operations\n    import ptls\n\n    # Remove the module directory from sys.path\n    sys.path.remove(module_path)\n\nelse:\n    print(f\"Module directory '{module_dir}' not found.\")\n\nos.chdir(standard_dir)","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:52:14.621536Z","iopub.execute_input":"2023-06-18T20:52:14.621866Z","iopub.status.idle":"2023-06-18T20:52:16.477196Z","shell.execute_reply.started":"2023-06-18T20:52:14.621833Z","shell.execute_reply":"2023-06-18T20:52:16.476455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -U pytorch-lightning --no-index --find-links=file:///kaggle/input/pytorch-lightning-195","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-18T20:52:16.478126Z","iopub.execute_input":"2023-06-18T20:52:16.478522Z","iopub.status.idle":"2023-06-18T20:52:26.150622Z","shell.execute_reply.started":"2023-06-18T20:52:16.478496Z","shell.execute_reply":"2023-06-18T20:52:26.149826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %cd /kaggle/working","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:52:26.152009Z","iopub.execute_input":"2023-06-18T20:52:26.152294Z","iopub.status.idle":"2023-06-18T20:52:26.156795Z","shell.execute_reply.started":"2023-06-18T20:52:26.152260Z","shell.execute_reply":"2023-06-18T20:52:26.155684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install pytorch-lifestream\n# !pip install \"torch<2\"\n# !pip install -U \"pytorch-lightning<2\"\n# !pip install -U \"torchvision<0.15.1\"\n\n!pip freeze | grep torch\n\nimport pickle\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom tqdm import tqdm\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import GroupKFold, KFold\nfrom catboost import CatBoostClassifier, Pool\nimport matplotlib.pyplot as plt\nimport warnings\nfrom itertools import combinations\nimport math\nwarnings.filterwarnings('ignore')\npd.set_option(\"display.max_columns\", None)\npd.set_option(\"display.max_rows\", 200)\n\nfrom ptls.preprocessing import PandasDataPreprocessor\nimport torch\nfrom ptls.frames.supervised import SequenceToTarget\nfrom ptls.nn import TrxEncoder, RnnSeqEncoder\nimport pytorch_lightning as pl\nfrom ptls.data_load.datasets import inference_data_loader\n\nmax_height = 1261.7737454550663 \nmin_height = -1992.3545688360275 \nmax_width = 543.6164243795992 \nmin_width = -918.1623490877204\nnum_px = 8\nnum_py = 6\n\ndef get_patch_index(x, y, min_image_height, max_image_height, min_image_width, max_image_width, num_patches_height = 8, num_patches_width = 6):\n    if np.isnan(x):\n        return num_px * num_py + 1\n    if np.isnan(y):\n        return num_px * num_py + 1\n    patch_height = (max_image_height - min_image_height + 1) / num_patches_height\n    patch_width = (max_image_width - min_image_width + 1) / num_patches_width\n    patch_x = (x - min_image_height) // patch_height\n    patch_y = (y - min_image_width) // patch_width\n    patch_index = patch_x * num_patches_width + patch_y + 1\n    if patch_index < 3:\n        patch_index = 3\n    if patch_index > 49:\n        patch_index = 49\n    return patch_index","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:52:26.158019Z","iopub.execute_input":"2023-06-18T20:52:26.158321Z","iopub.status.idle":"2023-06-18T20:52:39.108476Z","shell.execute_reply.started":"2023-06-18T20:52:26.158292Z","shell.execute_reply":"2023-06-18T20:52:39.107229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_emb_df(test, model):\n    \n    test = test[['session_id', 'level_group', 'level', 'event_name', 'elapsed_time', 'room_coor_x', 'room_coor_y']]\n    test.loc[test['room_coor_x'] > max_height, 'room_coor_x'] = max_height\n    test.loc[test['room_coor_x'] < min_height, 'room_coor_x'] = min_height\n    test.loc[test['room_coor_y'] > max_width, 'room_coor_y'] = max_width\n    test.loc[test['room_coor_x'] < min_width, 'room_coor_y'] = min_width\n    gps = []\n    for _, session in test.groupby('session_id'):\n        for _, gp in session.groupby('level_group'):\n            gp['patch'] = gp.apply(lambda row : get_patch_index(row['room_coor_x'], row['room_coor_y'], min_height, max_height, min_width, max_width, num_px, num_py), axis = 1)\n            gps.append(gp)\n\n    df_patshed = pd.concat(gps)\n#     df_patshed['time_diff'] = df_patshed.groupby('session_id')['elapsed_time'].diff()\n#     df_patshed['time_diff'] = df_patshed['time_diff'].fillna(0)\n#     df_patshed = df_patshed[df_patshed.time_diff >= 0]\n#     df_patshed['event_name_factorized'] =  pd.factorize(df_patshed['event_name'])[0]\n#     df_patshed['elapsed_time_to_datetime'] = pd.to_datetime(df_patshed['elapsed_time'])\n\n    df_patshed[\"sid_level\"] = df_patshed[\"session_id\"].astype(str) + ' ' + df_patshed[\"level_group\"].astype(str)\n#     df_patshed = df_patshed.drop([\"session_id\", \"level\", \"level_group\", \"event_name\", 'elapsed_time_to_datetime'], axis = 1)\n    df_patshed = df_patshed.drop([\"session_id\", \"level\", \"level_group\", \"event_name\"], axis = 1)\n\n    preprocessor = PandasDataPreprocessor(\n        col_id= 'sid_level',\n        col_event_time='elapsed_time',\n        event_time_transformation='none',\n        cols_category= ['patch'],\n        return_records=True,\n    )\n    df_patshed = preprocessor.fit_transform(df_patshed)\n    df_patshed = sorted(df_patshed, key=lambda x: x['sid_level'])\n\n#     trx_encoder_params = dict(\n#       embeddings_noise=0.003,\n#       embeddings={\n#           'patch': {'in': 55, 'out': 55},\n#       },\n#     )\n\n#     seq_encoder = RnnSeqEncoder(\n#       trx_encoder=TrxEncoder(**trx_encoder_params),\n#       hidden_size=256,\n#       type='gru',\n#     )\n\n#     seq_encoder.load_state_dict(torch.load('/kaggle/input/jw-emb-lg/jw-emb-lg.pt'))\n#     model = SequenceToTarget(seq_encoder)\n    model.eval();\n\n#     trainer = pl.Trainer(gpus=1 if torch.cuda.is_available() else 0)\n    trainer = pl.Trainer(accelerator=\"cpu\", enable_progress_bar=False)\n    test_dl = inference_data_loader(df_patshed, num_workers=0, batch_size=256)\n    test_embeds = torch.vstack(trainer.predict(model, test_dl))\n\n    test_df_patched = pd.DataFrame(data=test_embeds, columns=[f'embed_{i}' for i in range(test_embeds.shape[1])])\n    test_df_patched['sid_level'] = [x['sid_level'] for x in df_patshed]\n\n    emeb_df = pd.DataFrame(np.arange(len(test_df_patched)))\n    emeb_df['embeddings'] = test_df_patched.iloc[:, :-1].values.tolist()\n    emeb_df = emeb_df.drop([0], axis=1)\n    emeb_df['sid_level'] = test_df_patched['sid_level']\n\n    emeb_df['session_id'], emeb_df['level_gp'] = emeb_df['sid_level'].str.split(' ', 1).str\n    emeb_df = emeb_df.drop(['sid_level', 'level_gp'], axis = 1).set_index('session_id')\n\n    return emeb_df","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-06-18T20:52:39.110262Z","iopub.execute_input":"2023-06-18T20:52:39.110605Z","iopub.status.idle":"2023-06-18T20:52:39.132348Z","shell.execute_reply.started":"2023-06-18T20:52:39.110568Z","shell.execute_reply":"2023-06-18T20:52:39.130860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Infer Test Data**","metadata":{}},{"cell_type":"code","source":"quests_0_4 = [1, 3] \nquests_5_12 = [4, 5, 6, 7, 8, 9, 10, 11] \nquests_13_22 = [14, 15, 16, 17]","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:52:39.134333Z","iopub.execute_input":"2023-06-18T20:52:39.134882Z","iopub.status.idle":"2023-06-18T20:52:39.149064Z","shell.execute_reply.started":"2023-06-18T20:52:39.134852Z","shell.execute_reply":"2023-06-18T20:52:39.147844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model2 = {}\ndir_meta = '/kaggle/input/catboost-0699-emb/catboost-0.699-emb/' \nfor q in quests_0_4 + quests_5_12 + quests_13_22:\n    model2[f'{q}'] = CatBoostClassifier().load_model(dir_meta +f'q{q}.cbm')","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:52:39.150775Z","iopub.execute_input":"2023-06-18T20:52:39.151077Z","iopub.status.idle":"2023-06-18T20:52:44.541540Z","shell.execute_reply.started":"2023-06-18T20:52:39.151049Z","shell.execute_reply":"2023-06-18T20:52:44.540358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model2","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:52:44.545488Z","iopub.execute_input":"2023-06-18T20:52:44.545823Z","iopub.status.idle":"2023-06-18T20:52:44.552541Z","shell.execute_reply.started":"2023-06-18T20:52:44.545793Z","shell.execute_reply":"2023-06-18T20:52:44.551816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trx_encoder_params = dict(\n  embeddings_noise=0.003,\n  embeddings={\n      'patch': {'in': 55, 'out': 55},\n  },\n)\n\nseq_encoder = RnnSeqEncoder(\n  trx_encoder=TrxEncoder(**trx_encoder_params),\n  hidden_size=256,\n  type='gru',\n)\n\nseq_encoder.load_state_dict(torch.load('/kaggle/input/jw-emb-lg/jw-emb-lg.pt'))\nmodel_emb = SequenceToTarget(seq_encoder)","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:52:44.553556Z","iopub.execute_input":"2023-06-18T20:52:44.554877Z","iopub.status.idle":"2023-06-18T20:52:44.602175Z","shell.execute_reply.started":"2023-06-18T20:52:44.554825Z","shell.execute_reply":"2023-06-18T20:52:44.601093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jo_wilder\n\ntry:\n    jo_wilder.make_env.__called__ = False\n    env.__called__ = False\n    type(env)._state = type(type(env)._state).__dict__['INIT']\nexcept:\n    pass\n\nenv = jo_wilder.make_env()\niter_test = env.iter_test()    ","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:52:44.603210Z","iopub.execute_input":"2023-06-18T20:52:44.603468Z","iopub.status.idle":"2023-06-18T20:52:44.624440Z","shell.execute_reply.started":"2023-06-18T20:52:44.603441Z","shell.execute_reply":"2023-06-18T20:52:44.623670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_final = {}\ng_end4 = 0\ng_end5 = 0\n\nlist_q = {'0-4':quests_0_4, '5-12':quests_5_12, '13-22':quests_13_22}\nfor (test, sam_sub) in iter_test:\n    sam_sub['question'] = [int(label.split('_')[1][1:]) for label in sam_sub['session_id']]    \n    grp = test.level_group.values[0]\n    \n    session_id = test.session_id.values[0]\n    sid = sam_sub.iloc[0, :]['session_id'].split('_')[0]\n\n    sam_sub['correct'] = 1\n    sam_sub.loc[sam_sub.question.isin([5, 8, 10, 13, 15]), 'correct'] = 0  \n    old_train = delt_time_def(test[test.level_group == grp])\n    \n    test_clean = test.dropna(subset=['elapsed_time', 'room_coor_x', 'room_coor_y'])\n    emb_df = create_emb_df(test_clean[test_clean.level_group == grp], model_emb)\n    emb_df.index.name = None\n    emb_df.index = emb_df.index.astype('int64')\n       \n    for q in list_q[grp]:\n\n        new_train = feature_engineer(old_train, list_kol_f[q])\n        new_train = feature_quest_otvet(new_train, old_train, q, list_kol_f[q])\n        \n        df = new_train.merge(emb_df, left_index = True, right_index = True)\n        \n        model = model2[f'{q}']\n        test_pool = Pool(df, embedding_features=['embeddings'])\n        pred = model.predict_proba(test_pool)[0,1]\n        \n        mask = sam_sub.question == q \n        x = int(pred>0.63)\n        sam_sub.loc[mask,'correct'] = x \n\n    sam_sub = sam_sub[['session_id', 'correct']]      \n    env.predict(sam_sub)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-18T20:52:44.625507Z","iopub.execute_input":"2023-06-18T20:52:44.626092Z","iopub.status.idle":"2023-06-18T20:52:54.352878Z","shell.execute_reply.started":"2023-06-18T20:52:44.626049Z","shell.execute_reply":"2023-06-18T20:52:54.352069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA submission.csv","metadata":{"papermill":{"duration":0.011427,"end_time":"2023-02-07T01:02:45.502331","exception":false,"start_time":"2023-02-07T01:02:45.490904","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# sub = pd.read_csv('submission.csv')\n# print(sub.shape)\n# sub.mean()","metadata":{"papermill":{"duration":0.027432,"end_time":"2023-02-07T01:02:45.541022","exception":false,"start_time":"2023-02-07T01:02:45.51359","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T20:52:54.354736Z","iopub.execute_input":"2023-06-18T20:52:54.355697Z","iopub.status.idle":"2023-06-18T20:52:54.358986Z","shell.execute_reply.started":"2023-06-18T20:52:54.355634Z","shell.execute_reply":"2023-06-18T20:52:54.358303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub.head(54)","metadata":{"execution":{"iopub.status.busy":"2023-06-18T20:52:54.360172Z","iopub.execute_input":"2023-06-18T20:52:54.360684Z","iopub.status.idle":"2023-06-18T20:52:54.393944Z","shell.execute_reply.started":"2023-06-18T20:52:54.360637Z","shell.execute_reply":"2023-06-18T20:52:54.392684Z"},"trusted":true},"execution_count":null,"outputs":[]}]}