{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np\nfrom catboost import CatBoostClassifier\nimport pickle\nimport sys","metadata":{"execution":{"iopub.status.busy":"2023-06-10T04:11:58.491324Z","iopub.execute_input":"2023-06-10T04:11:58.491765Z","iopub.status.idle":"2023-06-10T04:11:58.497354Z","shell.execute_reply.started":"2023-06-10T04:11:58.491731Z","shell.execute_reply":"2023-06-10T04:11:58.496186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Train Data and Labels","metadata":{"papermill":{"duration":0.004542,"end_time":"2023-02-07T00:59:59.189777","exception":false,"start_time":"2023-02-07T00:59:59.185235","status":"completed"},"tags":[]}},{"cell_type":"code","source":"dtypes = {\"session_id\": 'int64',\n          \"index\": np.int16,\n          \"elapsed_time\": np.int32,\n          \"event_name\": 'category',\n          \"name\": 'category',\n          \"level\": np.int8,\n          \"page\": np.float16,\n          \"room_coor_x\": np.float16,\n          \"room_coor_y\": np.float16,\n          \"screen_coor_x\": np.float16,\n          \"screen_coor_y\": np.float16,\n          \"hover_duration\": np.float32,\n          \"text\": 'category',\n          \"fqid\": 'category',\n          \"room_fqid\": 'category',\n          \"text_fqid\": 'category',\n          \"fullscreen\": np.int8,\n          \"hq\": np.int8,\n          \"music\": np.int8,\n          \"level_group\": 'category'\n          }\nuse_col = list(dtypes.keys())","metadata":{"papermill":{"duration":59.284316,"end_time":"2023-02-07T01:00:58.478743","exception":false,"start_time":"2023-02-07T00:59:59.194427","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-10T04:11:58.697511Z","iopub.execute_input":"2023-06-10T04:11:58.697931Z","iopub.status.idle":"2023-06-10T04:11:58.707246Z","shell.execute_reply.started":"2023-06-10T04:11:58.697892Z","shell.execute_reply":"2023-06-10T04:11:58.706196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training = False\nlocal = False","metadata":{"execution":{"iopub.status.busy":"2023-06-10T04:11:58.743262Z","iopub.execute_input":"2023-06-10T04:11:58.744388Z","iopub.status.idle":"2023-06-10T04:11:58.748741Z","shell.execute_reply.started":"2023-06-10T04:11:58.744342Z","shell.execute_reply":"2023-06-10T04:11:58.747710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_dir = './kaggle/input/' if local else '/kaggle/input/'\ntargets = pd.read_csv(f'{input_dir}/predict-student-performance-from-game-play/train_labels.csv')\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]))\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]))\nfeature_df = pd.read_csv(f'{input_dir}/featur/feature_sort.csv')\n","metadata":{"papermill":{"duration":0.598155,"end_time":"2023-02-07T01:00:59.082015","exception":false,"start_time":"2023-02-07T01:00:58.48386","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-10T04:11:58.750543Z","iopub.execute_input":"2023-06-10T04:11:58.751148Z","iopub.status.idle":"2023-06-10T04:12:00.262386Z","shell.execute_reply.started":"2023-06-10T04:11:58.751112Z","shell.execute_reply":"2023-06-10T04:12:00.261152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineer","metadata":{"papermill":{"duration":0.005196,"end_time":"2023-02-07T01:00:59.092865","exception":false,"start_time":"2023-02-07T01:00:59.087669","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def delt_time_def(df):\n    df.sort_values(by=['session_id', 'elapsed_time'], inplace=True)\n    df['d_time'] = df['elapsed_time'].diff(1)\n    df['d_time'].fillna(0, inplace=True)\n    df['delt_time'] = df['d_time'].clip(0, 103000)\n    df['delt_time_next'] = df['delt_time'].shift(-1)\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-06-10T04:12:00.264077Z","iopub.execute_input":"2023-06-10T04:12:00.264573Z","iopub.status.idle":"2023-06-10T04:12:00.271813Z","shell.execute_reply.started":"2023-06-10T04:12:00.264536Z","shell.execute_reply":"2023-06-10T04:12:00.270609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train, kol_f):\n    global kol_col, kol_col_max\n    kol_col = 9\n    kol_col_max = 11+kol_f*2\n    col = [i for i in range(0,kol_col_max)]\n    new_train = pd.DataFrame(index=train['session_id'].unique(), columns=col, dtype=np.float16)  \n    new_train[10] = new_train.index # \"session_id\"    \n\n    new_train[0] = train.groupby(['session_id'])['d_time'].quantile(q=0.3)\n    new_train[1] = train.groupby(['session_id'])['d_time'].quantile(q=0.8)\n    new_train[2] = train.groupby(['session_id'])['d_time'].quantile(q=0.5)\n    new_train[3] = train.groupby(['session_id'])['d_time'].quantile(q=0.65)\n    new_train[4] = train.groupby(['session_id'])['hover_duration'].agg('mean')\n    new_train[5] = train.groupby(['session_id'])['hover_duration'].agg('std')    \n    new_train[6] = new_train[10].apply(lambda x: int(str(x)[:2])).astype(np.uint8) # \"year\"\n    new_train[7] = new_train[10].apply(lambda x: int(str(x)[2:4])+1).astype(np.uint8) # \"month\"\n    new_train[8] = new_train[10].apply(lambda x: int(str(x)[4:6])).astype(np.uint8) # \"day\"\n    new_train[9] = new_train[10].apply(lambda x: int(str(x)[6:8])).astype(np.uint8) + new_train[10].apply(lambda x: int(str(x)[8:10])).astype(np.uint8)/60\n    new_train[10] = 0\n    new_train = new_train.fillna(-1)\n    \n    return new_train","metadata":{"execution":{"iopub.status.busy":"2023-06-10T04:12:00.274553Z","iopub.execute_input":"2023-06-10T04:12:00.274910Z","iopub.status.idle":"2023-06-10T04:12:00.292181Z","shell.execute_reply.started":"2023-06-10T04:12:00.274875Z","shell.execute_reply":"2023-06-10T04:12:00.291077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_next_t(row_f, new_train, train, gran_1, gran_2, i):\n    global kol_col\n    kol_col +=1\n    col1 = row_f['col1']\n    val1 = row_f['val1']\n    maska = (train[col1] == val1)\n    if row_f['kol_col'] == 1:       \n        new_train[kol_col] = train[maska].groupby(['session_id'])['delt_time_next'].sum()\n        if gran_1:\n            kol_col +=1\n            new_train[kol_col] = train[maska].groupby(['session_id'])['delt_time'].mean()\n        if gran_2:\n            kol_col +=1\n            new_train[kol_col] = train[maska].groupby(['session_id'])['index'].count()          \n    elif row_f['kol_col'] == 2: \n        col2 = row_f['col2']\n        val2 = row_f['val2']\n        maska = maska & (train[col2] == val2)        \n        new_train[kol_col] = train[maska].groupby(['session_id'])['delt_time_next'].sum()\n        if gran_1:\n            kol_col +=1\n            new_train[kol_col] = train[maska].groupby(['session_id'])['delt_time'].mean()\n        if gran_2:\n            kol_col +=1\n            new_train[kol_col] = train[maska].groupby(['session_id'])['index'].count()\n    return new_train","metadata":{"execution":{"iopub.status.busy":"2023-06-10T04:12:00.293645Z","iopub.execute_input":"2023-06-10T04:12:00.294001Z","iopub.status.idle":"2023-06-10T04:12:00.311865Z","shell.execute_reply.started":"2023-06-10T04:12:00.293942Z","shell.execute_reply":"2023-06-10T04:12:00.310590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_next_t_otvet(row_f, new_train, train, gran_1, gran_2, i):\n    global kol_col\n    kol_col +=1\n    col1 = row_f['col1']\n    val1 = row_f['val1']\n    maska = (train[col1] == val1)\n    if row_f['kol_col'] == 1:      \n        new_train[kol_col] = train[maska]['delt_time_next'].sum()\n        if gran_1:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['delt_time'].mean()\n        if gran_2:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['index'].count()          \n    elif row_f['kol_col'] == 2: \n        col2 = row_f['col2']\n        val2 = row_f['val2']\n        maska = maska & (train[col2] == val2)        \n        new_train[kol_col] = train[maska]['delt_time_next'].sum()\n        if gran_1:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['delt_time'].mean()\n        if gran_2:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['index'].count()\n    return new_train","metadata":{"execution":{"iopub.status.busy":"2023-06-10T04:12:00.313536Z","iopub.execute_input":"2023-06-10T04:12:00.313879Z","iopub.status.idle":"2023-06-10T04:12:00.331104Z","shell.execute_reply.started":"2023-06-10T04:12:00.313848Z","shell.execute_reply":"2023-06-10T04:12:00.329716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def experiment_feature_next_t_otvet(row_f, new_train, train, gran_1, gran_2, i):\n    global kol_col\n    kol_col +=1\n    if row_f['kol_col'] == 1: \n        maska = train[row_f['col1']] == row_f['val1']\n        new_train[kol_col] = train[maska]['delt_time_next'].sum()\n        if gran_1:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['delt_time'].mean()\n        if gran_2:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['index'].count()          \n    elif row_f['kol_col'] == 2: \n        col2 = row_f['col2']\n        val2 = row_f['val2']\n        maska = (train[col1] == val1) & (train[col2] == val2)        \n        new_train[kol_col] = train[maska]['delt_time_next'].sum()\n        if gran_1:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['delt_time'].mean()\n        if gran_2:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['index'].count()\n    return new_train","metadata":{"execution":{"iopub.status.busy":"2023-06-10T04:12:00.332661Z","iopub.execute_input":"2023-06-10T04:12:00.333019Z","iopub.status.idle":"2023-06-10T04:12:00.350493Z","shell.execute_reply.started":"2023-06-10T04:12:00.332987Z","shell.execute_reply":"2023-06-10T04:12:00.349315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_quest_otvet(new_train, train, quest, kol_f):\n    global kol_col\n    kol_col = 9\n    g1 = 0.7 \n    g2 = 0.3 \n\n    feature_q = feature_df[feature_df['quest'] == quest].copy()\n    feature_q.reset_index(drop=True, inplace=True)\n    \n    gran1 = round(kol_f * g1)\n    gran2 = round(kol_f * g2)    \n    for i in range(0, kol_f):         \n        row_f = feature_q.loc[i]\n        new_train = feature_next_t_otvet(row_f, new_train, train, i < gran1, i <  gran2, i) \n    col = [i for i in range(0,kol_col+1)]\n    return new_train[col]","metadata":{"execution":{"iopub.status.busy":"2023-06-10T04:12:00.352305Z","iopub.execute_input":"2023-06-10T04:12:00.352666Z","iopub.status.idle":"2023-06-10T04:12:00.365168Z","shell.execute_reply.started":"2023-06-10T04:12:00.352635Z","shell.execute_reply":"2023-06-10T04:12:00.364183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer_new(new_train, train, feature_q, kol_f):\n    g1 = 0.7 \n    g2 = 0.3 \n    gran1 = round(kol_f * g1)\n    gran2 = round(kol_f * g2)    \n    for i in range(0, kol_f): \n        row_f = feature_q.loc[i]       \n        new_train = feature_next_t(row_f, new_train, train, i < gran1, i <  gran2, i)         \n    return new_train","metadata":{"execution":{"iopub.status.busy":"2023-06-10T04:12:00.368591Z","iopub.execute_input":"2023-06-10T04:12:00.368922Z","iopub.status.idle":"2023-06-10T04:12:00.382741Z","shell.execute_reply.started":"2023-06-10T04:12:00.368892Z","shell.execute_reply":"2023-06-10T04:12:00.381621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_quest(new_train, train, quest, kol_f):\n    global kol_col\n    kol_col = 9\n    feature_q = feature_df[feature_df['quest'] == quest].copy()\n    feature_q.reset_index(drop=True, inplace=True)\n    new_train = feature_engineer_new(new_train, train, feature_q, kol_f)\n    col = [i for i in range(0,kol_col+1)]\n    return new_train[col]","metadata":{"execution":{"iopub.status.busy":"2023-06-10T04:12:00.384174Z","iopub.execute_input":"2023-06-10T04:12:00.385155Z","iopub.status.idle":"2023-06-10T04:12:00.394415Z","shell.execute_reply.started":"2023-06-10T04:12:00.385116Z","shell.execute_reply":"2023-06-10T04:12:00.393437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model(old_train, quests, models, list_kol_f):\n\n    kol_quest = len(quests)\n    # ITERATE THRU QUESTIONS\n    for q in quests:\n        print('### quest ', q, end='')\n        new_train = feature_engineer(old_train, list_kol_f[q])\n        train_x = feature_quest(new_train, old_train, q, list_kol_f[q])\n        print(' ---- ', 'train_q.shape = ', train_x.shape)\n\n        # TRAIN DATA\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q == q].set_index(\n            'session').loc[train_users]\n\n        # TRAIN MODEL\n\n        model = CatBoostClassifier(\n            n_estimators = 300,\n            learning_rate= 0.045,\n            depth = 6\n        )\n\n        model.fit(train_x.astype('float32'), train_y['correct'], verbose=False)\n\n        # SAVE MODEL, PREDICT VALID OOF\n        models[f'{q}'] = model\n    print('***')\n\n    return models\n","metadata":{"execution":{"iopub.status.busy":"2023-06-10T04:12:00.396072Z","iopub.execute_input":"2023-06-10T04:12:00.396393Z","iopub.status.idle":"2023-06-10T04:12:00.408454Z","shell.execute_reply.started":"2023-06-10T04:12:00.396363Z","shell.execute_reply":"2023-06-10T04:12:00.407355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nmodels = {}\nlist_kol_f = {\n    1: 140, 3: 110,\n    4: 120, 5: 220, 6: 130, 7: 110, 8: 110, 9: 100, 10: 140, 11: 120,\n    14: 160, 15: 160, 16: 130, 17: 140\n}\nquests_0_4 = [1, 3]\nquests_5_12 = [4, 5, 6, 7, 8, 9, 10, 11]\nquests_13_22 = [14, 15, 16, 17]\nall_quests = np.concatenate((quests_0_4, quests_5_12, quests_13_22))\n","metadata":{"execution":{"iopub.status.busy":"2023-06-10T04:12:00.409877Z","iopub.execute_input":"2023-06-10T04:12:00.410689Z","iopub.status.idle":"2023-06-10T04:12:00.427582Z","shell.execute_reply.started":"2023-06-10T04:12:00.410640Z","shell.execute_reply":"2023-06-10T04:12:00.426402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport sys\n\nif training:\n    df0_4 = pd.read_csv(f'{input_dir}/featur/train_0_4t.csv', dtype=dtypes)\n    kol_lvl = (df0_4 .groupby(['session_id'])['level'].agg('nunique') < 5)\n    list_session = kol_lvl[kol_lvl].index\n    df0_4 = df0_4[~df0_4['session_id'].isin(list_session)]\n    df0_4 = delt_time_def(df0_4)\n\n    models = create_model(df0_4, quests_0_4, models, list_kol_f)\n    del df0_4\n\n    df5_12 = pd.read_csv(f'{input_dir}/featur/train_5_12t.csv', dtype=dtypes)\n    kol_lvl = (df5_12.groupby(['session_id'])['level'].agg('nunique') < 8)\n    list_session = kol_lvl[kol_lvl].index\n    df5_12 = df5_12[~df5_12['session_id'].isin(list_session)]\n    df5_12 = delt_time_def(df5_12)\n\n    models = create_model(df5_12, quests_5_12, models, list_kol_f)\n    del df5_12\n\n    df13_22 = pd.read_csv(\n        '/kaggle/input/featur/train_13_22t.csv', dtype=dtypes)\n    kol_lvl = (df13_22 .groupby(['session_id'])['level'].agg('nunique') < 10)\n    list_session = kol_lvl[kol_lvl].index\n    df13_22 = df13_22[~df13_22['session_id'].isin(list_session)]\n    df13_22 = delt_time_def(df13_22)\n\n    models = create_model(df13_22, quests_13_22, models, list_kol_f)\n\n    # Saving a Model\n    output = './catboost-models'\n    if not os.path.exists(output):\n        os.mkdir(output)\n    for q in all_quests:\n        models[str(q)].save_model(f'{output}/cat_model_{q}.bin')\n","metadata":{"execution":{"iopub.status.busy":"2023-06-10T04:12:00.429148Z","iopub.execute_input":"2023-06-10T04:12:00.429472Z","iopub.status.idle":"2023-06-10T04:12:00.442920Z","shell.execute_reply.started":"2023-06-10T04:12:00.429442Z","shell.execute_reply":"2023-06-10T04:12:00.442128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Model Reading\nif not training:\n    dir = f'{input_dir}catboost-mix-1/'\n    for q in all_quests:\n        models[str(q)] = CatBoostClassifier().load_model(dir+f'cat_model_{q}.bin')","metadata":{"execution":{"iopub.status.busy":"2023-06-10T04:12:00.444378Z","iopub.execute_input":"2023-06-10T04:12:00.444983Z","iopub.status.idle":"2023-06-10T04:12:00.607487Z","shell.execute_reply.started":"2023-06-10T04:12:00.444925Z","shell.execute_reply":"2023-06-10T04:12:00.606253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Infer Test Data**","metadata":{}},{"cell_type":"code","source":"\nimport jo_wilder\ntry:\n    jo_wilder.make_env.__called__ = False\n    env.__called__ = False\n    type(env)._state = type(type(env)._state).__dict__['INIT']\nexcept:\n    pass\n\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n","metadata":{"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-10T04:12:00.608700Z","iopub.execute_input":"2023-06-10T04:12:00.609044Z","iopub.status.idle":"2023-06-10T04:12:00.620922Z","shell.execute_reply.started":"2023-06-10T04:12:00.609013Z","shell.execute_reply":"2023-06-10T04:12:00.619610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nlist_q = {'0-4': quests_0_4, '5-12': quests_5_12, '13-22': quests_13_22}\nthreholds = {2: 0.55, 18: 0.56, 3: 0.57, 12: 0.6, 4: 0.61, 8: 0.64}\nbest_threshold = 0.63\n\nfor (test, sam_sub) in iter_test:\n    sam_sub['question'] = [int(label.split('_')[1][1:])\n                           for label in sam_sub['session_id']]\n    grp = test.level_group.values[0]\n    sam_sub['correct'] = 1\n    sam_sub.loc[sam_sub.question.isin([5, 8, 10, 13, 15]), 'correct'] = 0\n    old_train = delt_time_def(test[test.level_group == grp])\n    for q in list_q[grp]:\n\n        new_train = feature_engineer(old_train, list_kol_f[q])\n        new_train = feature_quest_otvet(new_train, old_train, q, list_kol_f[q])\n\n        clf = models[f'{q}']\n        p = clf.predict_proba(new_train.astype('float32'))[:, 1]\n\n        mask = sam_sub.question == q\n        threhold = threholds.get(q, best_threshold)\n        x = int(p[0] > best_threshold)\n        sam_sub.loc[mask, 'correct'] = x\n\n    sam_sub = sam_sub[['session_id', 'correct']]\n    env.predict(sam_sub)\n","metadata":{"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-10T04:12:00.623911Z","iopub.execute_input":"2023-06-10T04:12:00.624819Z","iopub.status.idle":"2023-06-10T04:12:12.789427Z","shell.execute_reply.started":"2023-06-10T04:12:00.624783Z","shell.execute_reply":"2023-06-10T04:12:12.788428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA submission.csv","metadata":{"papermill":{"duration":0.011427,"end_time":"2023-02-07T01:02:45.502331","exception":false,"start_time":"2023-02-07T01:02:45.490904","status":"completed"},"tags":[]}},{"cell_type":"code","source":"df = pd.read_csv('submission.csv')\nprint(df.shape)\ndf.head(60)","metadata":{"papermill":{"duration":0.027432,"end_time":"2023-02-07T01:02:45.541022","exception":false,"start_time":"2023-02-07T01:02:45.51359","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-10T04:12:12.790656Z","iopub.execute_input":"2023-06-10T04:12:12.791019Z","iopub.status.idle":"2023-06-10T04:12:12.812259Z","shell.execute_reply.started":"2023-06-10T04:12:12.790987Z","shell.execute_reply":"2023-06-10T04:12:12.811198Z"},"trusted":true},"execution_count":null,"outputs":[]}]}