{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np\nfrom catboost import CatBoostClassifier\nimport pickle\nimport sys","metadata":{"execution":{"iopub.status.busy":"2023-06-27T01:24:57.783364Z","iopub.execute_input":"2023-06-27T01:24:57.784182Z","iopub.status.idle":"2023-06-27T01:24:59.322937Z","shell.execute_reply.started":"2023-06-27T01:24:57.784082Z","shell.execute_reply":"2023-06-27T01:24:59.322009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Train Data and Labels","metadata":{"papermill":{"duration":0.004542,"end_time":"2023-02-07T00:59:59.189777","exception":false,"start_time":"2023-02-07T00:59:59.185235","status":"completed"},"tags":[]}},{"cell_type":"code","source":"dtypes = {\"session_id\": 'int64',\n          \"index\": np.int16,\n          \"elapsed_time\": np.int32,\n          \"event_name\": 'category',\n          \"name\": 'category',\n          \"level\": np.int8,\n          \"page\": np.float16,\n          \"room_coor_x\": np.float16,\n          \"room_coor_y\": np.float16,\n          \"screen_coor_x\": np.float16,\n          \"screen_coor_y\": np.float16,\n          \"hover_duration\": np.float32,\n          \"text\": 'category',\n          \"fqid\": 'category',\n          \"room_fqid\": 'category',\n          \"text_fqid\": 'category',\n          \"fullscreen\": np.int8,\n          \"hq\": np.int8,\n          \"music\": np.int8,\n          \"level_group\": 'category'\n          }\nuse_col = ['session_id', 'index', 'elapsed_time', 'event_name', 'name', 'level', 'page',\n           'room_coor_x', 'room_coor_y', 'hover_duration', 'text', 'fqid', 'room_fqid', 'text_fqid', 'level_group']","metadata":{"papermill":{"duration":59.284316,"end_time":"2023-02-07T01:00:58.478743","exception":false,"start_time":"2023-02-07T00:59:59.194427","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-27T01:24:59.324984Z","iopub.execute_input":"2023-06-27T01:24:59.325655Z","iopub.status.idle":"2023-06-27T01:24:59.334069Z","shell.execute_reply.started":"2023-06-27T01:24:59.325612Z","shell.execute_reply":"2023-06-27T01:24:59.333119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]) )\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nprint( targets.shape )\ntargets.head()","metadata":{"papermill":{"duration":0.598155,"end_time":"2023-02-07T01:00:59.082015","exception":false,"start_time":"2023-02-07T01:00:58.48386","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-27T01:24:59.336447Z","iopub.execute_input":"2023-06-27T01:24:59.337370Z","iopub.status.idle":"2023-06-27T01:25:01.139363Z","shell.execute_reply.started":"2023-06-27T01:24:59.337335Z","shell.execute_reply":"2023-06-27T01:25:01.137781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_df = pd.read_csv('/kaggle/input/featur/feature_sort.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-27T01:25:01.147960Z","iopub.execute_input":"2023-06-27T01:25:01.151833Z","iopub.status.idle":"2023-06-27T01:25:01.363609Z","shell.execute_reply.started":"2023-06-27T01:25:01.151771Z","shell.execute_reply":"2023-06-27T01:25:01.362497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineer","metadata":{"papermill":{"duration":0.005196,"end_time":"2023-02-07T01:00:59.092865","exception":false,"start_time":"2023-02-07T01:00:59.087669","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def delt_time_def(df):\n    df.sort_values(by=['session_id', 'elapsed_time'], inplace=True)\n    df['d_time'] = df['elapsed_time'].diff(1)\n    df['d_time'].fillna(0, inplace=True)\n    df['delt_time'] = df['d_time'].clip(0, 103000)\n    df['delt_time_next'] = df['delt_time'].shift(-1)\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-06-27T01:25:01.368098Z","iopub.execute_input":"2023-06-27T01:25:01.370919Z","iopub.status.idle":"2023-06-27T01:25:01.380372Z","shell.execute_reply.started":"2023-06-27T01:25:01.370875Z","shell.execute_reply":"2023-06-27T01:25:01.378452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train, kol_f):\n    global kol_col, kol_col_max\n    kol_col = 20\n    kol_col_max = 11+kol_f*2\n    col = [i for i in range(0,kol_col_max)]\n    new_train = pd.DataFrame(index=train['session_id'].unique(), columns=col, dtype=np.float16)  \n    # Index\n    new_train[21] = new_train.index # \"session_id\"    \n    # FE.d_time\n    new_train[0] = train.groupby(['session_id'])['d_time'].quantile(q=0.15)\n    new_train[1] = train.groupby(['session_id'])['d_time'].quantile(q=0.25)\n    new_train[2] = train.groupby(['session_id'])['d_time'].quantile(q=0.35)\n    new_train[3] = train.groupby(['session_id'])['d_time'].quantile(q=0.45)\n    new_train[4] = train.groupby(['session_id'])['d_time'].quantile(q=0.65)\n    new_train[5] = train.groupby(['session_id'])['d_time'].quantile(q=0.75)\n    new_train[6] = train.groupby(['session_id'])['d_time'].quantile(q=0.85)\n    new_train[7] = train.groupby(['session_id'])['d_time'].quantile(q=0.95)\n    # FE.hover_duration\n    new_train[8] = train.groupby(['session_id'])['hover_duration'].agg('mean')\n    new_train[9] = train.groupby(['session_id'])['hover_duration'].agg('median')\n    new_train[10] = train.groupby(['session_id'])['hover_duration'].agg('std')\n    new_train[11] = train.groupby(['session_id'])['hover_duration'].agg('min')\n    new_train[12] = train.groupby(['session_id'])['hover_duration'].agg('max')\n    q15 = lambda x: x.quantile(0.15); q15.__name__ = \"q0.15\"\n    q25 = lambda x: x.quantile(0.25); q25.__name__ = \"q0.50\"\n    q75 = lambda x: x.quantile(0.75); q75.__name__ = \"q0.75\"\n    q85 = lambda x: x.quantile(0.85); q85.__name__ = \"q0.85\"\n    new_train[13] = train.groupby(['session_id'])['hover_duration'].agg(q15)\n    new_train[14] = train.groupby(['session_id'])['hover_duration'].agg(q25)\n    new_train[15] = train.groupby(['session_id'])['hover_duration'].agg(q75)\n    new_train[16] = train.groupby(['session_id'])['hover_duration'].agg(q85)\n    # FE.dt\n    new_train[17] = new_train[21].apply(lambda x: int(str(x)[:2])).astype(np.uint8) # \"year\"\n    new_train[18] = new_train[21].apply(lambda x: int(str(x)[2:4])+1).astype(np.uint8) # \"month\"\n    new_train[19] = new_train[21].apply(lambda x: int(str(x)[4:6])).astype(np.uint8) # \"day\"\n    new_train[20] = new_train[21].apply(lambda x: int(str(x)[6:8])).astype(np.uint8) + new_train[21].apply(lambda x: int(str(x)[8:10])).astype(np.uint8)/60\n    new_train[21] = 0\n    new_train = new_train.fillna(-1)\n    \n    return new_train","metadata":{"execution":{"iopub.status.busy":"2023-06-27T01:25:01.385728Z","iopub.execute_input":"2023-06-27T01:25:01.387817Z","iopub.status.idle":"2023-06-27T01:25:01.412787Z","shell.execute_reply.started":"2023-06-27T01:25:01.387774Z","shell.execute_reply":"2023-06-27T01:25:01.411769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_next_t(row_f, new_train, train, gran_1, gran_2, i):\n    global kol_col\n    kol_col +=1\n    col1 = row_f['col1']\n    val1 = row_f['val1']\n    maska = (train[col1] == val1)\n    if row_f['kol_col'] == 1:       \n        new_train[kol_col] = train[maska].groupby(['session_id'])['delt_time_next'].sum()\n        if gran_1:\n            kol_col +=1\n            new_train[kol_col] = train[maska].groupby(['session_id'])['delt_time'].mean()\n        if gran_2:\n            kol_col +=1\n            new_train[kol_col] = train[maska].groupby(['session_id'])['index'].count()          \n    elif row_f['kol_col'] == 2: \n        col2 = row_f['col2']\n        val2 = row_f['val2']\n        maska = maska & (train[col2] == val2)        \n        new_train[kol_col] = train[maska].groupby(['session_id'])['delt_time_next'].sum()\n        if gran_1:\n            kol_col +=1\n            new_train[kol_col] = train[maska].groupby(['session_id'])['delt_time'].mean()\n        if gran_2:\n            kol_col +=1\n            new_train[kol_col] = train[maska].groupby(['session_id'])['index'].count()\n    return new_train","metadata":{"execution":{"iopub.status.busy":"2023-06-27T01:25:01.414129Z","iopub.execute_input":"2023-06-27T01:25:01.414777Z","iopub.status.idle":"2023-06-27T01:25:01.434861Z","shell.execute_reply.started":"2023-06-27T01:25:01.414743Z","shell.execute_reply":"2023-06-27T01:25:01.433588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_next_t_otvet(row_f, new_train, train, gran_1, gran_2, i):\n    global kol_col\n    kol_col +=1\n    col1 = row_f['col1']\n    val1 = row_f['val1']\n    maska = (train[col1] == val1)\n    if row_f['kol_col'] == 1:      \n        new_train[kol_col] = train[maska]['delt_time_next'].sum()\n        if gran_1:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['delt_time'].mean()\n        if gran_2:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['index'].count()          \n    elif row_f['kol_col'] == 2: \n        col2 = row_f['col2']\n        val2 = row_f['val2']\n        maska = maska & (train[col2] == val2)        \n        new_train[kol_col] = train[maska]['delt_time_next'].sum()\n        if gran_1:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['delt_time'].mean()\n        if gran_2:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['index'].count()\n    return new_train","metadata":{"execution":{"iopub.status.busy":"2023-06-27T01:25:01.439902Z","iopub.execute_input":"2023-06-27T01:25:01.442525Z","iopub.status.idle":"2023-06-27T01:25:01.455394Z","shell.execute_reply.started":"2023-06-27T01:25:01.442469Z","shell.execute_reply":"2023-06-27T01:25:01.454413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def experiment_feature_next_t_otvet(row_f, new_train, train, gran_1, gran_2, i):\n    global kol_col\n    kol_col +=1\n    if row_f['kol_col'] == 1: \n        maska = train[row_f['col1']] == row_f['val1']\n        new_train[kol_col] = train[maska]['delt_time_next'].sum()\n        if gran_1:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['delt_time'].mean()\n        if gran_2:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['index'].count()          \n    elif row_f['kol_col'] == 2: \n        col2 = row_f['col2']\n        val2 = row_f['val2']\n        maska = (train[col1] == val1) & (train[col2] == val2)        \n        new_train[kol_col] = train[maska]['delt_time_next'].sum()\n        if gran_1:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['delt_time'].mean()\n        if gran_2:\n            kol_col +=1\n            new_train[kol_col] = train[maska]['index'].count()\n    return new_train","metadata":{"execution":{"iopub.status.busy":"2023-06-27T01:25:01.457323Z","iopub.execute_input":"2023-06-27T01:25:01.457667Z","iopub.status.idle":"2023-06-27T01:25:01.471230Z","shell.execute_reply.started":"2023-06-27T01:25:01.457636Z","shell.execute_reply":"2023-06-27T01:25:01.470309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_quest_otvet(new_train, train, quest, kol_f):\n    global kol_col\n    kol_col = 20\n    g1 = 0.7 \n    g2 = 0.3 \n\n    feature_q = feature_df[feature_df['quest'] == quest].copy()\n    feature_q.reset_index(drop=True, inplace=True)\n    \n    gran1 = round(kol_f * g1)\n    gran2 = round(kol_f * g2)    \n    for i in range(0, kol_f):         \n        row_f = feature_q.loc[i]\n        new_train = feature_next_t_otvet(row_f, new_train, train, i < gran1, i <  gran2, i) \n    col = [i for i in range(0,kol_col+1)]\n    return new_train[col]","metadata":{"execution":{"iopub.status.busy":"2023-06-27T01:25:01.475319Z","iopub.execute_input":"2023-06-27T01:25:01.475948Z","iopub.status.idle":"2023-06-27T01:25:01.485438Z","shell.execute_reply.started":"2023-06-27T01:25:01.475902Z","shell.execute_reply":"2023-06-27T01:25:01.483951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer_new(new_train, train, feature_q, kol_f):\n    g1 = 0.7 \n    g2 = 0.3 \n    gran1 = round(kol_f * g1)\n    gran2 = round(kol_f * g2)    \n    for i in range(0, kol_f): \n        row_f = feature_q.loc[i]       \n        new_train = feature_next_t(row_f, new_train, train, i < gran1, i <  gran2, i)         \n    return new_train","metadata":{"execution":{"iopub.status.busy":"2023-06-27T01:25:01.487110Z","iopub.execute_input":"2023-06-27T01:25:01.488175Z","iopub.status.idle":"2023-06-27T01:25:01.501223Z","shell.execute_reply.started":"2023-06-27T01:25:01.488102Z","shell.execute_reply":"2023-06-27T01:25:01.500119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_quest(new_train, train, quest, kol_f):\n    global kol_col\n    kol_col = 20\n    feature_q = feature_df[feature_df['quest'] == quest].copy()\n    feature_q.reset_index(drop=True, inplace=True)\n    new_train = feature_engineer_new(new_train, train, feature_q, kol_f)\n    col = [i for i in range(0,kol_col+1)]\n    return new_train[col]","metadata":{"execution":{"iopub.status.busy":"2023-06-27T01:25:01.502941Z","iopub.execute_input":"2023-06-27T01:25:01.503373Z","iopub.status.idle":"2023-06-27T01:25:01.517307Z","shell.execute_reply.started":"2023-06-27T01:25:01.503326Z","shell.execute_reply":"2023-06-27T01:25:01.516258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---\n## Model Def\n---","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBClassifier\nfrom sklearn.neural_network import MLPClassifier\ndef create_model(old_train, quests, models, list_kol_f):\n    \n    kol_quest = len(quests)\n    # ITERATE THRU QUESTIONS\n    for q in quests:\n        print('### quest ', q, end='')\n        new_train = feature_engineer(old_train, list_kol_f[q])\n        train_x = feature_quest(new_train, old_train, q, list_kol_f[q])\n        print (' ---- ', 'train_q.shape = ', train_x.shape)\n           \n        # TRAIN DATA\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q==q].set_index('session').loc[train_users]\n\n        # ---------------------------------------------------------------- #\n        # Model.CatBoostClassifier\n        # ---------------------------------------------------------------- #\n        print(\"[INFO]Proc CatBoostClassifier\")\n        cat_model = CatBoostClassifier(\n            n_estimators = 300,\n            learning_rate= 0.045,\n            depth = 6\n        )\n        cat_model.fit(train_x.astype('float32'), train_y['correct'], verbose=False)\n        models[f'{q}_catboost'] = cat_model\n        # ---------------------------------------------------------------- #\n        # Model.XGBClassifier\n        # ---------------------------------------------------------------- #\n        print(\"[INFO]Proc XGBClassifier\")\n        xgb_model = XGBClassifier(\n            learning_rate=0.02,\n            max_depth= 4,\n            n_estimators= 800,\n            gamma = 0.01, \n            alpha = 0.05,  \n        )\n        xgb_model.fit(train_x.astype('float32'), train_y['correct'])\n        models[f'{q}_xgboost'] = xgb_model\n        # ---------------------------------------------------------------- #\n        # Model.MLP\n        # ---------------------------------------------------------------- #\n        print(\"[INFO]Proc MLP\")\n        mlp_model = MLPClassifier(max_iter=400)\n        mlp_model.fit(train_x.astype('float32').fillna(0), train_y['correct'])\n        models[f'{q}_mlp'] = mlp_model\n    \n    \n    return models","metadata":{"execution":{"iopub.status.busy":"2023-06-27T01:25:01.521177Z","iopub.execute_input":"2023-06-27T01:25:01.521821Z","iopub.status.idle":"2023-06-27T01:25:01.553490Z","shell.execute_reply.started":"2023-06-27T01:25:01.521769Z","shell.execute_reply":"2023-06-27T01:25:01.552242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = {}\nbest_threshold = 0.62","metadata":{"execution":{"iopub.status.busy":"2023-06-27T01:25:01.562752Z","iopub.execute_input":"2023-06-27T01:25:01.563119Z","iopub.status.idle":"2023-06-27T01:25:01.567739Z","shell.execute_reply.started":"2023-06-27T01:25:01.563089Z","shell.execute_reply":"2023-06-27T01:25:01.566713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_kol_f = {\n    1:140,3:110,\n    4:120, 5:220, 6:130, 7:110, 8:110, 9:100, 10:140, 11:120,\n    14: 160, 15:160, 16:130, 17:140             \n             }","metadata":{"execution":{"iopub.status.busy":"2023-06-27T01:25:01.569258Z","iopub.execute_input":"2023-06-27T01:25:01.569957Z","iopub.status.idle":"2023-06-27T01:25:01.578640Z","shell.execute_reply.started":"2023-06-27T01:25:01.569923Z","shell.execute_reply":"2023-06-27T01:25:01.577609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---\n## Train\n---","metadata":{}},{"cell_type":"code","source":"df0_4 = pd.read_csv('/kaggle/input/featur/train_0_4t.csv', dtype=dtypes) \nkol_lvl = (df0_4 .groupby(['session_id'])['level'].agg('nunique') < 5)\nlist_session = kol_lvl[kol_lvl].index\ndf0_4  = df0_4 [~df0_4 ['session_id'].isin(list_session)]\ndf0_4 = delt_time_def(df0_4)\n\nquests_0_4 = [1, 3] \n# list_kol_f = {1:140,3:110}\n\nmodels = create_model(df0_4, quests_0_4, models, list_kol_f)\ndel df0_4","metadata":{"execution":{"iopub.status.busy":"2023-06-27T01:25:01.580296Z","iopub.execute_input":"2023-06-27T01:25:01.580953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df5_12 = pd.read_csv('/kaggle/input/featur/train_5_12t.csv', dtype=dtypes)\nkol_lvl = (df5_12.groupby(['session_id'])['level'].agg('nunique') < 8)\nlist_session = kol_lvl[kol_lvl].index\ndf5_12 = df5_12[~df5_12['session_id'].isin(list_session)]\ndf5_12 = delt_time_def(df5_12)\nquests_5_12 = [4, 5, 6, 7, 8, 9, 10, 11] \n\n# list_kol_f = {4:110, 5:220, 6:120, 7:110, 8:110, 9:100, 10:140, 11:120}\n\nmodels = create_model(df5_12, quests_5_12, models, list_kol_f)\ndel df5_12","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df13_22 = pd.read_csv('/kaggle/input/featur/train_13_22t.csv', dtype=dtypes) \nkol_lvl = (df13_22 .groupby(['session_id'])['level'].agg('nunique') < 10)\nlist_session = kol_lvl[kol_lvl].index\ndf13_22  = df13_22 [~df13_22 ['session_id'].isin(list_session)]\ndf13_22 = delt_time_def(df13_22)\n\nquests_13_22 = [14, 15, 16, 17] \n# list_kol_f = {14: 160, 15:160, 16:105, 17:140}\n\nmodels = create_model(df13_22, quests_13_22, models, list_kol_f)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---\n## Save Model\n---","metadata":{}},{"cell_type":"code","source":"# Saving a Model\nimport joblib\nfor q in quests_0_4 + quests_5_12 + quests_13_22:\n    models[f\"{q}_catboost\"].save_model(f'cat_model_{q}.bin')\n    models[f\"{q}_xgboost\"].save_model(f'xgb_model_{q}.bin')\n    joblib.dump(models[f\"{q}_mlp\"], f'mlp_model_{q}.joblib')","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}