{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-13T08:27:39.370655Z","iopub.execute_input":"2023-03-13T08:27:39.371129Z","iopub.status.idle":"2023-03-13T08:27:39.378056Z","shell.execute_reply.started":"2023-03-13T08:27:39.371091Z","shell.execute_reply":"2023-03-13T08:27:39.376847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-13T08:27:39.379931Z","iopub.execute_input":"2023-03-13T08:27:39.380553Z","iopub.status.idle":"2023-03-13T08:28:16.442377Z","shell.execute_reply.started":"2023-03-13T08:27:39.380480Z","shell.execute_reply":"2023-03-13T08:28:16.441309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ntarget_df['session'] = target_df.session_id.apply(lambda x: int(x.split('_')[0]) )\ntarget_df['q'] = target_df.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\ntarget_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-13T08:28:16.443760Z","iopub.execute_input":"2023-03-13T08:28:16.444353Z","iopub.status.idle":"2023-03-13T08:28:17.054668Z","shell.execute_reply.started":"2023-03-13T08:28:16.444318Z","shell.execute_reply":"2023-03-13T08:28:17.053718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['word_cnt'] = train_df['text'].astype(str).apply(lambda x: 0 if x=='nan' else len(x.split()))\ntmp = pd.DataFrame(train_df.groupby(['session_id','level_group'])['word_cnt'].agg('sum'))\ndisplay(tmp)\ndel tmp\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-13T08:28:17.056845Z","iopub.execute_input":"2023-03-13T08:28:17.057401Z","iopub.status.idle":"2023-03-13T08:28:28.326778Z","shell.execute_reply.started":"2023-03-13T08:28:17.057367Z","shell.execute_reply":"2023-03-13T08:28:28.325596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATS   = ['event_name', 'fqid', 'room_fqid', 'text']\nNUMS   = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\nEVENTS = ['navigate_click','person_click','cutscene_click','object_click',\n          'map_hover','notification_click','map_click','observation_click', 'checkpoint']\n#DIALOGS = ['who', 'when', 'where', 'what', 'who', 'how']\nDIALOGS = ['that', 'this', 'it']","metadata":{"execution":{"iopub.status.busy":"2023-03-13T08:28:28.328726Z","iopub.execute_input":"2023-03-13T08:28:28.329312Z","iopub.status.idle":"2023-03-13T08:28:28.335394Z","shell.execute_reply.started":"2023-03-13T08:28:28.329273Z","shell.execute_reply":"2023-03-13T08:28:28.334192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train):\n    dfs = []\n    for c in CATS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)    \n    for c in EVENTS: \n        train[c] = (train.event_name == c).astype('int8')\n    for c in EVENTS + ['elapsed_time']:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('sum')\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n    \n    for c in DIALOGS:\n        tmp = train.groupby(['session_id','level_group'])['text'].apply(lambda x: (x.str.count(c)).sum())\n        tmp.name = c\n        dfs.append(tmp)\n    \n    #train['word_cnt'] = train['text'].astype(str).apply(lambda x: 0 if x=='nan' else len(x.split()))\n    #tmp = train.groupby(['session_id','level_group'])['word_cnt'].agg('sum')\n    #tmp.name = \"word_cnt\"\n    #dfs.append(tmp)\n    \n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)    \n    df = df.reset_index()\n    df = df.set_index('session_id')\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-03-13T08:28:28.337057Z","iopub.execute_input":"2023-03-13T08:28:28.337487Z","iopub.status.idle":"2023-03-13T08:28:28.352106Z","shell.execute_reply.started":"2023-03-13T08:28:28.337453Z","shell.execute_reply":"2023-03-13T08:28:28.350806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf = feature_engineer(train_df)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-13T08:28:28.353698Z","iopub.execute_input":"2023-03-13T08:28:28.354182Z","iopub.status.idle":"2023-03-13T08:30:29.191655Z","shell.execute_reply.started":"2023-03-13T08:28:28.354141Z","shell.execute_reply":"2023-03-13T08:30:29.190240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_df\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-13T08:30:29.193697Z","iopub.execute_input":"2023-03-13T08:30:29.194091Z","iopub.status.idle":"2023-03-13T08:30:29.200272Z","shell.execute_reply.started":"2023-03-13T08:30:29.194053Z","shell.execute_reply":"2023-03-13T08:30:29.199091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\nALL_USERS = df.index.unique()","metadata":{"execution":{"iopub.status.busy":"2023-03-13T08:30:29.203942Z","iopub.execute_input":"2023-03-13T08:30:29.204927Z","iopub.status.idle":"2023-03-13T08:30:29.219533Z","shell.execute_reply.started":"2023-03-13T08:30:29.204881Z","shell.execute_reply":"2023-03-13T08:30:29.217570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import GroupKFold\nfrom lightgbm import LGBMRegressor\nN_FOLDS = 3\n\ngkf = GroupKFold(n_splits=N_FOLDS)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nmodels = {}\n\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    for t in range(1,19):\n                \n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = target_df.loc[target_df.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = target_df.loc[target_df.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL\n        model = LGBMRegressor(learning_rate=0.027, \\\n                    num_leaves=15, \\\n                    n_estimators=200, \\\n                    min_child_samples=20, \\\n                    boosting_type='gbdt',\n                    subsample_for_bin=1000,\n                    max_depth=-1,\n                    colsample_bytree=0.8)\n        model.fit(train_x[FEATURES].astype('float32'), train_y['correct'])\n                \n        models[f'{i}_{grp}_{t}'] = model\n        oof.loc[valid_users, t-1] = model.predict(valid_x[FEATURES])\n        \n    print()","metadata":{"execution":{"iopub.status.busy":"2023-03-13T08:30:29.221277Z","iopub.execute_input":"2023-03-13T08:30:29.221669Z","iopub.status.idle":"2023-03-13T08:31:03.543462Z","shell.execute_reply.started":"2023-03-13T08:30:29.221634Z","shell.execute_reply":"2023-03-13T08:31:03.542317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"true = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = target_df.loc[target_df.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-03-13T08:31:03.545084Z","iopub.execute_input":"2023-03-13T08:31:03.545721Z","iopub.status.idle":"2023-03-13T08:31:03.624297Z","shell.execute_reply.started":"2023-03-13T08:31:03.545681Z","shell.execute_reply":"2023-03-13T08:31:03.623252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nfrom sklearn.metrics import f1_score\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-03-13T08:31:03.625636Z","iopub.execute_input":"2023-03-13T08:31:03.626187Z","iopub.status.idle":"2023-03-13T08:31:06.792639Z","shell.execute_reply.started":"2023-03-13T08:31:03.626153Z","shell.execute_reply":"2023-03-13T08:31:06.791361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del target_df, df, oof, true\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-13T08:31:06.794066Z","iopub.execute_input":"2023-03-13T08:31:06.794642Z","iopub.status.idle":"2023-03-13T08:31:06.799796Z","shell.execute_reply.started":"2023-03-13T08:31:06.794603Z","shell.execute_reply":"2023-03-13T08:31:06.798588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-03-13T08:31:06.801295Z","iopub.execute_input":"2023-03-13T08:31:06.801731Z","iopub.status.idle":"2023-03-13T08:31:06.831848Z","shell.execute_reply.started":"2023-03-13T08:31:06.801695Z","shell.execute_reply":"2023-03-13T08:31:06.830607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(qid, grp, test):\n    val = 0\n    for fold in range(N_FOLDS):\n        val += models[f'{fold}_{grp}_{qid}'].predict(test[FEATURES])[0]\n    return val > best_threshold*N_FOLDS","metadata":{"execution":{"iopub.status.busy":"2023-03-13T08:31:06.833362Z","iopub.execute_input":"2023-03-13T08:31:06.833759Z","iopub.status.idle":"2023-03-13T08:31:06.842998Z","shell.execute_reply.started":"2023-03-13T08:31:06.833723Z","shell.execute_reply":"2023-03-13T08:31:06.841632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for (sample_submission, test) in iter_test:\n       \n    df = feature_engineer(test)\n        \n    grp = test.level_group.values[0]\n    sample_submission['qid'] = sample_submission['session_id'].apply(lambda x: x.split(\"_\")[1][1:]).astype(int)\n    sample_submission['correct'] = sample_submission['qid'].apply(lambda x: predict(x, grp, df)).astype(int)\n    del sample_submission['qid']\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-03-13T08:31:06.844681Z","iopub.execute_input":"2023-03-13T08:31:06.845025Z","iopub.status.idle":"2023-03-13T08:31:07.787287Z","shell.execute_reply.started":"2023-03-13T08:31:06.844992Z","shell.execute_reply":"2023-03-13T08:31:07.786088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"submission.csv\")\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-13T08:31:07.788952Z","iopub.execute_input":"2023-03-13T08:31:07.789305Z","iopub.status.idle":"2023-03-13T08:31:07.800404Z","shell.execute_reply.started":"2023-03-13T08:31:07.789272Z","shell.execute_reply":"2023-03-13T08:31:07.799564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}