{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import time\nimport joblib, gc\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom matplotlib.colors import ListedColormap\nimport sklearn.metrics as sm\nimport seaborn as sns\nfrom scipy.stats import kstest, ks_2samp\n\nfrom sklearn.linear_model import LinearRegression,LogisticRegression,LassoCV,RidgeCV,ElasticNetCV\nfrom sklearn.metrics import accuracy_score,classification_report,confusion_matrix,f1_score\nfrom sklearn.model_selection import train_test_split,cross_val_score,cross_validate,KFold, StratifiedKFold, GridSearchCV, GroupKFold\nfrom sklearn.neighbors import KNeighborsClassifier,KNeighborsRegressor,NearestCentroid\n\nfrom sklearn.tree import DecisionTreeRegressor, DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestRegressor, RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.feature_selection import SelectFromModel\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, LabelEncoder, OneHotEncoder\nfrom sklearn.feature_selection import SelectKBest, RFE, RFECV\n\nfrom sklearn.svm import LinearSVC, SVC, SVR\nfrom sklearn.decomposition import PCA\n\nfrom sklearn.cluster import KMeans, AgglomerativeClustering\nfrom sklearn.datasets import make_blobs\nfrom scipy.cluster.hierarchy import dendrogram\n","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:21:27.932482Z","iopub.execute_input":"2023-05-11T01:21:27.933427Z","iopub.status.idle":"2023-05-11T01:21:41.053250Z","shell.execute_reply.started":"2023-05-11T01:21:27.933380Z","shell.execute_reply":"2023-05-11T01:21:41.052072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Loading the Dataset\n- Specify the datatype of each feature to reduce loading time\n- referenced [this](http://www.kaggle.com/code/gusthema/student-performance-w-tensorflow-decision-forests) thread","metadata":{}},{"cell_type":"code","source":"# specify data type and downcast each into smaller type to reduce memory usage when loading dataset\ndtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\ndata = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\", dtype = dtypes)\ndata","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:21:41.055765Z","iopub.execute_input":"2023-05-11T01:21:41.056914Z","iopub.status.idle":"2023-05-11T01:23:56.040966Z","shell.execute_reply.started":"2023-05-11T01:21:41.056875Z","shell.execute_reply":"2023-05-11T01:23:56.039925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/test.csv\", dtype = dtypes)\ntest_data","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:23:56.042463Z","iopub.execute_input":"2023-05-11T01:23:56.043088Z","iopub.status.idle":"2023-05-11T01:23:56.126002Z","shell.execute_reply.started":"2023-05-11T01:23:56.043039Z","shell.execute_reply":"2023-05-11T01:23:56.125033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Check Missing Data ","metadata":{}},{"cell_type":"code","source":"train_missing = data.isna().sum() / len(data)\ntest_missing = test_data.isna().sum() / len(test_data)\nplt.figure(figsize = (10, 4))\nplt.subplot(1, 2, 1)\nplt.bar(train_missing.index,\n        train_missing.values,\n        color=['red' if ratio >= 0.6 else 'grey' for ratio in train_missing.values])\nplt.xlabel('Feature', fontsize=12)\nplt.ylabel('Missing values ratio', fontsize=12)\nplt.title('Missing values in Train Set', fontsize=16)\nplt.xticks(rotation=90)\n# plt.legend(handles=[mpatches.Patch(color='grey'),\n#                     mpatches.Patch(color='red')], \n#            labels=['Partially missing values', 'Completely missing values'])\nplt.subplot(1, 2, 2)\nplt.bar(test_missing.index,\n        test_missing.values,\n        color=['red' if ratio >= 0.6 else 'grey' for ratio in test_missing.values])\nplt.title('Missing values in Test Set', fontsize=16)\nplt.xticks(rotation=90)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:23:56.127337Z","iopub.execute_input":"2023-05-11T01:23:56.127944Z","iopub.status.idle":"2023-05-11T01:23:57.368388Z","shell.execute_reply.started":"2023-05-11T01:23:56.127912Z","shell.execute_reply":"2023-05-11T01:23:57.366812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(data[data['fullscreen'] == 1]['fullscreen']))\nprint(len(data[data['hq'] == 1]['hq']))\nprint(len(data[data['music'] == 1]['music']))","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:23:57.372330Z","iopub.execute_input":"2023-05-11T01:23:57.373062Z","iopub.status.idle":"2023-05-11T01:23:57.474286Z","shell.execute_reply.started":"2023-05-11T01:23:57.373009Z","shell.execute_reply":"2023-05-11T01:23:57.473122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Distribution of missing values for features are roughly the same for Test set and Training set\n- We choose to drop page, hover_duration, text, and text_fqid\n- threshold of 0.6 was used; although different threshold could be used,\n- it seems unnecessary given other features are not missing that much\n- We further choose to drop fullscreen, hq, and music as they are almost non-varying(lower than 0.5% of the values are different)","metadata":{}},{"cell_type":"code","source":"# because we'll be running classifications(trees), we don't need to check\n# multicolinearlity issues between the features as trees have built-in mechanisms to drop\n# features with high multicolinearlities\n\ndrop = ['page', 'hover_duration', 'text','text_fqid','fullscreen','hq','music']\nuseful = data.columns.difference(drop)\ndata = data[useful]\n\ncategorical = ['event_name', 'name', 'fqid', 'room_fqid']\nnumerical = ['elapsed_time', 'level', 'room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y']\n\nevents = list(np.unique(data['event_name']))\n# events = ['checkpoint', 'cutscene_click', 'map_click', 'map_hover', 'navigate_click', 'notebook_click', 'notification_click',\n#          'object_click', 'object_hover', 'observation_click', 'person_click']\nnames = list(np.unique(data['name']))\n# names = ['basic', 'close', 'next', 'open', 'prev', 'undefined']","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:23:57.475622Z","iopub.execute_input":"2023-05-11T01:23:57.476028Z","iopub.status.idle":"2023-05-11T01:25:22.647409Z","shell.execute_reply.started":"2023-05-11T01:23:57.476000Z","shell.execute_reply":"2023-05-11T01:25:22.646495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# identify unique sessions('session_id') and their counts\nunique_sessions = data.groupby('session_id').session_id.agg('count')\ndisplay(unique_sessions)","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:25:22.649049Z","iopub.execute_input":"2023-05-11T01:25:22.649685Z","iopub.status.idle":"2023-05-11T01:25:23.153305Z","shell.execute_reply.started":"2023-05-11T01:25:22.649648Z","shell.execute_reply":"2023-05-11T01:25:23.152313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Load the labels <br>\nsession_id is a combination of both the session and the question number<br>\nsplit these into session id and qeustion number for the ease of computation","metadata":{}},{"cell_type":"code","source":"# load the labels (= correct value for all 18 questions for each session)\nlabels = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\")\nlabels['session'] = labels.session_id.apply(lambda x: int(x.split('_')[0]))\nlabels['q'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]))\nlabels['q'] = labels['q'].astype(int)\nlabels","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:25:23.154888Z","iopub.execute_input":"2023-05-11T01:25:23.155510Z","iopub.status.idle":"2023-05-11T01:25:24.684510Z","shell.execute_reply.started":"2023-05-11T01:25:23.155474Z","shell.execute_reply":"2023-05-11T01:25:24.683576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_unique = list(np.unique(labels['session']))\nprint(len(labels_unique)*3)\nlabels['q'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:25:24.686257Z","iopub.execute_input":"2023-05-11T01:25:24.686948Z","iopub.status.idle":"2023-05-11T01:25:24.725782Z","shell.execute_reply.started":"2023-05-11T01:25:24.686910Z","shell.execute_reply":"2023-05-11T01:25:24.724753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualize the correct answers of the labels\n\nplt.figure(figsize=(3, 3))\nplt.suptitle(\"Correct column values for the entire dataset\", fontsize=14, y=0.94)\nplot_df = labels.correct.value_counts()\nplot_df.plot(kind=\"bar\", color=['cornflowerblue', 'lightcoral'])\n\n\nplt.figure(figsize=(10, 20))\nplt.subplots_adjust(hspace=0.5, wspace=0.5)\nplt.suptitle(\"Correct column values for each question\", fontsize=14, y=0.94)\nfor n in range(1,19):\n    #print(n, str(n))\n    ax = plt.subplot(6, 3, n)\n\n    # filter df and plot ticker on the new subplot axis\n    plot_df = labels.loc[labels.q == n]\n    plot_df = plot_df.correct.value_counts()\n    plot_df.plot(ax=ax, kind=\"bar\", color=['cornflowerblue', 'lightcoral'])\n    \n    # chart formatting\n    ax.set_title(\"Question \" + str(n))\n    ax.set_xlabel(\"\")","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:25:24.727367Z","iopub.execute_input":"2023-05-11T01:25:24.727713Z","iopub.status.idle":"2023-05-11T01:25:27.551740Z","shell.execute_reply.started":"2023-05-11T01:25:24.727685Z","shell.execute_reply":"2023-05-11T01:25:27.550761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we need to make prediction for each segment;\n# the same session_id (20090312431273200) for example has 3 segments based on \n# level_group: 0-4, 5-12, 13-22\n# We need to predict for each group of the unique session_id, not on every single sample\n# Hence, group the session_id by level_group\n\ndfs = []\nfor cat in categorical:\n    tmp = data.groupby(['session_id','level_group'])[cat].agg('nunique')\n    tmp.name = tmp.name + \"_nunique\"\n    dfs.append(tmp)\nfor num in numerical:\n    tmp1 = data.groupby(['session_id', 'level_group'])[num].agg('mean')\n    dfs.append(tmp1)\nfor num in numerical:\n    tmp2 = data.groupby(['session_id', 'level_group'])[num].agg('std')\n    tmp2.name = tmp2.name + \"_std\"\n    dfs.append(tmp2)\ndata_seg = pd.concat(dfs, axis=1)\ndata_seg = data_seg.fillna(-1)\ndata_seg = data_seg.reset_index()\ndata_seg = data_seg.set_index('session_id')\n\ndata_seg","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:25:27.553287Z","iopub.execute_input":"2023-05-11T01:25:27.553662Z","iopub.status.idle":"2023-05-11T01:26:07.317196Z","shell.execute_reply.started":"2023-05-11T01:25:27.553622Z","shell.execute_reply":"2023-05-11T01:26:07.315781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_seg.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:26:07.319550Z","iopub.execute_input":"2023-05-11T01:26:07.320295Z","iopub.status.idle":"2023-05-11T01:26:07.454775Z","shell.execute_reply.started":"2023-05-11T01:26:07.320255Z","shell.execute_reply":"2023-05-11T01:26:07.453828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# final processing to extract the data that will be trained with models;\n# we'll train following 'features' on 'users' info(sample)\n\nfeatures = data_seg.columns[1:]\nusers = data_seg.index.unique()\nusers","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:26:07.456343Z","iopub.execute_input":"2023-05-11T01:26:07.456993Z","iopub.status.idle":"2023-05-11T01:26:07.468629Z","shell.execute_reply.started":"2023-05-11T01:26:07.456958Z","shell.execute_reply":"2023-05-11T01:26:07.467816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Logistic Regression\n- Behavior of the models referenced [this](https://www.kaggle.com/code/rizkykiky/baseline-logistic-regression-only-group-level) thread","metadata":{}},{"cell_type":"code","source":"folds = GroupKFold(n_splits = 5)\nprediction = pd.DataFrame(data = np.zeros((len(users), 18)), index = users)\nmodels = {}\nscores = pd.DataFrame(index = [f'FOLD_{i}' for i in range(5)], columns = list(range(1,19)))\n\nfor i, (t,v) in enumerate(folds.split(X = data_seg, groups = data_seg.index)):\n    for l in range(1,19):\n        if l <= 3: grp = '0-4'\n        if l <= 13: grp = '5-12'\n        if l <= 22: grp = '13-22'\n        xtrain = data_seg.iloc[t]\n        xtrain = xtrain.loc[xtrain.level_group == grp]\n        train_users = xtrain.index.values\n        ytrain = labels.loc[labels.q==l].set_index('session').loc[train_users]\n        \n        xval = data_seg.iloc[v]\n        xval = xval.loc[xval.level_group == grp]\n        val_users = xval.index.values\n        yval = labels.loc[labels.q == l].set_index('session').loc[val_users]\n        \n        model = LogisticRegression(random_state = 0, solver = 'saga', max_iter = 1500, n_jobs = -1, C = 0.1)\n        model.fit(xtrain[features], ytrain['correct'],)\n        \n        yhat = model.predict_proba(xval[features])[:, 1]\n        score = f1_score(yval['correct'], np.round(yhat).astype(int))\n        \n        models[f'fold{i}-level{l}'] = model\n        joblib.dump(model, f'model-fold{i}-level{l}.pkl')\n        prediction.loc[val_users, l-1] = yhat\n        scores.loc[f'FOLD_{i}',l] = score\n        del xtrain, train_users, ytrain, xval, val_users, yval, model, yhat, score\n        gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:26:07.475846Z","iopub.execute_input":"2023-05-11T01:26:07.476502Z","iopub.status.idle":"2023-05-11T01:41:21.426423Z","shell.execute_reply.started":"2023-05-11T01:26:07.476465Z","shell.execute_reply":"2023-05-11T01:41:21.425324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_logistic = scores\nscores_logistic","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:41:21.428595Z","iopub.execute_input":"2023-05-11T01:41:21.429292Z","iopub.status.idle":"2023-05-11T01:41:21.450640Z","shell.execute_reply.started":"2023-05-11T01:41:21.429256Z","shell.execute_reply":"2023-05-11T01:41:21.449321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.mean(scores_logistic)","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:41:21.452244Z","iopub.execute_input":"2023-05-11T01:41:21.452597Z","iopub.status.idle":"2023-05-11T01:41:21.478988Z","shell.execute_reply.started":"2023-05-11T01:41:21.452567Z","shell.execute_reply":"2023-05-11T01:41:21.477877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictionCopy = prediction.copy()\nfor k in range(18):\n    tmp = labels.loc[labels.q == k+1].set_index('session').loc[users]\n    predictionCopy[k] = tmp.correct.values\n    \nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.3, 0.95, 0.01):\n    preds = (prediction.values.reshape((-1)) > threshold).astype(int)\n    m = f1_score(predictionCopy.values.reshape((-1)), preds, average='macro')\n    scores.append(m)\n    thresholds.append(threshold)\n    if m > best_score:\n        best_score = m; best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:41:21.480273Z","iopub.execute_input":"2023-05-11T01:41:21.480690Z","iopub.status.idle":"2023-05-11T01:41:31.604421Z","shell.execute_reply.started":"2023-05-11T01:41:21.480661Z","shell.execute_reply":"2023-05-11T01:41:31.603077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT THRESHOLD VS. F1_SCORE\n\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='lightcoral')\nplt.scatter([best_threshold], [best_score], color='cornflowerblue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:41:31.606115Z","iopub.execute_input":"2023-05-11T01:41:31.606439Z","iopub.status.idle":"2023-05-11T01:41:31.944401Z","shell.execute_reply.started":"2023-05-11T01:41:31.606411Z","shell.execute_reply":"2023-05-11T01:41:31.943449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_per_question = {}\nfor k in range(18):\n    m = f1_score(predictionCopy[k].values, (prediction[k].values > best_threshold).astype(int), average = 'macro')\n    print(f'Q{k+1}: F1 = ', m)\n\nm = f1_score(predictionCopy.values.reshape((-1)), (prediction.values.reshape((-1)) > best_threshold).astype(int), average='macro')\nprint('Overall F1 = ', m)","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:41:31.945562Z","iopub.execute_input":"2023-05-11T01:41:31.946501Z","iopub.status.idle":"2023-05-11T01:41:32.285363Z","shell.execute_reply.started":"2023-05-11T01:41:31.946463Z","shell.execute_reply":"2023-05-11T01:41:32.283956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Gradient Boosting Classifier","metadata":{}},{"cell_type":"code","source":"folds = GroupKFold(n_splits = 5)\nprediction = pd.DataFrame(data = np.zeros((len(users), 18)), index = users)\nmodels = {}\nscores = pd.DataFrame(index = [f'FOLD_{i}' for i in range(5)], columns = list(range(1,19)))\n\nfor i, (t,v) in enumerate(folds.split(X = data_seg, groups = data_seg.index)):\n    gradient_params = {\n        'loss': 'exponential',\n        'subsample': 0.5,\n        'max_depth': 8,\n        'max_features': 'log2',\n        'verbose': 0\n    }\n    # iterate through questions 1 - 18\n    for j in range(1,19):\n        if j <= 3: grp = '0-4'\n        if j <= 13: grp = '5-12'\n        if j <= 22: grp = '13-22'\n        # train data\n        train_x = data_seg.iloc[t]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = labels.loc[labels.q == j].set_index('session').loc[train_users]\n        # validate data\n        valid_x = data_seg.iloc[v]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = labels.loc[labels.q == j].set_index('session').loc[valid_users]\n        \n        #train model\n        clf = GradientBoostingClassifier(**gradient_params)\n        clf.fit(train_x[features], train_y['correct'],)\n        \n        yhat = clf.predict_proba(valid_x[features])[:, 1]\n        score = f1_score(valid_y['correct'], np.round(yhat).astype(int))\n        \n        models[f'{grp}_{j}'] = clf\n        joblib.dump(clf, f'model-fold{i}-levle{j}.pkl')\n        prediction.loc[valid_users, j-1] = yhat\n        scores.loc[f'FOLD_{i}', j] = score\n        del train_x, train_users, train_y, valid_x, valid_users, valid_y, clf, yhat, score\n        gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:41:32.287107Z","iopub.execute_input":"2023-05-11T01:41:32.287450Z","iopub.status.idle":"2023-05-11T01:48:56.286952Z","shell.execute_reply.started":"2023-05-11T01:41:32.287422Z","shell.execute_reply":"2023-05-11T01:48:56.285718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_gb = scores\nscores_gb","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:48:56.288926Z","iopub.execute_input":"2023-05-11T01:48:56.289279Z","iopub.status.idle":"2023-05-11T01:48:56.309837Z","shell.execute_reply.started":"2023-05-11T01:48:56.289249Z","shell.execute_reply":"2023-05-11T01:48:56.308858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.mean(scores_gb)","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:48:56.311263Z","iopub.execute_input":"2023-05-11T01:48:56.311898Z","iopub.status.idle":"2023-05-11T01:48:56.333474Z","shell.execute_reply.started":"2023-05-11T01:48:56.311863Z","shell.execute_reply":"2023-05-11T01:48:56.332315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictionCopy = prediction.copy()\nfor j in range(18):\n    tmp = labels.loc[labels.q == j+1].set_index('session').loc[users]\n    predictionCopy[j] = tmp.correct.values\n    \nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.3, 0.95, 0.01):\n    preds = (prediction.values.reshape((-1)) > threshold).astype(int)\n    m = f1_score(predictionCopy.values.reshape((-1)), preds, average='macro')\n    scores.append(m)\n    thresholds.append(threshold)\n    if m > best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:48:56.334969Z","iopub.execute_input":"2023-05-11T01:48:56.335513Z","iopub.status.idle":"2023-05-11T01:49:07.309371Z","shell.execute_reply.started":"2023-05-11T01:48:56.335481Z","shell.execute_reply":"2023-05-11T01:49:07.308361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT THRESHOLD VS. F1_SCORE\n\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='lightcoral')\nplt.scatter([best_threshold], [best_score], color='cornflowerblue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Gradient Boosting Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:49:07.310850Z","iopub.execute_input":"2023-05-11T01:49:07.311377Z","iopub.status.idle":"2023-05-11T01:49:07.606728Z","shell.execute_reply.started":"2023-05-11T01:49:07.311347Z","shell.execute_reply":"2023-05-11T01:49:07.605768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_per_question = {}\nfor k in range(18):\n    m = f1_score(predictionCopy[k].values, (prediction[k].values > best_threshold).astype(int), average = 'macro')\n    print(f'Q{k+1}: F1 = ', m)\n\nm = f1_score(predictionCopy.values.reshape((-1)), (prediction.values.reshape((-1)) > best_threshold).astype(int), average='macro')\nprint('Overall F1 = ', m)","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:49:07.608081Z","iopub.execute_input":"2023-05-11T01:49:07.608632Z","iopub.status.idle":"2023-05-11T01:49:07.965251Z","shell.execute_reply.started":"2023-05-11T01:49:07.608585Z","shell.execute_reply":"2023-05-11T01:49:07.964344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### SVC","metadata":{}},{"cell_type":"code","source":"folds = GroupKFold(n_splits = 5)\nprediction = pd.DataFrame(data = np.zeros((len(users), 18)), index = users)\nmodels = {}\nscores = pd.DataFrame(index = [f'FOLD_{i}' for i in range(5)], columns = list(range(1,19)))\n\nfor i, (t,v) in enumerate(folds.split(X = data_seg, groups = data_seg.index)):\n    svm_params = {\n        'kernel': 'linear',\n        'probability': True,\n        'max_iter': 100\n    }\n    # iterate through questions 1 - 18\n    for j in range(1,19):\n        if j <= 3: grp = '0-4'\n        if j <= 13: grp = '5-12'\n        if j <= 22: grp = '13-22'\n        # train data\n        train_x = data_seg.iloc[t]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = labels.loc[labels.q == j].set_index('session').loc[train_users]\n        # validate data\n        valid_x = data_seg.iloc[v]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = labels.loc[labels.q == j].set_index('session').loc[valid_users]\n        \n        #train model\n        svm = SVC(**svm_params)\n        svm.fit(train_x[features], train_y['correct'],)\n        \n        yhat = svm.predict_proba(valid_x[features])[:, 1]\n        score = f1_score(valid_y['correct'], np.round(yhat).astype(int))\n        \n        models[f'{grp}_{j}'] = svm\n        joblib.dump(svm, f'model-fold{i}-levle{j}.pkl')\n        prediction.loc[valid_users, j-1] = yhat\n        scores.loc[f'FOLD_{i}', j] = score\n        del train_x, train_users, train_y, valid_x, valid_users, valid_y, svm, yhat, score\n        gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:49:07.966939Z","iopub.execute_input":"2023-05-11T01:49:07.967575Z","iopub.status.idle":"2023-05-11T01:49:50.392912Z","shell.execute_reply.started":"2023-05-11T01:49:07.967541Z","shell.execute_reply":"2023-05-11T01:49:50.391646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_svm = scores\nscores_svm","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:49:50.395958Z","iopub.execute_input":"2023-05-11T01:49:50.396634Z","iopub.status.idle":"2023-05-11T01:49:50.418075Z","shell.execute_reply.started":"2023-05-11T01:49:50.396584Z","shell.execute_reply":"2023-05-11T01:49:50.416789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.mean(scores_svm)","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:49:50.419506Z","iopub.execute_input":"2023-05-11T01:49:50.419906Z","iopub.status.idle":"2023-05-11T01:49:50.432193Z","shell.execute_reply.started":"2023-05-11T01:49:50.419876Z","shell.execute_reply":"2023-05-11T01:49:50.430910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictionCopy = prediction.copy()\nfor j in range(18):\n    tmp = labels.loc[labels.q == j+1].set_index('session').loc[users]\n    predictionCopy[j] = tmp.correct.values\n    \nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.3, 0.95, 0.01):\n    preds = (prediction.values.reshape((-1)) > threshold).astype(int)\n    m = f1_score(predictionCopy.values.reshape((-1)), preds, average='macro')\n    scores.append(m)\n    thresholds.append(threshold)\n    if m > best_score:\n        best_score = m\n        best_threshold = threshold\n        \n# PLOT THRESHOLD VS. F1_SCORE\n\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='lightcoral')\nplt.scatter([best_threshold], [best_score], color='cornflowerblue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'SVM Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:49:50.433982Z","iopub.execute_input":"2023-05-11T01:49:50.434345Z","iopub.status.idle":"2023-05-11T01:50:01.270172Z","shell.execute_reply.started":"2023-05-11T01:49:50.434301Z","shell.execute_reply":"2023-05-11T01:50:01.268676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_per_question = {}\nfor k in range(18):\n    m = f1_score(predictionCopy[k].values, (prediction[k].values > best_threshold).astype(int), average = 'macro')\n    print(f'Q{k+1}: F1 = ', m)\n\nm = f1_score(predictionCopy.values.reshape((-1)), (prediction.values.reshape((-1)) > best_threshold).astype(int), average='macro')\nprint('Overall F1 = ', m)","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:50:01.273405Z","iopub.execute_input":"2023-05-11T01:50:01.273804Z","iopub.status.idle":"2023-05-11T01:50:01.615883Z","shell.execute_reply.started":"2023-05-11T01:50:01.273770Z","shell.execute_reply":"2023-05-11T01:50:01.614856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### KNN Classification","metadata":{}},{"cell_type":"code","source":"folds = GroupKFold(n_splits = 5)\nprediction = pd.DataFrame(data = np.zeros((len(users), 18)), index = users)\nmodels = {}\nscores = pd.DataFrame(index = [f'FOLD_{i}' for i in range(5)], columns = list(range(1,19)))\n\nfor i, (t,v) in enumerate(folds.split(X = data_seg, groups = data_seg.index)):\n    knn_params = {\n        'n_neighbors': 5,\n        'n_jobs': -1\n    }\n    # iterate through questions 1 - 18\n    for j in range(1,19):\n        if j <= 3: grp = '0-4'\n        if j <= 13: grp = '5-12'\n        if j <= 22: grp = '13-22'\n        # train data\n        train_x = data_seg.iloc[t]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = labels.loc[labels.q == j].set_index('session').loc[train_users]\n        # validate data\n        valid_x = data_seg.iloc[v]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = labels.loc[labels.q == j].set_index('session').loc[valid_users]\n        \n        #train model\n        knn = KNeighborsClassifier(**knn_params)\n        knn.fit(train_x[features], train_y['correct'],)\n        \n        yhat = knn.predict_proba(valid_x[features])[:, 1]\n        score = f1_score(valid_y['correct'], np.round(yhat).astype(int))\n        \n        models[f'{grp}_{j}'] = knn\n        joblib.dump(knn, f'model-fold{i}-levle{j}.pkl')\n        prediction.loc[valid_users, j-1] = yhat\n        scores.loc[f'FOLD_{i}', j] = score\n        del train_x, train_users, train_y, valid_x, valid_users, valid_y, knn, yhat, score\n        gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:50:01.617263Z","iopub.execute_input":"2023-05-11T01:50:01.618097Z","iopub.status.idle":"2023-05-11T01:50:56.848096Z","shell.execute_reply.started":"2023-05-11T01:50:01.618058Z","shell.execute_reply":"2023-05-11T01:50:56.846809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_knn = scores\nscores_knn","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:50:56.850750Z","iopub.execute_input":"2023-05-11T01:50:56.851423Z","iopub.status.idle":"2023-05-11T01:50:56.871922Z","shell.execute_reply.started":"2023-05-11T01:50:56.851387Z","shell.execute_reply":"2023-05-11T01:50:56.870819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.mean(scores_knn)","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:50:56.873298Z","iopub.execute_input":"2023-05-11T01:50:56.874365Z","iopub.status.idle":"2023-05-11T01:50:56.886701Z","shell.execute_reply.started":"2023-05-11T01:50:56.874333Z","shell.execute_reply":"2023-05-11T01:50:56.885825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictionCopy = prediction.copy()\nfor j in range(18):\n    tmp = labels.loc[labels.q == j+1].set_index('session').loc[users]\n    predictionCopy[j] = tmp.correct.values\n    \nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.3, 0.95, 0.01):\n    preds = (prediction.values.reshape((-1)) > threshold).astype(int)\n    m = f1_score(predictionCopy.values.reshape((-1)), preds, average='macro')\n    scores.append(m)\n    thresholds.append(threshold)\n    if m > best_score:\n        best_score = m\n        best_threshold = threshold\n        \n# PLOT THRESHOLD VS. F1_SCORE\n\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='lightcoral')\nplt.scatter([best_threshold], [best_score], color='cornflowerblue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'KNN Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:50:56.888047Z","iopub.execute_input":"2023-05-11T01:50:56.888974Z","iopub.status.idle":"2023-05-11T01:51:08.309456Z","shell.execute_reply.started":"2023-05-11T01:50:56.888941Z","shell.execute_reply":"2023-05-11T01:51:08.308567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Random Forest Classifier","metadata":{}},{"cell_type":"code","source":"folds = GroupKFold(n_splits = 5)\nprediction = pd.DataFrame(data = np.zeros((len(users), 18)), index = users)\nmodels = {}\nscores = pd.DataFrame(index = [f'FOLD_{i}' for i in range(5)], columns = list(range(1,19)))\n\nfor i, (t,v) in enumerate(folds.split(X = data_seg, groups = data_seg.index)):\n    forest_params = {\n        'n_estimators': 50,\n        'criterion': 'entropy',\n        'n_jobs': -1,\n        'max_depth': 8,\n        'max_features': 'log2',\n        'verbose': 0\n    }\n    # iterate through questions 1 - 18\n    for j in range(1,19):\n        if j <= 3: grp = '0-4'\n        if j <= 13: grp = '5-12'\n        if j <= 22: grp = '13-22'\n        # train data\n        train_x = data_seg.iloc[t]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = labels.loc[labels.q == j].set_index('session').loc[train_users]\n        # validate data\n        valid_x = data_seg.iloc[v]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = labels.loc[labels.q == j].set_index('session').loc[valid_users]\n        \n        #train model\n        clf = RandomForestClassifier(**forest_params)\n        clf.fit(train_x[features], train_y['correct'],)\n        \n        yhat = clf.predict_proba(valid_x[features])[:, 1]\n        score = f1_score(valid_y['correct'], np.round(yhat).astype(int))\n        \n        models[f'{grp}_{j}'] = clf\n        joblib.dump(clf, f'model-fold{i}-levle{j}.pkl')\n        prediction.loc[valid_users, j-1] = yhat\n        scores.loc[f'FOLD_{i}', j] = score\n        del train_x, train_users, train_y, valid_x, valid_users, valid_y, clf, yhat, score\n        gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:51:08.310791Z","iopub.execute_input":"2023-05-11T01:51:08.311292Z","iopub.status.idle":"2023-05-11T01:54:16.035440Z","shell.execute_reply.started":"2023-05-11T01:51:08.311259Z","shell.execute_reply":"2023-05-11T01:54:16.033985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_forest = scores\nscores_forest","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:54:16.037163Z","iopub.execute_input":"2023-05-11T01:54:16.037496Z","iopub.status.idle":"2023-05-11T01:54:16.057576Z","shell.execute_reply.started":"2023-05-11T01:54:16.037466Z","shell.execute_reply":"2023-05-11T01:54:16.056676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.mean(scores_forest)","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:54:16.059053Z","iopub.execute_input":"2023-05-11T01:54:16.059668Z","iopub.status.idle":"2023-05-11T01:54:16.083293Z","shell.execute_reply.started":"2023-05-11T01:54:16.059632Z","shell.execute_reply":"2023-05-11T01:54:16.082179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictionCopy = prediction.copy()\nfor j in range(18):\n    tmp = labels.loc[labels.q == j+1].set_index('session').loc[users]\n    predictionCopy[j] = tmp.correct.values\n    \nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.3, 0.95, 0.01):\n    preds = (prediction.values.reshape((-1)) > threshold).astype(int)\n    m = f1_score(predictionCopy.values.reshape((-1)), preds, average='macro')\n    scores.append(m)\n    thresholds.append(threshold)\n    if m > best_score:\n        best_score = m\n        best_threshold = threshold\n        \n# PLOT THRESHOLD VS. F1_SCORE\n\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='lightcoral')\nplt.scatter([best_threshold], [best_score], color='cornflowerblue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Random Forest Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:54:16.084936Z","iopub.execute_input":"2023-05-11T01:54:16.085540Z","iopub.status.idle":"2023-05-11T01:54:27.016908Z","shell.execute_reply.started":"2023-05-11T01:54:16.085491Z","shell.execute_reply":"2023-05-11T01:54:27.015897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_per_question = {}\nfor k in range(18):\n    m = f1_score(predictionCopy[k].values, (prediction[k].values > best_threshold).astype(int), average = 'macro')\n    print(f'Q{k+1}: F1 = ', m)\n\nm = f1_score(predictionCopy.values.reshape((-1)), (prediction.values.reshape((-1)) > best_threshold).astype(int), average='macro')\nprint('Overall F1 = ', m)","metadata":{"execution":{"iopub.status.busy":"2023-05-11T01:54:27.018449Z","iopub.execute_input":"2023-05-11T01:54:27.019055Z","iopub.status.idle":"2023-05-11T01:54:27.361859Z","shell.execute_reply.started":"2023-05-11T01:54:27.019018Z","shell.execute_reply":"2023-05-11T01:54:27.360227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Conclusion & Prediction\n- Random Forest Classification has been tested to have the best F1 score\n- We'll predict the test data set using Random Foreset Classification","metadata":{}},{"cell_type":"code","source":"test_data\ndrop = ['page', 'hover_duration', 'text','text_fqid','fullscreen','hq','music']\nuseful = test_data.columns.difference(drop)\ntest_data = test_data[useful]\ntest_data\ncategorical = ['event_name', 'name', 'fqid', 'room_fqid']\nnumerical = ['elapsed_time', 'level', 'room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y']\n\nevents = list(np.unique(test_data['event_name']))\nnames = list(np.unique(test_data['name']))\n\ndfs = []\nfor cat in categorical:\n    tmp = test_data.groupby(['session_id','level_group'])[cat].agg('nunique')\n    tmp.name = tmp.name + \"_nunique\"\n    dfs.append(tmp)\nfor num in numerical:\n    tmp1 = test_data.groupby(['session_id', 'level_group'])[num].agg('mean')\n    dfs.append(tmp1)\nfor num in numerical:\n    tmp2 = test_data.groupby(['session_id', 'level_group'])[num].agg('std')\n    tmp2.name = tmp2.name + \"_std\"\n    dfs.append(tmp2)\ntest_seg = pd.concat(dfs, axis=1)\ntest_seg = test_seg.fillna(-1)\ntest_seg = test_seg.reset_index()\ntest_seg = test_seg.set_index('session_id')\ntest_seg\n\nfeatures = test_seg.columns[1:]\nusers = test_seg.index.unique()\nusers\n","metadata":{"execution":{"iopub.status.busy":"2023-05-11T03:08:49.722218Z","iopub.execute_input":"2023-05-11T03:08:49.722702Z","iopub.status.idle":"2023-05-11T03:08:49.818451Z","shell.execute_reply.started":"2023-05-11T03:08:49.722662Z","shell.execute_reply":"2023-05-11T03:08:49.817422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# feed test_seg into the model\nfolds = GroupKFold(n_splits = 3)\nprediction = pd.DataFrame(data = np.zeros((len(users), 18)), index = users)\nmodels = {}\nscores = pd.DataFrame(index = [f'FOLD_{i}' for i in range(3)], columns = list(range(1,19)))\n\nfor i, (t,v) in enumerate(folds.split(X = test_seg, groups = test_seg.index)):\n    forest_params = {\n        'n_estimators': 50,\n        'criterion': 'entropy',\n        'n_jobs': -1,\n        'max_depth': 8,\n        'max_features': 'log2',\n        'verbose': 0\n    }\n    # iterate through questions 1 - 18\n    for j in range(1,19):\n        if j <= 3: grp = '0-4'\n        if j <= 13: grp = '5-12'\n        if j <= 22: grp = '13-22'\n        # train data (from the best model)\n        train_x = data_seg.iloc[t]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = labels.loc[labels.q == j].set_index('session').loc[train_users]\n        # predict\n        clf = RandomForestClassifier(**forest_params)\n        clf.fit(train_x[features], train_y['correct'],)\n        valid_x = test_seg.iloc[v]\n        valid_users = valid_x.index.values\n        valid_x = valid_x[features]\n        pred = clf.predict(valid_x)\n#         print(pred)\n        prediction.loc[valid_users, j-1] = pred\n        \n        del train_x, train_users, train_y, valid_x, valid_users, clf, pred\n\nsubmission = prediction\n","metadata":{"execution":{"iopub.status.busy":"2023-05-11T03:10:36.717778Z","iopub.execute_input":"2023-05-11T03:10:36.718305Z","iopub.status.idle":"2023-05-11T03:10:46.510588Z","shell.execute_reply.started":"2023-05-11T03:10:36.718269Z","shell.execute_reply":"2023-05-11T03:10:46.509488Z"},"trusted":true},"execution_count":null,"outputs":[]}]}