{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import time\nimport joblib, gc\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom matplotlib.colors import ListedColormap\nimport sklearn.metrics as sm\nimport seaborn as sns\nfrom scipy.stats import kstest, ks_2samp\n\nfrom sklearn.linear_model import LinearRegression,LogisticRegression,LassoCV,RidgeCV,ElasticNetCV\nfrom sklearn.metrics import accuracy_score,classification_report,confusion_matrix,f1_score\nfrom sklearn.model_selection import train_test_split,cross_val_score,cross_validate,KFold, StratifiedKFold, GridSearchCV, GroupKFold\nfrom sklearn.neighbors import KNeighborsClassifier,KNeighborsRegressor,NearestCentroid\n\nfrom sklearn.tree import DecisionTreeRegressor, DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestRegressor, RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.feature_selection import SelectFromModel\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, LabelEncoder, OneHotEncoder\nfrom sklearn.feature_selection import SelectKBest, RFE, RFECV\n\nfrom sklearn.svm import LinearSVC, SVC, SVR\nfrom sklearn.decomposition import PCA\n\nfrom sklearn.cluster import KMeans, AgglomerativeClustering\nfrom sklearn.datasets import make_blobs\nfrom scipy.cluster.hierarchy import dendrogram\n","metadata":{"papermill":{"duration":0.023295,"end_time":"2022-06-03T21:13:10.412151","exception":false,"start_time":"2022-06-03T21:13:10.388856","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-12T01:25:13.127309Z","iopub.execute_input":"2023-05-12T01:25:13.128099Z","iopub.status.idle":"2023-05-12T01:25:15.099259Z","shell.execute_reply.started":"2023-05-12T01:25:13.128000Z","shell.execute_reply":"2023-05-12T01:25:15.097888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Loading the Dataset\n- Specify the datatype of each feature to reduce loading time\n- referenced [this](http://www.kaggle.com/code/gusthema/student-performance-w-tensorflow-decision-forests) thread","metadata":{}},{"cell_type":"code","source":"# specify data type and downcast each into smaller type to reduce memory usage when loading dataset\ndtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\ndata = pd.read_csv(\"/kaggle/input/student-performance-and-game-play/train.csv\", dtype = dtypes)\ndata","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:25:15.103450Z","iopub.execute_input":"2023-05-12T01:25:15.103840Z","iopub.status.idle":"2023-05-12T01:27:33.394923Z","shell.execute_reply.started":"2023-05-12T01:25:15.103793Z","shell.execute_reply":"2023-05-12T01:27:33.393304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv(\"/kaggle/input/student-performance-and-game-play/test.csv\", dtype = dtypes)\ntest_data","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:27:33.396494Z","iopub.execute_input":"2023-05-12T01:27:33.396967Z","iopub.status.idle":"2023-05-12T01:27:33.482020Z","shell.execute_reply.started":"2023-05-12T01:27:33.396932Z","shell.execute_reply":"2023-05-12T01:27:33.481133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Check Missing Data ","metadata":{}},{"cell_type":"code","source":"train_missing = data.isna().sum() / len(data)\ntest_missing = test_data.isna().sum() / len(test_data)\nplt.figure(figsize = (10, 4))\nplt.subplot(1, 2, 1)\nplt.bar(train_missing.index,\n        train_missing.values,\n        color=['red' if ratio >= 0.6 else 'grey' for ratio in train_missing.values])\nplt.xlabel('Feature', fontsize=12)\nplt.ylabel('Missing values ratio', fontsize=12)\nplt.title('Missing values in Train Set', fontsize=16)\nplt.xticks(rotation=90)\n# plt.legend(handles=[mpatches.Patch(color='grey'),\n#                     mpatches.Patch(color='red')], \n#            labels=['Partially missing values', 'Completely missing values'])\nplt.subplot(1, 2, 2)\nplt.bar(test_missing.index,\n        test_missing.values,\n        color=['red' if ratio >= 0.6 else 'grey' for ratio in test_missing.values])\nplt.title('Missing values in Test Set', fontsize=16)\nplt.xticks(rotation=90)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:27:33.484424Z","iopub.execute_input":"2023-05-12T01:27:33.484985Z","iopub.status.idle":"2023-05-12T01:27:34.874437Z","shell.execute_reply.started":"2023-05-12T01:27:33.484952Z","shell.execute_reply":"2023-05-12T01:27:34.873414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(data[data['fullscreen'] == 1]['fullscreen']))\nprint(len(data[data['hq'] == 1]['hq']))\nprint(len(data[data['music'] == 1]['music']))","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:27:34.875740Z","iopub.execute_input":"2023-05-12T01:27:34.876282Z","iopub.status.idle":"2023-05-12T01:27:34.972893Z","shell.execute_reply.started":"2023-05-12T01:27:34.876248Z","shell.execute_reply":"2023-05-12T01:27:34.971666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# because we'll be running classifications(trees), we don't need to check\n# multicolinearlity issues between the features as trees have built-in mechanisms to drop\n# features with high multicolinearlities\n\ndrop = ['page', 'hover_duration', 'text','text_fqid','fullscreen','hq','music']\nuseful = data.columns.difference(drop)\ndata = data[useful]\n\ncategorical = ['event_name', 'name', 'fqid', 'room_fqid']\nnumerical = ['elapsed_time', 'level', 'room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y']\n\nevents = list(np.unique(data['event_name']))\n# events = ['checkpoint', 'cutscene_click', 'map_click', 'map_hover', 'navigate_click', 'notebook_click', 'notification_click',\n#          'object_click', 'object_hover', 'observation_click', 'person_click']\nnames = list(np.unique(data['name']))\n# names = ['basic', 'close', 'next', 'open', 'prev', 'undefined']","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:27:34.974473Z","iopub.execute_input":"2023-05-12T01:27:34.974849Z","iopub.status.idle":"2023-05-12T01:29:00.939521Z","shell.execute_reply.started":"2023-05-12T01:27:34.974797Z","shell.execute_reply":"2023-05-12T01:29:00.937931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# identify unique sessions('session_id') and their counts\nunique_sessions = data.groupby('session_id').session_id.agg('count')\ndisplay(unique_sessions)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:29:00.941292Z","iopub.execute_input":"2023-05-12T01:29:00.941681Z","iopub.status.idle":"2023-05-12T01:29:01.570996Z","shell.execute_reply.started":"2023-05-12T01:29:00.941644Z","shell.execute_reply":"2023-05-12T01:29:01.569451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load the labels (= correct value for all 18 questions for each session)\nlabels = pd.read_csv(\"/kaggle/input/student-performance-and-game-play/train_labels.csv\")\nlabels['session'] = labels.session_id.apply(lambda x: int(x.split('_')[0]))\nlabels['q'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]))\nlabels['q'] = labels['q'].astype(int)\nlabels","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:29:01.572523Z","iopub.execute_input":"2023-05-12T01:29:01.573440Z","iopub.status.idle":"2023-05-12T01:29:03.137417Z","shell.execute_reply.started":"2023-05-12T01:29:01.573398Z","shell.execute_reply":"2023-05-12T01:29:03.136500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_unique = list(np.unique(labels['session']))\nprint(len(labels_unique)*3)\nlabels['q'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:29:03.138791Z","iopub.execute_input":"2023-05-12T01:29:03.139654Z","iopub.status.idle":"2023-05-12T01:29:03.177016Z","shell.execute_reply.started":"2023-05-12T01:29:03.139616Z","shell.execute_reply":"2023-05-12T01:29:03.175846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualize the correct answers of the labels\n\nplt.figure(figsize=(3, 3))\nplt.suptitle(\"Correct column values for the entire dataset\", fontsize=14, y=0.94)\nplot_df = labels.correct.value_counts()\nplot_df.plot(kind=\"bar\", color=['cornflowerblue', 'lightcoral'])\n\n\nplt.figure(figsize=(10, 20))\nplt.subplots_adjust(hspace=0.5, wspace=0.5)\nplt.suptitle(\"Correct column values for each question\", fontsize=14, y=0.94)\nfor n in range(1,19):\n    #print(n, str(n))\n    ax = plt.subplot(6, 3, n)\n\n    # filter df and plot ticker on the new subplot axis\n    plot_df = labels.loc[labels.q == n]\n    plot_df = plot_df.correct.value_counts()\n    plot_df.plot(ax=ax, kind=\"bar\", color=['cornflowerblue', 'lightcoral'])\n    \n    # chart formatting\n    ax.set_title(\"Question \" + str(n))\n    ax.set_xlabel(\"\")","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:29:03.181265Z","iopub.execute_input":"2023-05-12T01:29:03.181753Z","iopub.status.idle":"2023-05-12T01:29:05.259008Z","shell.execute_reply.started":"2023-05-12T01:29:03.181717Z","shell.execute_reply":"2023-05-12T01:29:05.257755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we need to make prediction for each segment;\n# the same session_id (20090312431273200) for example has 3 segments based on \n# level_group: 0-4, 5-12, 13-22\n# We need to predict for each group of the unique session_id, not on every single sample\n# Hence, group the session_id by level_group\ndef feature_engineer(data):\n    dfs = []\n    for cat in categorical:\n        tmp = data.groupby(['session_id','level_group'])[cat].agg('nunique')\n        tmp.name = tmp.name + \"_nunique\"\n        dfs.append(tmp)\n    for num in numerical:\n        tmp1 = data.groupby(['session_id', 'level_group'])[num].agg('mean')\n        dfs.append(tmp1)\n    for num in numerical:\n        tmp2 = data.groupby(['session_id', 'level_group'])[num].agg('std')\n        tmp2.name = tmp2.name + \"_std\"\n        dfs.append(tmp2)\n    data_seg = pd.concat(dfs, axis=1)\n    data_seg = data_seg.fillna(-1)\n    data_seg = data_seg.reset_index()\n    data_seg = data_seg.set_index('session_id')\n\n    return data_seg\n\ndata_seg = feature_engineer(data)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:29:05.260568Z","iopub.execute_input":"2023-05-12T01:29:05.260941Z","iopub.status.idle":"2023-05-12T01:29:44.754570Z","shell.execute_reply.started":"2023-05-12T01:29:05.260907Z","shell.execute_reply":"2023-05-12T01:29:44.753059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# final processing to extract the data that will be trained with models;\n# we'll train following 'features' on 'users' info(sample)\n\nfeatures = data_seg.columns[1:]\nusers = data_seg.index.unique()\nusers","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:29:44.757419Z","iopub.execute_input":"2023-05-12T01:29:44.758210Z","iopub.status.idle":"2023-05-12T01:29:44.772044Z","shell.execute_reply.started":"2023-05-12T01:29:44.758146Z","shell.execute_reply":"2023-05-12T01:29:44.770876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Logistic Regression\n- Behavior of the models referenced [this](https://www.kaggle.com/code/rizkykiky/baseline-logistic-regression-only-group-level) thread","metadata":{}},{"cell_type":"code","source":"folds = GroupKFold(n_splits = 5)\nprediction = pd.DataFrame(data = np.zeros((len(users), 18)), index = users)\nmodels = {}\nscores = pd.DataFrame(index = [f'FOLD_{i}' for i in range(5)], columns = list(range(1,19)))\n\nfor i, (t,v) in enumerate(folds.split(X = data_seg, groups = data_seg.index)):\n    for l in range(1,19):\n        if l <= 3: grp = '0-4'\n        if l <= 13: grp = '5-12'\n        if l <= 22: grp = '13-22'\n        xtrain = data_seg.iloc[t]\n        xtrain = xtrain.loc[xtrain.level_group == grp]\n        train_users = xtrain.index.values\n        ytrain = labels.loc[labels.q==l].set_index('session').loc[train_users]\n        \n        xval = data_seg.iloc[v]\n        xval = xval.loc[xval.level_group == grp]\n        val_users = xval.index.values\n        yval = labels.loc[labels.q == l].set_index('session').loc[val_users]\n        \n        model = LogisticRegression(random_state = 0, solver = 'saga', max_iter = 1500, n_jobs = -1, C = 0.1)\n        model.fit(xtrain[features], ytrain['correct'],)\n        \n        yhat = model.predict_proba(xval[features])[:, 1]\n        score = f1_score(yval['correct'], np.round(yhat).astype(int))\n        \n        models[f'fold{i}-level{l}'] = model\n        joblib.dump(model, f'model-fold{i}-level{l}.pkl')\n        prediction.loc[val_users, l-1] = yhat\n        scores.loc[f'FOLD_{i}',l] = score\n        del xtrain, train_users, ytrain, xval, val_users, yval, model, yhat, score\n        gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:29:44.773459Z","iopub.execute_input":"2023-05-12T01:29:44.773831Z","iopub.status.idle":"2023-05-12T01:45:20.984384Z","shell.execute_reply.started":"2023-05-12T01:29:44.773784Z","shell.execute_reply":"2023-05-12T01:45:20.982928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_logistic = scores\nscores_logistic","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:45:20.986189Z","iopub.execute_input":"2023-05-12T01:45:20.986559Z","iopub.status.idle":"2023-05-12T01:45:21.015056Z","shell.execute_reply.started":"2023-05-12T01:45:20.986525Z","shell.execute_reply":"2023-05-12T01:45:21.013502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.mean(scores_logistic)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:45:21.017228Z","iopub.execute_input":"2023-05-12T01:45:21.017953Z","iopub.status.idle":"2023-05-12T01:45:21.031209Z","shell.execute_reply.started":"2023-05-12T01:45:21.017907Z","shell.execute_reply":"2023-05-12T01:45:21.030240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictionCopy = prediction.copy()\nfor k in range(18):\n    tmp = labels.loc[labels.q == k+1].set_index('session').loc[users]\n    predictionCopy[k] = tmp.correct.values\n    \nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.3, 0.95, 0.01):\n    preds = (prediction.values.reshape((-1)) > threshold).astype(int)\n    m = f1_score(predictionCopy.values.reshape((-1)), preds, average='macro')\n    scores.append(m)\n    thresholds.append(threshold)\n    if m > best_score:\n        best_score = m; best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:45:21.032438Z","iopub.execute_input":"2023-05-12T01:45:21.033524Z","iopub.status.idle":"2023-05-12T01:45:34.148805Z","shell.execute_reply.started":"2023-05-12T01:45:21.033486Z","shell.execute_reply":"2023-05-12T01:45:34.147553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT THRESHOLD VS. F1_SCORE\n\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='lightcoral')\nplt.scatter([best_threshold], [best_score], color='cornflowerblue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:45:34.150558Z","iopub.execute_input":"2023-05-12T01:45:34.150939Z","iopub.status.idle":"2023-05-12T01:45:34.423972Z","shell.execute_reply.started":"2023-05-12T01:45:34.150905Z","shell.execute_reply":"2023-05-12T01:45:34.423086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_per_question = {}\nfor k in range(18):\n    m = f1_score(predictionCopy[k].values, (prediction[k].values > best_threshold).astype(int), average = 'macro')\n    print(f'Q{k+1}: F1 = ', m)\n\nm = f1_score(predictionCopy.values.reshape((-1)), (prediction.values.reshape((-1)) > best_threshold).astype(int), average='macro')\nprint('Overall F1 = ', m)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:45:34.428292Z","iopub.execute_input":"2023-05-12T01:45:34.428653Z","iopub.status.idle":"2023-05-12T01:45:34.827712Z","shell.execute_reply.started":"2023-05-12T01:45:34.428621Z","shell.execute_reply":"2023-05-12T01:45:34.826454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Gradient Boosting Classifier","metadata":{}},{"cell_type":"code","source":"folds = GroupKFold(n_splits = 5)\nprediction = pd.DataFrame(data = np.zeros((len(users), 18)), index = users)\nmodels = {}\nscores = pd.DataFrame(index = [f'FOLD_{i}' for i in range(5)], columns = list(range(1,19)))\n\nfor i, (t,v) in enumerate(folds.split(X = data_seg, groups = data_seg.index)):\n    gradient_params = {\n        'loss': 'exponential',\n        'subsample': 0.5,\n        'max_depth': 8,\n        'max_features': 'log2',\n        'verbose': 0\n    }\n    # iterate through questions 1 - 18\n    for j in range(1,19):\n        if j <= 3: grp = '0-4'\n        if j <= 13: grp = '5-12'\n        if j <= 22: grp = '13-22'\n        # train data\n        train_x = data_seg.iloc[t]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = labels.loc[labels.q == j].set_index('session').loc[train_users]\n        # validate data\n        valid_x = data_seg.iloc[v]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = labels.loc[labels.q == j].set_index('session').loc[valid_users]\n        \n        #train model\n        clf = GradientBoostingClassifier(**gradient_params)\n        clf.fit(train_x[features], train_y['correct'],)\n        \n        yhat = clf.predict_proba(valid_x[features])[:, 1]\n        score = f1_score(valid_y['correct'], np.round(yhat).astype(int))\n        \n        models[f'{grp}_{j}'] = clf\n        joblib.dump(clf, f'model-fold{i}-levle{j}.pkl')\n        prediction.loc[valid_users, j-1] = yhat\n        scores.loc[f'FOLD_{i}', j] = score\n        del train_x, train_users, train_y, valid_x, valid_users, valid_y, clf, yhat, score\n        gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:45:34.829546Z","iopub.execute_input":"2023-05-12T01:45:34.829922Z","iopub.status.idle":"2023-05-12T01:52:14.013546Z","shell.execute_reply.started":"2023-05-12T01:45:34.829890Z","shell.execute_reply":"2023-05-12T01:52:14.012301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_gb = scores\nscores_gb","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:52:14.015495Z","iopub.execute_input":"2023-05-12T01:52:14.015896Z","iopub.status.idle":"2023-05-12T01:52:14.039145Z","shell.execute_reply.started":"2023-05-12T01:52:14.015860Z","shell.execute_reply":"2023-05-12T01:52:14.038056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.mean(scores_gb)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:52:14.040928Z","iopub.execute_input":"2023-05-12T01:52:14.041644Z","iopub.status.idle":"2023-05-12T01:52:14.061840Z","shell.execute_reply.started":"2023-05-12T01:52:14.041603Z","shell.execute_reply":"2023-05-12T01:52:14.060689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictionCopy = prediction.copy()\nfor j in range(18):\n    tmp = labels.loc[labels.q == j+1].set_index('session').loc[users]\n    predictionCopy[j] = tmp.correct.values\n    \nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.3, 0.95, 0.01):\n    preds = (prediction.values.reshape((-1)) > threshold).astype(int)\n    m = f1_score(predictionCopy.values.reshape((-1)), preds, average='macro')\n    scores.append(m)\n    thresholds.append(threshold)\n    if m > best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:52:14.063544Z","iopub.execute_input":"2023-05-12T01:52:14.063920Z","iopub.status.idle":"2023-05-12T01:52:27.799534Z","shell.execute_reply.started":"2023-05-12T01:52:14.063886Z","shell.execute_reply":"2023-05-12T01:52:27.798434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT THRESHOLD VS. F1_SCORE\n\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='lightcoral')\nplt.scatter([best_threshold], [best_score], color='cornflowerblue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Gradient Boosting Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:52:27.801493Z","iopub.execute_input":"2023-05-12T01:52:27.804226Z","iopub.status.idle":"2023-05-12T01:52:28.040428Z","shell.execute_reply.started":"2023-05-12T01:52:27.804174Z","shell.execute_reply":"2023-05-12T01:52:28.038651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_per_question = {}\nfor k in range(18):\n    m = f1_score(predictionCopy[k].values, (prediction[k].values > best_threshold).astype(int), average = 'macro')\n    print(f'Q{k+1}: F1 = ', m)\n\nm = f1_score(predictionCopy.values.reshape((-1)), (prediction.values.reshape((-1)) > best_threshold).astype(int), average='macro')\nprint('Overall F1 = ', m)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:52:28.042499Z","iopub.execute_input":"2023-05-12T01:52:28.043323Z","iopub.status.idle":"2023-05-12T01:52:28.466544Z","shell.execute_reply.started":"2023-05-12T01:52:28.043268Z","shell.execute_reply":"2023-05-12T01:52:28.465518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### SVC","metadata":{}},{"cell_type":"code","source":"folds = GroupKFold(n_splits = 5)\nprediction = pd.DataFrame(data = np.zeros((len(users), 18)), index = users)\nmodels = {}\nscores = pd.DataFrame(index = [f'FOLD_{i}' for i in range(5)], columns = list(range(1,19)))\n\nfor i, (t,v) in enumerate(folds.split(X = data_seg, groups = data_seg.index)):\n    svm_params = {\n        'kernel': 'linear',\n        'probability': True,\n        'max_iter': 100\n    }\n    # iterate through questions 1 - 18\n    for j in range(1,19):\n        if j <= 3: grp = '0-4'\n        if j <= 13: grp = '5-12'\n        if j <= 22: grp = '13-22'\n        # train data\n        train_x = data_seg.iloc[t]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = labels.loc[labels.q == j].set_index('session').loc[train_users]\n        # validate data\n        valid_x = data_seg.iloc[v]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = labels.loc[labels.q == j].set_index('session').loc[valid_users]\n        \n        #train model\n        svm = SVC(**svm_params)\n        svm.fit(train_x[features], train_y['correct'],)\n        \n        yhat = svm.predict_proba(valid_x[features])[:, 1]\n        score = f1_score(valid_y['correct'], np.round(yhat).astype(int))\n        \n        models[f'{grp}_{j}'] = svm\n        joblib.dump(svm, f'model-fold{i}-levle{j}.pkl')\n        prediction.loc[valid_users, j-1] = yhat\n        scores.loc[f'FOLD_{i}', j] = score\n        del train_x, train_users, train_y, valid_x, valid_users, valid_y, svm, yhat, score\n        gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:52:28.468289Z","iopub.execute_input":"2023-05-12T01:52:28.468964Z","iopub.status.idle":"2023-05-12T01:53:03.383366Z","shell.execute_reply.started":"2023-05-12T01:52:28.468926Z","shell.execute_reply":"2023-05-12T01:53:03.382000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_svm = scores\nscores_svm","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:53:03.385782Z","iopub.execute_input":"2023-05-12T01:53:03.386248Z","iopub.status.idle":"2023-05-12T01:53:03.414472Z","shell.execute_reply.started":"2023-05-12T01:53:03.386210Z","shell.execute_reply":"2023-05-12T01:53:03.413339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.mean(scores_svm)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:53:03.416085Z","iopub.execute_input":"2023-05-12T01:53:03.417289Z","iopub.status.idle":"2023-05-12T01:53:03.435444Z","shell.execute_reply.started":"2023-05-12T01:53:03.417237Z","shell.execute_reply":"2023-05-12T01:53:03.434033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictionCopy = prediction.copy()\nfor j in range(18):\n    tmp = labels.loc[labels.q == j+1].set_index('session').loc[users]\n    predictionCopy[j] = tmp.correct.values\n    \nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.3, 0.95, 0.01):\n    preds = (prediction.values.reshape((-1)) > threshold).astype(int)\n    m = f1_score(predictionCopy.values.reshape((-1)), preds, average='macro')\n    scores.append(m)\n    thresholds.append(threshold)\n    if m > best_score:\n        best_score = m\n        best_threshold = threshold\n        \n# PLOT THRESHOLD VS. F1_SCORE\n\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='lightcoral')\nplt.scatter([best_threshold], [best_score], color='cornflowerblue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'SVM Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:53:03.444623Z","iopub.execute_input":"2023-05-12T01:53:03.445071Z","iopub.status.idle":"2023-05-12T01:53:17.309316Z","shell.execute_reply.started":"2023-05-12T01:53:03.445035Z","shell.execute_reply":"2023-05-12T01:53:17.308330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_per_question = {}\nfor k in range(18):\n    m = f1_score(predictionCopy[k].values, (prediction[k].values > best_threshold).astype(int), average = 'macro')\n    print(f'Q{k+1}: F1 = ', m)\n\nm = f1_score(predictionCopy.values.reshape((-1)), (prediction.values.reshape((-1)) > best_threshold).astype(int), average='macro')\nprint('Overall F1 = ', m)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:53:17.311921Z","iopub.execute_input":"2023-05-12T01:53:17.312664Z","iopub.status.idle":"2023-05-12T01:53:17.723008Z","shell.execute_reply.started":"2023-05-12T01:53:17.312619Z","shell.execute_reply":"2023-05-12T01:53:17.721987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### KNN Classification","metadata":{}},{"cell_type":"code","source":"folds = GroupKFold(n_splits = 5)\nprediction = pd.DataFrame(data = np.zeros((len(users), 18)), index = users)\nmodels = {}\nscores = pd.DataFrame(index = [f'FOLD_{i}' for i in range(5)], columns = list(range(1,19)))\n\nfor i, (t,v) in enumerate(folds.split(X = data_seg, groups = data_seg.index)):\n    knn_params = {\n        'n_neighbors': 5,\n        'n_jobs': -1\n    }\n    # iterate through questions 1 - 18\n    for j in range(1,19):\n        if j <= 3: grp = '0-4'\n        if j <= 13: grp = '5-12'\n        if j <= 22: grp = '13-22'\n        # train data\n        train_x = data_seg.iloc[t]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = labels.loc[labels.q == j].set_index('session').loc[train_users]\n        # validate data\n        valid_x = data_seg.iloc[v]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = labels.loc[labels.q == j].set_index('session').loc[valid_users]\n        \n        #train model\n        knn = KNeighborsClassifier(**knn_params)\n        knn.fit(train_x[features], train_y['correct'],)\n        \n        yhat = knn.predict_proba(valid_x[features])[:, 1]\n        score = f1_score(valid_y['correct'], np.round(yhat).astype(int))\n        \n        models[f'{grp}_{j}'] = knn\n        joblib.dump(knn, f'model-fold{i}-levle{j}.pkl')\n        prediction.loc[valid_users, j-1] = yhat\n        scores.loc[f'FOLD_{i}', j] = score\n        del train_x, train_users, train_y, valid_x, valid_users, valid_y, knn, yhat, score\n        gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:53:17.724283Z","iopub.execute_input":"2023-05-12T01:53:17.725442Z","iopub.status.idle":"2023-05-12T01:57:44.645196Z","shell.execute_reply.started":"2023-05-12T01:53:17.725398Z","shell.execute_reply":"2023-05-12T01:57:44.644006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_knn = scores\nscores_knn","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:57:44.647089Z","iopub.execute_input":"2023-05-12T01:57:44.647456Z","iopub.status.idle":"2023-05-12T01:57:44.674630Z","shell.execute_reply.started":"2023-05-12T01:57:44.647422Z","shell.execute_reply":"2023-05-12T01:57:44.673557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.mean(scores_knn)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:57:44.676347Z","iopub.execute_input":"2023-05-12T01:57:44.677093Z","iopub.status.idle":"2023-05-12T01:57:44.693840Z","shell.execute_reply.started":"2023-05-12T01:57:44.677039Z","shell.execute_reply":"2023-05-12T01:57:44.692158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictionCopy = prediction.copy()\nfor j in range(18):\n    tmp = labels.loc[labels.q == j+1].set_index('session').loc[users]\n    predictionCopy[j] = tmp.correct.values\n    \nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.3, 0.95, 0.01):\n    preds = (prediction.values.reshape((-1)) > threshold).astype(int)\n    m = f1_score(predictionCopy.values.reshape((-1)), preds, average='macro')\n    scores.append(m)\n    thresholds.append(threshold)\n    if m > best_score:\n        best_score = m\n        best_threshold = threshold\n        \n# PLOT THRESHOLD VS. F1_SCORE\n\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='lightcoral')\nplt.scatter([best_threshold], [best_score], color='cornflowerblue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'KNN Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:57:44.695784Z","iopub.execute_input":"2023-05-12T01:57:44.696972Z","iopub.status.idle":"2023-05-12T01:57:59.081253Z","shell.execute_reply.started":"2023-05-12T01:57:44.696927Z","shell.execute_reply":"2023-05-12T01:57:59.079997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Random Forest Classifier","metadata":{}},{"cell_type":"code","source":"folds = GroupKFold(n_splits = 5)\nprediction = pd.DataFrame(data = np.zeros((len(users), 18)), index = users)\nmodels = {}\nscores = pd.DataFrame(index = [f'FOLD_{i}' for i in range(5)], columns = list(range(1,19)))\n\nfor i, (t,v) in enumerate(folds.split(X = data_seg, groups = data_seg.index)):\n    forest_params = {\n        'n_estimators': 50,\n        'criterion': 'entropy',\n        'n_jobs': -1,\n        'max_depth': 8,\n        'max_features': 'log2',\n        'verbose': 0\n    }\n    # iterate through questions 1 - 18\n    for j in range(1,19):\n        if j <= 3: grp = '0-4'\n        if j <= 13: grp = '5-12'\n        if j <= 22: grp = '13-22'\n        # train data\n        train_x = data_seg.iloc[t]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = labels.loc[labels.q == j].set_index('session').loc[train_users]\n        # validate data\n        valid_x = data_seg.iloc[v]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = labels.loc[labels.q == j].set_index('session').loc[valid_users]\n        \n        #train model\n        clf = RandomForestClassifier(**forest_params)\n        clf.fit(train_x[features], train_y['correct'])\n        \n        yhat = clf.predict_proba(valid_x[features])[:, 1]\n        score = f1_score(valid_y['correct'], np.round(yhat).astype(int))\n        \n        models[f'{grp}_{j}'] = clf\n        joblib.dump(clf, f'model-fold{i}-levle{j}.pkl')\n        prediction.loc[valid_users, j-1] = yhat\n        scores.loc[f'FOLD_{i}', j] = score\n        del train_x, train_users, train_y, valid_x, valid_users, valid_y, clf, yhat, score\n        gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T01:57:59.083063Z","iopub.execute_input":"2023-05-12T01:57:59.083732Z","iopub.status.idle":"2023-05-12T02:00:44.644055Z","shell.execute_reply.started":"2023-05-12T01:57:59.083691Z","shell.execute_reply":"2023-05-12T02:00:44.642664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_forest = scores\nscores_forest","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:00:44.649408Z","iopub.execute_input":"2023-05-12T02:00:44.649811Z","iopub.status.idle":"2023-05-12T02:00:44.679045Z","shell.execute_reply.started":"2023-05-12T02:00:44.649775Z","shell.execute_reply":"2023-05-12T02:00:44.677801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(models))\nmodels","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:00:44.680439Z","iopub.execute_input":"2023-05-12T02:00:44.680805Z","iopub.status.idle":"2023-05-12T02:00:44.706969Z","shell.execute_reply.started":"2023-05-12T02:00:44.680763Z","shell.execute_reply":"2023-05-12T02:00:44.705931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictionCopy = prediction.copy()\nfor j in range(18):\n    tmp = labels.loc[labels.q == j+1].set_index('session').loc[users]\n    predictionCopy[j] = tmp.correct.values\n    \nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.3, 0.95, 0.01):\n    preds = (prediction.values.reshape((-1)) > threshold).astype(int)\n    m = f1_score(predictionCopy.values.reshape((-1)), preds, average='macro')\n    scores.append(m)\n    thresholds.append(threshold)\n    if m > best_score:\n        best_score = m\n        best_threshold = threshold\n        \n# PLOT THRESHOLD VS. F1_SCORE\n\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='lightcoral')\nplt.scatter([best_threshold], [best_score], color='cornflowerblue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Random Forest Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:00:44.708678Z","iopub.execute_input":"2023-05-12T02:00:44.709433Z","iopub.status.idle":"2023-05-12T02:00:58.684287Z","shell.execute_reply.started":"2023-05-12T02:00:44.709393Z","shell.execute_reply":"2023-05-12T02:00:58.683183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_per_question = {}\nfor k in range(18):\n    m = f1_score(predictionCopy[k].values, (prediction[k].values > best_threshold).astype(int), average = 'macro')\n    print(f'Q{k+1}: F1 = ', m)\n\nm = f1_score(predictionCopy.values.reshape((-1)), (prediction.values.reshape((-1)) > best_threshold).astype(int), average='macro')\nprint('Overall F1 = ', m)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:00:58.686750Z","iopub.execute_input":"2023-05-12T02:00:58.687921Z","iopub.status.idle":"2023-05-12T02:00:59.117680Z","shell.execute_reply.started":"2023-05-12T02:00:58.687867Z","shell.execute_reply":"2023-05-12T02:00:59.116532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Conclusion & Prediction\n- Random Forest Classification has been tested to have the best F1 score\n- We'll predict the test data set using Random Foreset Classification","metadata":{}},{"cell_type":"code","source":"import jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"papermill":{"duration":0.036198,"end_time":"2022-06-03T21:13:10.454318","exception":false,"start_time":"2022-06-03T21:13:10.41812","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-12T02:00:59.119476Z","iopub.execute_input":"2023-05-12T02:00:59.120318Z","iopub.status.idle":"2023-05-12T02:00:59.159278Z","shell.execute_reply.started":"2023-05-12T02:00:59.120274Z","shell.execute_reply":"2023-05-12T02:00:59.158225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\n# env.predict(sample_submission)\nfor (test, sample_submission) in iter_test:\n    \n    # FEATURE ENGINEER TEST DATA\n    df = feature_engineer(test)\n    \n    # INFER TEST DATA\n    grp = test.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        if(grp == '13-22'):\n            clf = models[f'{grp}_{t}']\n            p = clf.predict(df[features].astype('float32'))\n    #         p = clf.predict_proba(df[FEATURES].astype('float32'))[:,1]\n            mask = sample_submission.session_id.str.contains(f'q{t}')\n            sample_submission.loc[mask,'correct'] = ( p >  0.63 ).astype(int)\n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-05-12T02:00:59.160992Z","iopub.execute_input":"2023-05-12T02:00:59.161860Z","iopub.status.idle":"2023-05-12T02:01:01.093153Z","shell.execute_reply.started":"2023-05-12T02:00:59.161808Z","shell.execute_reply":"2023-05-12T02:01:01.091923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## the end result is a submission file containing all test session predictions\n! head submission.csv","metadata":{"papermill":{"duration":0.767504,"end_time":"2022-06-03T21:13:11.572788","exception":false,"start_time":"2022-06-03T21:13:10.805284","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-12T02:01:01.098532Z","iopub.execute_input":"2023-05-12T02:01:01.101224Z","iopub.status.idle":"2023-05-12T02:01:02.286889Z","shell.execute_reply.started":"2023-05-12T02:01:01.101174Z","shell.execute_reply":"2023-05-12T02:01:02.284749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}