{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder, LabelEncoder\nfrom sklearn.ensemble import HistGradientBoostingClassifier\nfrom sklearn import linear_model\nfrom sklearn import preprocessing\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import f1_score\n#to ignore warning when using append instead of concat\nimport warnings\nwarnings.filterwarnings('ignore')\nimport gc","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:14:12.767426Z","iopub.execute_input":"2023-05-03T00:14:12.767793Z","iopub.status.idle":"2023-05-03T00:14:13.777946Z","shell.execute_reply.started":"2023-05-03T00:14:12.767761Z","shell.execute_reply":"2023-05-03T00:14:13.776866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\ntrain = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\nprint(train.shape)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:14:13.782011Z","iopub.execute_input":"2023-05-03T00:14:13.783023Z","iopub.status.idle":"2023-05-03T00:16:04.450049Z","shell.execute_reply.started":"2023-05-03T00:14:13.782981Z","shell.execute_reply":"2023-05-03T00:16:04.448774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\nprint(labels.shape)\n\nlabels['level'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nlabels['session_id'] = labels.session_id.apply(lambda x: int(x.split('_')[0]) )\n\nlabels.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:04.451743Z","iopub.execute_input":"2023-05-03T00:16:04.452460Z","iopub.status.idle":"2023-05-03T00:16:05.406860Z","shell.execute_reply.started":"2023-05-03T00:16:04.452417Z","shell.execute_reply":"2023-05-03T00:16:05.405735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/test.csv')\n#test.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:05.410070Z","iopub.execute_input":"2023-05-03T00:16:05.410465Z","iopub.status.idle":"2023-05-03T00:16:05.414473Z","shell.execute_reply.started":"2023-05-03T00:16:05.410427Z","shell.execute_reply":"2023-05-03T00:16:05.413448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = ['session_id', 'index', 'elapsed_time', 'event_name', 'name', 'level', 'page', \n            'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'hover_duration', \n            'text', 'fqid', 'room_fqid', 'text_fqid', 'fullscreen', 'hq', 'music', 'level_group']\n\ncategorical_features = ['event_name', 'name','text', 'fqid', 'room_fqid', 'text_fqid','level_group']\nnumeric_features = [col for col in train.columns if col not in categorical_features]","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:05.416215Z","iopub.execute_input":"2023-05-03T00:16:05.416943Z","iopub.status.idle":"2023-05-03T00:16:05.426640Z","shell.execute_reply.started":"2023-05-03T00:16:05.416891Z","shell.execute_reply":"2023-05-03T00:16:05.425688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"level_1 = train[train['level_group'] == '0-4']\nlevel_2 = train[train['level_group'] == '5-12']\nlevel_3 = train[train['level_group'] == '13-22']","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:05.429888Z","iopub.execute_input":"2023-05-03T00:16:05.430574Z","iopub.status.idle":"2023-05-03T00:16:08.105275Z","shell.execute_reply.started":"2023-05-03T00:16:05.430538Z","shell.execute_reply":"2023-05-03T00:16:08.104201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"slevel_1 = level_1.sample(n=5000)\nprint(slevel_1.shape)\nslevel_2 = level_2.sample(n=5000)\nprint(slevel_2.shape)\nslevel_3 = level_3.sample(n=5000)\nprint(slevel_3.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:08.106983Z","iopub.execute_input":"2023-05-03T00:16:08.107378Z","iopub.status.idle":"2023-05-03T00:16:08.999743Z","shell.execute_reply.started":"2023-05-03T00:16:08.107339Z","shell.execute_reply":"2023-05-03T00:16:08.998623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"strain= slevel_1.append(slevel_2)\nstrain = strain.append(slevel_3)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:09.001323Z","iopub.execute_input":"2023-05-03T00:16:09.001741Z","iopub.status.idle":"2023-05-03T00:16:09.021400Z","shell.execute_reply.started":"2023-05-03T00:16:09.001701Z","shell.execute_reply":"2023-05-03T00:16:09.020397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_encoder = preprocessing.LabelEncoder()\nfor feature in categorical_features:\n    strain[feature] = label_encoder.fit_transform(strain[feature])\n    #test[feature]=label_encoder.fit_transform(test[feature])","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:09.022809Z","iopub.execute_input":"2023-05-03T00:16:09.023477Z","iopub.status.idle":"2023-05-03T00:16:09.058827Z","shell.execute_reply.started":"2023-05-03T00:16:09.023439Z","shell.execute_reply":"2023-05-03T00:16:09.057956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shown = strain[cols].hist(figsize=(20,30), layout=(5,4))","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:09.063088Z","iopub.execute_input":"2023-05-03T00:16:09.063362Z","iopub.status.idle":"2023-05-03T00:16:11.926212Z","shell.execute_reply.started":"2023-05-03T00:16:09.063329Z","shell.execute_reply":"2023-05-03T00:16:11.925283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize = (8,8))\ncorr = strain[cols].corr()\nsns.barplot(data=corr,orient = 'h', errorbar = None)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:11.927364Z","iopub.execute_input":"2023-05-03T00:16:11.929184Z","iopub.status.idle":"2023-05-03T00:16:12.272815Z","shell.execute_reply.started":"2023-05-03T00:16:11.929144Z","shell.execute_reply":"2023-05-03T00:16:12.271853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.scatter(strain['room_coor_x'], strain['room_coor_y'])\nplt.show()\n\nplt.scatter(strain['screen_coor_x'], strain['screen_coor_y'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:12.274368Z","iopub.execute_input":"2023-05-03T00:16:12.274786Z","iopub.status.idle":"2023-05-03T00:16:12.749959Z","shell.execute_reply.started":"2023-05-03T00:16:12.274748Z","shell.execute_reply":"2023-05-03T00:16:12.748872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"strain.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:12.751654Z","iopub.execute_input":"2023-05-03T00:16:12.752012Z","iopub.status.idle":"2023-05-03T00:16:12.765536Z","shell.execute_reply.started":"2023-05-03T00:16:12.751975Z","shell.execute_reply":"2023-05-03T00:16:12.764642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"strain = strain[strain['level'] <= 18]\nstrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:12.769016Z","iopub.execute_input":"2023-05-03T00:16:12.769278Z","iopub.status.idle":"2023-05-03T00:16:12.792870Z","shell.execute_reply.started":"2023-05-03T00:16:12.769254Z","shell.execute_reply":"2023-05-03T00:16:12.791966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to_drop = ['page', 'hover_duration']\nto_mean = ['fqid', 'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'text', 'text_fqid']\n\n#hanlde na features(dropped)\nfor feature in to_drop:\n    strain.drop(feature, axis=1, inplace=True)\n    #test.drop(feature, axis=1, inplace=True)\n    \n#test.drop('session_level', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:12.794500Z","iopub.execute_input":"2023-05-03T00:16:12.794873Z","iopub.status.idle":"2023-05-03T00:16:12.803753Z","shell.execute_reply.started":"2023-05-03T00:16:12.794837Z","shell.execute_reply":"2023-05-03T00:16:12.803063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for feature in to_mean:\n    t_mean = strain[feature].mean()\n    #st_mean = test[feature].mean()\n    strain[feature] = strain[feature].fillna(t_mean)\n    #test[feature] = test[feature].fillna(st_mean)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:12.805193Z","iopub.execute_input":"2023-05-03T00:16:12.806228Z","iopub.status.idle":"2023-05-03T00:16:12.817093Z","shell.execute_reply.started":"2023-05-03T00:16:12.806191Z","shell.execute_reply":"2023-05-03T00:16:12.816036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged = strain.merge(labels, on=['level','session_id'])","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:12.820559Z","iopub.execute_input":"2023-05-03T00:16:12.820919Z","iopub.status.idle":"2023-05-03T00:16:12.887498Z","shell.execute_reply.started":"2023-05-03T00:16:12.820891Z","shell.execute_reply":"2023-05-03T00:16:12.886332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = merged['correct']\nX = merged.drop(['correct'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:12.889022Z","iopub.execute_input":"2023-05-03T00:16:12.890353Z","iopub.status.idle":"2023-05-03T00:16:12.899329Z","shell.execute_reply.started":"2023-05-03T00:16:12.890306Z","shell.execute_reply":"2023-05-03T00:16:12.898001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.30, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:12.901254Z","iopub.execute_input":"2023-05-03T00:16:12.901679Z","iopub.status.idle":"2023-05-03T00:16:12.911820Z","shell.execute_reply.started":"2023-05-03T00:16:12.901643Z","shell.execute_reply":"2023-05-03T00:16:12.910706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_selection import VarianceThreshold\n\nprint('Variance Threshold')\n\nselvt = VarianceThreshold(threshold=1)\nselvt = selvt.fit_transform(X_train)\nselvt","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:12.913743Z","iopub.execute_input":"2023-05-03T00:16:12.914444Z","iopub.status.idle":"2023-05-03T00:16:12.959456Z","shell.execute_reply.started":"2023-05-03T00:16:12.914404Z","shell.execute_reply":"2023-05-03T00:16:12.958417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HGBClf = HistGradientBoostingClassifier(max_bins=255, max_iter=100)\nHGBClf.fit(X_train, y_train)\nHGB_preds = HGBClf.predict(X_test)\n\n#gives accuracy 0.6626506024096386 without tuning\nSGDClf = linear_model.SGDClassifier(max_iter = 1000, tol=1e-3,penalty = \"elasticnet\")\nSGDClf.fit(X_train, y_train)\nSGD_preds = SGDClf.predict(X_test)\n\n#gives accuracy 0.5542168674698795 without tuning\ndtree = DecisionTreeClassifier(max_depth=10)\ndtree.fit(X_train, y_train)\ndtree_preds = dtree.predict(X_test)\n\n#create grid search params for each model\nHGBClf_params = {\n    \"max_iter\": [50, 100, 200, 500, 1000],\n    \"max_bins\": [50, 100, 200]\n}\n\nSGD_params = {\n    \"penalty\": ['l2', 'l1', 'elasticnet'],\n    \"tol\": [1e-3]\n}\n\ndtree_params = {\n    \"max_depth\": [2, 5, 10, 100],\n    \"min_samples_split\": [2, 4, 10, 20]\n}\n\n#perform grid search\nGS = GridSearchCV(SGDClf, param_grid=SGD_params, scoring=[\"accuracy\"], refit=\"accuracy\", cv=5)\nGS.fit(X_train, y_train)\nGS_preds = GS.predict(X_test)\n\n#show metrics\naccuracy = accuracy_score(y_test, GS_preds)\nf1 = f1_score(y_test, GS_preds)\nprint(\"\\naccuracy: \")\nprint(accuracy) \nprint(\"\\nf1: \")\nprint(f1)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:12.960892Z","iopub.execute_input":"2023-05-03T00:16:12.961546Z","iopub.status.idle":"2023-05-03T00:16:18.218770Z","shell.execute_reply.started":"2023-05-03T00:16:12.961512Z","shell.execute_reply":"2023-05-03T00:16:18.217736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:18.220349Z","iopub.execute_input":"2023-05-03T00:16:18.221067Z","iopub.status.idle":"2023-05-03T00:16:18.246824Z","shell.execute_reply.started":"2023-05-03T00:16:18.221030Z","shell.execute_reply":"2023-05-03T00:16:18.245439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter = 0\n# The API will deliver two dataframes in this specific order,\n# for every session+level grouping (one group per session for each checkpoint)\nfor (sample_submission, test) in iter_test:\n        \n    ## users make predictions here using the test data\n    t= sample_submission\n    le = preprocessing.LabelEncoder()\n    for feature in categorical_features:\n        t[feature]=le.fit_transform(t[feature])\n    for feature in to_drop:\n        t.drop(feature, axis=1, inplace=True)\n    \n    for feature in to_mean:\n        st_mean = t[feature].mean()\n        t[feature] = t[feature].fillna(st_mean)\n    \n    sample_submission['correct'] = GS.predict(t)\n    \n    ## env.predict appends the session+level sample_submission to the overall\n    ## submission\n    env.predict(test)\n    counter += 1","metadata":{"execution":{"iopub.status.busy":"2023-05-03T00:16:18.248454Z","iopub.execute_input":"2023-05-03T00:16:18.249078Z","iopub.status.idle":"2023-05-03T00:16:18.470189Z","shell.execute_reply.started":"2023-05-03T00:16:18.249037Z","shell.execute_reply":"2023-05-03T00:16:18.468607Z"},"trusted":true},"execution_count":null,"outputs":[]}]}