{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder, LabelEncoder\nfrom sklearn.ensemble import HistGradientBoostingClassifier\nfrom sklearn import linear_model\nfrom sklearn import preprocessing\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import f1_score\n#to ignore warning when using append instead of concat\nimport warnings\nwarnings.filterwarnings('ignore')\nimport gc","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:20:27.007002Z","iopub.execute_input":"2023-05-15T07:20:27.007308Z","iopub.status.idle":"2023-05-15T07:20:29.198863Z","shell.execute_reply.started":"2023-05-15T07:20:27.007279Z","shell.execute_reply":"2023-05-15T07:20:29.197570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\ntrain = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\nprint(train.shape)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:20:29.201597Z","iopub.execute_input":"2023-05-15T07:20:29.202392Z","iopub.status.idle":"2023-05-15T07:22:11.803850Z","shell.execute_reply.started":"2023-05-15T07:20:29.202347Z","shell.execute_reply":"2023-05-15T07:22:11.802503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\nprint(labels.shape)\n\nlabels['level'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nlabels['session_id'] = labels.session_id.apply(lambda x: int(x.split('_')[0]) )\n\nlabels.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:11.805783Z","iopub.execute_input":"2023-05-15T07:22:11.807991Z","iopub.status.idle":"2023-05-15T07:22:13.118906Z","shell.execute_reply.started":"2023-05-15T07:22:11.807940Z","shell.execute_reply":"2023-05-15T07:22:13.117753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"penambahan kolom level sesuai tingkat kesulitan dan + pembaruan koilom sesion_id pada label(file \"train_labels\")\n","metadata":{}},{"cell_type":"code","source":"#test = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/test.csv')\n#test.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:13.121802Z","iopub.execute_input":"2023-05-15T07:22:13.122260Z","iopub.status.idle":"2023-05-15T07:22:13.127419Z","shell.execute_reply.started":"2023-05-15T07:22:13.122222Z","shell.execute_reply":"2023-05-15T07:22:13.126379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = ['session_id', 'index', 'elapsed_time', 'event_name', 'name', 'level', 'page', \n            'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'hover_duration', \n            'text', 'fqid', 'room_fqid', 'text_fqid', 'fullscreen', 'hq', 'music', 'level_group']\n\ncategorical_features = ['event_name', 'name','text', 'fqid', 'room_fqid', 'text_fqid','level_group']\nnumeric_features = [col for col in train.columns if col not in categorical_features]","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:13.128819Z","iopub.execute_input":"2023-05-15T07:22:13.129847Z","iopub.status.idle":"2023-05-15T07:22:13.140360Z","shell.execute_reply.started":"2023-05-15T07:22:13.129809Z","shell.execute_reply":"2023-05-15T07:22:13.139320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(numeric_features)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:13.141908Z","iopub.execute_input":"2023-05-15T07:22:13.142363Z","iopub.status.idle":"2023-05-15T07:22:13.155006Z","shell.execute_reply.started":"2023-05-15T07:22:13.142322Z","shell.execute_reply":"2023-05-15T07:22:13.153842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"level_1 = train[train['level_group'] == '0-4']\nlevel_2 = train[train['level_group'] == '5-12']\nlevel_3 = train[train['level_group'] == '13-22']","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:13.156996Z","iopub.execute_input":"2023-05-15T07:22:13.157405Z","iopub.status.idle":"2023-05-15T07:22:15.846722Z","shell.execute_reply.started":"2023-05-15T07:22:13.157366Z","shell.execute_reply":"2023-05-15T07:22:15.845657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(level_1.shape\n     )","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:15.848169Z","iopub.execute_input":"2023-05-15T07:22:15.848839Z","iopub.status.idle":"2023-05-15T07:22:15.855010Z","shell.execute_reply.started":"2023-05-15T07:22:15.848797Z","shell.execute_reply":"2023-05-15T07:22:15.853833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"slevel_1 = level_1.sample(n=5000)\nprint(slevel_1.shape)\nslevel_2 = level_2.sample(n=5000)\nprint(slevel_2.shape)\nslevel_3 = level_3.sample(n=5000)\nprint(slevel_3.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:15.856642Z","iopub.execute_input":"2023-05-15T07:22:15.857396Z","iopub.status.idle":"2023-05-15T07:22:16.798990Z","shell.execute_reply.started":"2023-05-15T07:22:15.857347Z","shell.execute_reply":"2023-05-15T07:22:16.797721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"strain= slevel_1.append(slevel_2)\nstrain = strain.append(slevel_3)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:16.804324Z","iopub.execute_input":"2023-05-15T07:22:16.804654Z","iopub.status.idle":"2023-05-15T07:22:16.826322Z","shell.execute_reply.started":"2023-05-15T07:22:16.804622Z","shell.execute_reply":"2023-05-15T07:22:16.825263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_encoder = preprocessing.LabelEncoder()\nfor feature in categorical_features:\n    strain[feature] = label_encoder.fit_transform(strain[feature])\n    #test[feature]=label_encoder.fit_transform(test[feature])","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:16.828108Z","iopub.execute_input":"2023-05-15T07:22:16.828519Z","iopub.status.idle":"2023-05-15T07:22:16.875950Z","shell.execute_reply.started":"2023-05-15T07:22:16.828478Z","shell.execute_reply":"2023-05-15T07:22:16.874993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shown = strain[cols].hist(figsize=(20,30), layout=(5,4))","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:16.877717Z","iopub.execute_input":"2023-05-15T07:22:16.878101Z","iopub.status.idle":"2023-05-15T07:22:20.244822Z","shell.execute_reply.started":"2023-05-15T07:22:16.878050Z","shell.execute_reply":"2023-05-15T07:22:20.243651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize = (8,8))\ncorr = strain[cols].corr()\nsns.barplot(data=corr,orient = 'h', errorbar = None)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:20.246005Z","iopub.execute_input":"2023-05-15T07:22:20.246399Z","iopub.status.idle":"2023-05-15T07:22:20.622817Z","shell.execute_reply.started":"2023-05-15T07:22:20.246361Z","shell.execute_reply":"2023-05-15T07:22:20.621664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.scatter(strain['room_coor_x'], strain['room_coor_y'])\nplt.show()\n\nplt.scatter(strain['screen_coor_x'], strain['screen_coor_y'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:20.624757Z","iopub.execute_input":"2023-05-15T07:22:20.625208Z","iopub.status.idle":"2023-05-15T07:22:21.138817Z","shell.execute_reply.started":"2023-05-15T07:22:20.625162Z","shell.execute_reply":"2023-05-15T07:22:21.137632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"strain.isna().sum()\n#pembersihan data dengan isna","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:21.140622Z","iopub.execute_input":"2023-05-15T07:22:21.141016Z","iopub.status.idle":"2023-05-15T07:22:21.155607Z","shell.execute_reply.started":"2023-05-15T07:22:21.140976Z","shell.execute_reply":"2023-05-15T07:22:21.153728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"strain = strain[strain['level'] <= 18]\nstrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:21.157301Z","iopub.execute_input":"2023-05-15T07:22:21.157952Z","iopub.status.idle":"2023-05-15T07:22:21.186986Z","shell.execute_reply.started":"2023-05-15T07:22:21.157913Z","shell.execute_reply":"2023-05-15T07:22:21.185938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to_drop = ['page', 'hover_duration']\nto_mean = ['fqid', 'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'text', 'text_fqid']\n\n#hanlde na features(dropped)\nfor feature in to_drop:\n    strain.drop(feature, axis=1, inplace=True)\n    #test.drop(feature, axis=1, inplace=True)\n    \n#test.drop('session_level', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:21.190590Z","iopub.execute_input":"2023-05-15T07:22:21.190923Z","iopub.status.idle":"2023-05-15T07:22:21.200181Z","shell.execute_reply.started":"2023-05-15T07:22:21.190892Z","shell.execute_reply":"2023-05-15T07:22:21.198789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for feature in to_mean:\n    t_mean = strain[feature].mean()\n    #st_mean = test[feature].mean()\n    strain[feature] = strain[feature].fillna(t_mean)\n    #test[feature] = test[feature].fillna(st_mean)\nstrain[feature]","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:21.202370Z","iopub.execute_input":"2023-05-15T07:22:21.202807Z","iopub.status.idle":"2023-05-15T07:22:21.218782Z","shell.execute_reply.started":"2023-05-15T07:22:21.202765Z","shell.execute_reply":"2023-05-15T07:22:21.217528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged = strain.merge(labels, on=['level','session_id'])\n","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:21.220700Z","iopub.execute_input":"2023-05-15T07:22:21.221088Z","iopub.status.idle":"2023-05-15T07:22:21.284893Z","shell.execute_reply.started":"2023-05-15T07:22:21.221048Z","shell.execute_reply":"2023-05-15T07:22:21.283685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = merged['correct']\nX = merged.drop(['correct'],axis=1)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:26:07.964508Z","iopub.execute_input":"2023-05-15T07:26:07.965269Z","iopub.status.idle":"2023-05-15T07:26:07.973976Z","shell.execute_reply.started":"2023-05-15T07:26:07.965227Z","shell.execute_reply":"2023-05-15T07:26:07.972614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.30, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:21.298461Z","iopub.execute_input":"2023-05-15T07:22:21.298947Z","iopub.status.idle":"2023-05-15T07:22:21.309964Z","shell.execute_reply.started":"2023-05-15T07:22:21.298903Z","shell.execute_reply":"2023-05-15T07:22:21.308740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_selection import VarianceThreshold\n\nprint('Variance Threshold')\n\nselvt = VarianceThreshold(threshold=1)\nselvt = selvt.fit_transform(X_train)\nselvt","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:21.311773Z","iopub.execute_input":"2023-05-15T07:22:21.312335Z","iopub.status.idle":"2023-05-15T07:22:21.369289Z","shell.execute_reply.started":"2023-05-15T07:22:21.312288Z","shell.execute_reply":"2023-05-15T07:22:21.368162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HGBClf = HistGradientBoostingClassifier(max_bins=255, max_iter=100)\nHGBClf.fit(X_train, y_train)\nHGB_preds = HGBClf.predict(X_test)\n\n#gives accuracy 0.6626506024096386 without tuning\nSGDClf = linear_model.SGDClassifier(max_iter = 1000, tol=1e-3,penalty = \"elasticnet\")\nSGDClf.fit(X_train, y_train)\nSGD_preds = SGDClf.predict(X_test)\n\n#gives accuracy 0.5542168674698795 without tuning\ndtree = DecisionTreeClassifier(max_depth=10)\ndtree.fit(X_train, y_train)\ndtree_preds = dtree.predict(X_test)\n\n#create grid search params for each model\nHGBClf_params = {\n    \"max_iter\": [50, 100, 200, 500, 1000],\n    \"max_bins\": [50, 100, 200]\n}\n\nSGD_params = {\n    \"penalty\": ['l2', 'l1', 'elasticnet'],\n    \"tol\": [1e-3]\n}\n\ndtree_params = {\n    \"max_depth\": [2, 5, 10, 100],\n    \"min_samples_split\": [2, 4, 10, 20]\n}\n\n#perform grid search\nGS = GridSearchCV(SGDClf, param_grid=SGD_params, scoring=[\"accuracy\"], refit=\"accuracy\", cv=5)\nGS.fit(X_train, y_train)\nGS_preds = GS.predict(X_test)\n\n#show metrics\naccuracy = accuracy_score(y_test, GS_preds)\nf1 = f1_score(y_test, GS_preds)\nprint(\"\\naccuracy: \")\nprint(accuracy) \nprint(\"\\nf1: \")\nprint(f1)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:21.370617Z","iopub.execute_input":"2023-05-15T07:22:21.371629Z","iopub.status.idle":"2023-05-15T07:22:27.026960Z","shell.execute_reply.started":"2023-05-15T07:22:21.371587Z","shell.execute_reply":"2023-05-15T07:22:27.025723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Kode di atas merupakan implementasi dari beberapa algoritma machine learning untuk melakukan klasifikasi pada data.\na HistGradientBoostingClassifier dari scikit-learn yang menggunakan model gradient boosting untuk melakukan klasifikasi pada data. Dalam implementasinya, model ini diatur menggunakan hyperparameter max_bins sebesar 255 dan max_iter sebesar 100. Model kemudian di-fit pada data training X_train dan y_train, dan kemudian digunakan untuk melakukan prediksi pada data testing X_test dengan menyimpan hasil prediksi pada variabel HGB_preds","metadata":{}},{"cell_type":"code","source":"import jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:27.028556Z","iopub.execute_input":"2023-05-15T07:22:27.029690Z","iopub.status.idle":"2023-05-15T07:22:27.060199Z","shell.execute_reply.started":"2023-05-15T07:22:27.029629Z","shell.execute_reply":"2023-05-15T07:22:27.058651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter = 0\n# The API will deliver two dataframes in this specific order,\n# for every session+level grouping (one group per session for each checkpoint)\nfor (sample_submission, test) in iter_test:\n        \n    ## users make predictions here using the test data\n    t= sample_submission\n    le = preprocessing.LabelEncoder()\n    for feature in categorical_features:\n        t[feature]=le.fit_transform(t[feature])\n    for feature in to_drop:\n        t.drop(feature, axis=1, inplace=True)\n    \n    for feature in to_mean:\n        st_mean = t[feature].mean()\n        t[feature] = t[feature].fillna(st_mean)\n    \n    sample_submission['correct'] = GS.predict(t)\n    \n    ## env.predict appends the session+level sample_submission to the overall\n    ## submission\n    env.predict(test)\n    counter += 1","metadata":{"execution":{"iopub.status.busy":"2023-05-15T07:22:27.061732Z","iopub.execute_input":"2023-05-15T07:22:27.062421Z","iopub.status.idle":"2023-05-15T07:22:27.291486Z","shell.execute_reply.started":"2023-05-15T07:22:27.062377Z","shell.execute_reply":"2023-05-15T07:22:27.290023Z"},"trusted":true},"execution_count":null,"outputs":[]}]}