{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-14T00:22:56.824740Z","iopub.execute_input":"2023-05-14T00:22:56.825078Z","iopub.status.idle":"2023-05-14T00:22:56.875287Z","shell.execute_reply.started":"2023-05-14T00:22:56.825049Z","shell.execute_reply":"2023-05-14T00:22:56.874447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tmp = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\",usecols=[0]) #first we are just getting the user id","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:22:56.877248Z","iopub.execute_input":"2023-05-14T00:22:56.877790Z","iopub.status.idle":"2023-05-14T00:22:56.881115Z","shell.execute_reply.started":"2023-05-14T00:22:56.877760Z","shell.execute_reply":"2023-05-14T00:22:56.880404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tmp = tmp.groupby('session_id').session_id.agg('count') #then we are grouping all the same ids together and showing the total amount they appear. Example: id: 20090312431273200 appears 881 times","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:22:56.882686Z","iopub.execute_input":"2023-05-14T00:22:56.883178Z","iopub.status.idle":"2023-05-14T00:22:56.892828Z","shell.execute_reply.started":"2023-05-14T00:22:56.883150Z","shell.execute_reply":"2023-05-14T00:22:56.892163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nimport xgboost as xgb\nfrom sklearn.metrics import f1_score\nimport pickle","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:22:56.894314Z","iopub.execute_input":"2023-05-14T00:22:56.894740Z","iopub.status.idle":"2023-05-14T00:22:57.603921Z","shell.execute_reply.started":"2023-05-14T00:22:56.894714Z","shell.execute_reply":"2023-05-14T00:22:57.602920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PIECES = 10\n# CHUNK = int( np.ceil(len(tmp)/PIECES) ) #there are 23,562 ids, and we want to divide it by 10. Thus, 1/10 is 2357","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:22:57.605943Z","iopub.execute_input":"2023-05-14T00:22:57.606321Z","iopub.status.idle":"2023-05-14T00:22:57.608819Z","shell.execute_reply.started":"2023-05-14T00:22:57.606295Z","shell.execute_reply":"2023-05-14T00:22:57.608256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reads = []\n# skips = [0]\n# for k in range(PIECES):\n#     a = k*CHUNK #start of current fraction\n#     b = (k+1)*CHUNK #end of current fraction\n#     if b>len(tmp): b=len(tmp) #if b is greater than len(tmp) == position + 1, then use len(tmp)\n#     r = tmp.iloc[a:b].sum() #get sum from start of fraction to (end of fraction) - 1\n#     reads.append(r)\n#     skips.append(skips[-1]+r)","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:22:57.609759Z","iopub.execute_input":"2023-05-14T00:22:57.610116Z","iopub.status.idle":"2023-05-14T00:22:57.621707Z","shell.execute_reply.started":"2023-05-14T00:22:57.610091Z","shell.execute_reply":"2023-05-14T00:22:57.620592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(f'To avoid memory error, we will read train in {PIECES} pieces of sizes:')\n# print(skips) #this shows how many we have to skip\n# print(reads) #shows the size of each piece","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:22:57.623881Z","iopub.execute_input":"2023-05-14T00:22:57.624732Z","iopub.status.idle":"2023-05-14T00:22:57.638618Z","shell.execute_reply.started":"2023-05-14T00:22:57.624656Z","shell.execute_reply":"2023-05-14T00:22:57.637199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', nrows=reads[0])\n# #first batch, 2684191 rows","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:22:57.639977Z","iopub.execute_input":"2023-05-14T00:22:57.640240Z","iopub.status.idle":"2023-05-14T00:22:57.651660Z","shell.execute_reply.started":"2023-05-14T00:22:57.640215Z","shell.execute_reply":"2023-05-14T00:22:57.650078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# targets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\n# #this is if the answer is correct or not for each id-question pair","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:22:57.653753Z","iopub.execute_input":"2023-05-14T00:22:57.654261Z","iopub.status.idle":"2023-05-14T00:22:57.664451Z","shell.execute_reply.started":"2023-05-14T00:22:57.654233Z","shell.execute_reply":"2023-05-14T00:22:57.663279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# targets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]) ) #remove _xx in session_id and make a new row called session","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:22:57.665491Z","iopub.execute_input":"2023-05-14T00:22:57.665861Z","iopub.status.idle":"2023-05-14T00:22:57.677030Z","shell.execute_reply.started":"2023-05-14T00:22:57.665836Z","shell.execute_reply":"2023-05-14T00:22:57.676107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# targets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]) ) #get the question number from session_id and make a new role called q","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:22:57.679666Z","iopub.execute_input":"2023-05-14T00:22:57.680664Z","iopub.status.idle":"2023-05-14T00:22:57.688564Z","shell.execute_reply.started":"2023-05-14T00:22:57.680611Z","shell.execute_reply":"2023-05-14T00:22:57.687495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #making a list of strings for categories, numbers, and events\n# CATS = ['event_name', 'fqid', 'room_fqid', 'text']\n\n# NUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y',\n#         'screen_coor_x', 'screen_coor_y', 'hover_duration']\n\n# EVENTS = ['navigate_click','person_click','cutscene_click','object_click',\n#           'map_hover','notification_click','map_click','observation_click',\n#           'checkpoint']","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:22:57.689851Z","iopub.execute_input":"2023-05-14T00:22:57.690083Z","iopub.status.idle":"2023-05-14T00:22:57.700646Z","shell.execute_reply.started":"2023-05-14T00:22:57.690060Z","shell.execute_reply":"2023-05-14T00:22:57.698928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def feature_engineer(train):\n\n#     dfs = []\n#     for c in CATS: #go from 0 to (len(cats))-1 on and from start and each step including len(cats)-1 let c be the string at the given position\n#         tmp = train.groupby(['session_id','level_group'])[c].agg('nunique') #find how many times a given level group for a certain id has event_name, fqid, room_fqid, and text occur\n#         tmp.name = tmp.name + '_nunique'\n#         dfs.append(tmp) # add the dataframe\n#     for c in NUMS: #same idea as c in Cats\n#         tmp = train.groupby(['session_id','level_group'])[c].agg('mean') #find the mean elapsed time, level, page, ... for a given session_ids level group\n#         tmp.name = tmp.name + '_mean'\n#         dfs.append(tmp)\n#     for c in NUMS:\n#         tmp = train.groupby(['session_id','level_group'])[c].agg('std') #find the standard deviation for elapsed_time, level, ... for a given session_ids level group\n#         tmp.name = tmp.name + '_std'\n#         dfs.append(tmp)\n#     for c in EVENTS: #here we are separating all the events into their separate rows\n#         train[c] = (train.event_name == c).astype('int8')\n#     for c in EVENTS + ['elapsed_time']: #now we are getting the sum for each event for a unique id and elapsed_time\n#         tmp = train.groupby(['session_id','level_group'])[c].agg('sum')\n#         tmp.name = tmp.name + '_sum'\n#         dfs.append(tmp)\n#     train = train.drop(EVENTS,axis=1)\n\n#     df = pd.concat(dfs,axis=1)\n#     df = df.fillna(-1)\n#     df = df.reset_index()\n#     # df = df.set_index('session_id')\n#     return df","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:22:57.702023Z","iopub.execute_input":"2023-05-14T00:22:57.702314Z","iopub.status.idle":"2023-05-14T00:22:57.714752Z","shell.execute_reply.started":"2023-05-14T00:22:57.702288Z","shell.execute_reply":"2023-05-14T00:22:57.712932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n\n# # PROCESS TRAIN DATA IN PIECES\n# all_pieces = []\n# print(f'Processing train as {PIECES} pieces to avoid memory error... ')\n# for k in range(PIECES):\n#     print(k,', ',end='')\n#     SKIPS = 0\n#     if k>0: SKIPS = range(1,skips[k]+1)\n#     train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv',\n#                         nrows=reads[k], skiprows=SKIPS)\n#     df = feature_engineer(train)\n#     all_pieces.append(df)","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:22:57.720298Z","iopub.execute_input":"2023-05-14T00:22:57.720604Z","iopub.status.idle":"2023-05-14T00:22:57.730496Z","shell.execute_reply.started":"2023-05-14T00:22:57.720579Z","shell.execute_reply":"2023-05-14T00:22:57.729499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # CONCATENATE ALL PIECES INTO ONE DATAFRAME\n# import gc\n# del train; gc.collect()\n# df = pd.concat(all_pieces, axis=0)\n# print('Shape of all train data after feature engineering:', df.shape )","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:22:57.732219Z","iopub.execute_input":"2023-05-14T00:22:57.732544Z","iopub.status.idle":"2023-05-14T00:22:57.745618Z","shell.execute_reply.started":"2023-05-14T00:22:57.732517Z","shell.execute_reply":"2023-05-14T00:22:57.744261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/predict-student-performance/data.csv')\ndf.to_csv('data.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:22:57.747487Z","iopub.execute_input":"2023-05-14T00:22:57.747817Z","iopub.status.idle":"2023-05-14T00:22:59.806558Z","shell.execute_reply.started":"2023-05-14T00:22:57.747788Z","shell.execute_reply":"2023-05-14T00:22:59.805162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:22:59.808990Z","iopub.execute_input":"2023-05-14T00:22:59.809398Z","iopub.status.idle":"2023-05-14T00:22:59.860115Z","shell.execute_reply.started":"2023-05-14T00:22:59.809370Z","shell.execute_reply":"2023-05-14T00:22:59.859175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv('/kaggle/input/predict-student-performance/targets.csv')\ntargets.to_csv('targets.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:22:59.861108Z","iopub.execute_input":"2023-05-14T00:22:59.861355Z","iopub.status.idle":"2023-05-14T00:23:01.054355Z","shell.execute_reply.started":"2023-05-14T00:22:59.861332Z","shell.execute_reply":"2023-05-14T00:23:01.052848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_tmp = df.loc[df['level_group'] == '0-4']\ny_tmp = targets.loc[targets['q'] == 1]\ny_tmp = y_tmp['correct']\n\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X_tmp, y_tmp, test_size=0.1, stratify=y_tmp, random_state=0)\n\n#removing level_group from both training and testing group\nX_train.pop('level_group')\nX_test.pop('level_group')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:01.055632Z","iopub.execute_input":"2023-05-14T00:23:01.056828Z","iopub.status.idle":"2023-05-14T00:23:01.097147Z","shell.execute_reply.started":"2023-05-14T00:23:01.056768Z","shell.execute_reply":"2023-05-14T00:23:01.095353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pickle.dump(pipe_xgboost_1,open('model_1.sav','wb'))#saving model to a .sav file for later use\nmodel_1=pickle.load(open('/kaggle/input/predict-student-performance/model_1.sav', 'rb')) #loading model to model_1\n#demonstrating that it is working properly\ny_predict = model_1.predict(X_test)\nprint('question 1 f1_score: ' + str(f1_score(y_test, y_predict)))#cv train had 0.8424909787934283 f1 score, thus good and no overfitting\npickle.dump(model_1,open('model_1.sav','wb'))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:01.099047Z","iopub.execute_input":"2023-05-14T00:23:01.099381Z","iopub.status.idle":"2023-05-14T00:23:01.214913Z","shell.execute_reply.started":"2023-05-14T00:23:01.099355Z","shell.execute_reply":"2023-05-14T00:23:01.214244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#showing how to store the model into a list of xgboost models\nmodels = [model_1] #making a list of models to store\nmodels","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:01.217985Z","iopub.execute_input":"2023-05-14T00:23:01.219540Z","iopub.status.idle":"2023-05-14T00:23:01.250696Z","shell.execute_reply.started":"2023-05-14T00:23:01.219508Z","shell.execute_reply":"2023-05-14T00:23:01.249629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#make model for question 2\nX_tmp = df.loc[df['level_group'] == '0-4']\ny_tmp = targets.loc[targets['q'] == 2]\ny_tmp = y_tmp['correct']\nX_train, X_test, y_train, y_test = train_test_split(X_tmp, y_tmp, test_size=0.1, stratify=y_tmp, random_state=0)\nX_train.pop('level_group')\nX_test.pop('level_group')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:01.254900Z","iopub.execute_input":"2023-05-14T00:23:01.256581Z","iopub.status.idle":"2023-05-14T00:23:01.292626Z","shell.execute_reply.started":"2023-05-14T00:23:01.256542Z","shell.execute_reply":"2023-05-14T00:23:01.290920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pickle.dump(pipe_xgboost_2,open('model_2.sav','wb'))\nmodel_2=pickle.load(open('/kaggle/input/predict-student-performance/model_2.sav', 'rb'))\nmodels.append(model_2)\ny_predict = model_2.predict(X_test)\nprint('question 2 f1_score: ' + str(f1_score(y_test, y_predict)))\npickle.dump(model_2,open('model_2.sav','wb'))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:01.293898Z","iopub.execute_input":"2023-05-14T00:23:01.294397Z","iopub.status.idle":"2023-05-14T00:23:01.392474Z","shell.execute_reply.started":"2023-05-14T00:23:01.294368Z","shell.execute_reply":"2023-05-14T00:23:01.391521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#make model for question 3\nX_tmp = df.loc[df['level_group'] == '0-4']\ny_tmp = targets.loc[targets['q'] == 3]\ny_tmp = y_tmp['correct']\nX_train, X_test, y_train, y_test = train_test_split(X_tmp, y_tmp, test_size=0.1, stratify=y_tmp, random_state=0)\nX_train.pop('level_group')\nX_test.pop('level_group')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:01.395935Z","iopub.execute_input":"2023-05-14T00:23:01.397457Z","iopub.status.idle":"2023-05-14T00:23:01.431005Z","shell.execute_reply.started":"2023-05-14T00:23:01.397427Z","shell.execute_reply":"2023-05-14T00:23:01.429772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pickle.dump(pipe_xgboost_3,open('model_3.sav','wb'))\nmodel_3=pickle.load(open('/kaggle/input/predict-student-performance/model_3.sav', 'rb'))\nmodels.append(model_3)\ny_predict = model_3.predict(X_test)\nprint('question 3 f1_score: ' + str(f1_score(y_test, y_predict)))\npickle.dump(model_3,open('model_3.sav','wb'))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:01.434595Z","iopub.execute_input":"2023-05-14T00:23:01.436399Z","iopub.status.idle":"2023-05-14T00:23:01.534799Z","shell.execute_reply.started":"2023-05-14T00:23:01.436363Z","shell.execute_reply":"2023-05-14T00:23:01.533807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#level_groups '0-4' to train model for questions 1-3\n#level groups '5-12' to train questions 4 thru 13\n#level groups '13-22' to train questions 14 thru 18","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:01.538331Z","iopub.execute_input":"2023-05-14T00:23:01.539989Z","iopub.status.idle":"2023-05-14T00:23:01.545342Z","shell.execute_reply.started":"2023-05-14T00:23:01.539955Z","shell.execute_reply":"2023-05-14T00:23:01.544232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#make model for question 4\nX_tmp = df.loc[df['level_group'] == '5-12']\ny_tmp = targets.loc[targets['q'] == 4]\ny_tmp = y_tmp['correct']\nX_train, X_test, y_train, y_test = train_test_split(X_tmp, y_tmp, test_size=0.1, stratify=y_tmp, random_state=0)\nX_train.pop('level_group')\nX_test.pop('level_group')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:01.548355Z","iopub.execute_input":"2023-05-14T00:23:01.548827Z","iopub.status.idle":"2023-05-14T00:23:01.586598Z","shell.execute_reply.started":"2023-05-14T00:23:01.548799Z","shell.execute_reply":"2023-05-14T00:23:01.585571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pickle.dump(pipe_xgboost_4,open('model_4.sav','wb'))\nmodel_4=pickle.load(open('/kaggle/input/predict-student-performance/model_4.sav', 'rb'))\nmodels.append(model_4)\ny_predict = model_4.predict(X_test)\nprint('question 4 f1_score: ' + str(f1_score(y_test, y_predict)))\npickle.dump(model_4,open('model_4.sav','wb'))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:01.589745Z","iopub.execute_input":"2023-05-14T00:23:01.592688Z","iopub.status.idle":"2023-05-14T00:23:01.683820Z","shell.execute_reply.started":"2023-05-14T00:23:01.592637Z","shell.execute_reply":"2023-05-14T00:23:01.683083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#make model for question 5\nX_tmp = df.loc[df['level_group'] == '5-12']\ny_tmp = targets.loc[targets['q'] == 5]\ny_tmp = y_tmp['correct']\nX_train, X_test, y_train, y_test = train_test_split(X_tmp, y_tmp, test_size=0.1, stratify=y_tmp, random_state=0)\nX_train.pop('level_group')\nX_test.pop('level_group')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:01.686267Z","iopub.execute_input":"2023-05-14T00:23:01.686592Z","iopub.status.idle":"2023-05-14T00:23:01.716619Z","shell.execute_reply.started":"2023-05-14T00:23:01.686566Z","shell.execute_reply":"2023-05-14T00:23:01.715872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#picle implementation\n# pickle.dump(pipe_xgboost_5,open('model_5.sav','wb'))\nmodel_5=pickle.load(open('/kaggle/input/predict-student-performance/model_5.sav', 'rb'))\nmodels.append(model_5)\ny_predict = model_5.predict(X_test)\nprint('question 5 f1_score: ' + str(f1_score(y_test, y_predict)))\npickle.dump(model_5,open('model_5.sav','wb'))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:01.719453Z","iopub.execute_input":"2023-05-14T00:23:01.721501Z","iopub.status.idle":"2023-05-14T00:23:01.812570Z","shell.execute_reply.started":"2023-05-14T00:23:01.721464Z","shell.execute_reply":"2023-05-14T00:23:01.811928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#getting data for question 6\nX_tmp = df.loc[df['level_group'] == '5-12']\ny_tmp = targets.loc[targets['q'] == 6]\ny_tmp = y_tmp['correct']\nX_train, X_test, y_train, y_test = train_test_split(X_tmp, y_tmp, test_size=0.1, stratify=y_tmp, random_state=0)\nX_train.pop('level_group')\nX_test.pop('level_group')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:01.813436Z","iopub.execute_input":"2023-05-14T00:23:01.813791Z","iopub.status.idle":"2023-05-14T00:23:01.842594Z","shell.execute_reply.started":"2023-05-14T00:23:01.813767Z","shell.execute_reply":"2023-05-14T00:23:01.841890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #defining pipeline for question 6\n# pipe_xgboost_6 = make_pipeline(StandardScaler(), PCA(n_components=13, random_state=0), xgb.XGBClassifier(random_state=1))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:01.843533Z","iopub.execute_input":"2023-05-14T00:23:01.843877Z","iopub.status.idle":"2023-05-14T00:23:01.848790Z","shell.execute_reply.started":"2023-05-14T00:23:01.843853Z","shell.execute_reply":"2023-05-14T00:23:01.847860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# #hyperparameter training for question 6\n# learning_rate = [0.001, 0.01, 0.1]\n# max_depth = [2, 3, 4, 5]\n# param_grid = [{'xgbclassifier__n_estimators': [1000], 'xgbclassifier__learning_rate': learning_rate, 'xgbclassifier__max_depth': max_depth}]\n# gs = GridSearchCV(estimator=pipe_xgboost_6,\n#                  param_grid=param_grid,\n#                  scoring='f1',\n#                  refit=True,\n#                  cv=5,\n#                  n_jobs=-1)\n# gs = gs.fit(X_train, y_train)\n# print(gs.best_score_)\n# print(gs.best_params_)","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:01.851349Z","iopub.execute_input":"2023-05-14T00:23:01.851936Z","iopub.status.idle":"2023-05-14T00:23:01.859232Z","shell.execute_reply.started":"2023-05-14T00:23:01.851910Z","shell.execute_reply":"2023-05-14T00:23:01.857921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #get test data accuracy for 6 to determine performance on the dataset\n# y_predict = pipe_xgboost_6.predict(X_test)\n# print('question 6 f1_score: ' + str(f1_score(y_test, y_predict)))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:01.860365Z","iopub.execute_input":"2023-05-14T00:23:01.860767Z","iopub.status.idle":"2023-05-14T00:23:01.870872Z","shell.execute_reply.started":"2023-05-14T00:23:01.860717Z","shell.execute_reply":"2023-05-14T00:23:01.869863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pickle.dump(pipe_xgboost_6,open('model_6.sav','wb'))\nmodel_6=pickle.load(open('/kaggle/input/predict-student-performance/model_6.sav', 'rb'))\nmodels.append(model_6)\ny_predict = model_6.predict(X_test)\nprint('question 6 f1_score: ' + str(f1_score(y_test, y_predict)))\npickle.dump(model_6,open('model_6.sav','wb'))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:01.872052Z","iopub.execute_input":"2023-05-14T00:23:01.872957Z","iopub.status.idle":"2023-05-14T00:23:01.980440Z","shell.execute_reply.started":"2023-05-14T00:23:01.872927Z","shell.execute_reply":"2023-05-14T00:23:01.979812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#getting data for question 7\nX_tmp = df.loc[df['level_group'] == '5-12']\ny_tmp = targets.loc[targets['q'] == 7]\ny_tmp = y_tmp['correct']\nX_train, X_test, y_train, y_test = train_test_split(X_tmp, y_tmp, test_size=0.1, stratify=y_tmp, random_state=0)\nX_train.pop('level_group')\nX_test.pop('level_group')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:01.981463Z","iopub.execute_input":"2023-05-14T00:23:01.981889Z","iopub.status.idle":"2023-05-14T00:23:02.013155Z","shell.execute_reply.started":"2023-05-14T00:23:01.981863Z","shell.execute_reply":"2023-05-14T00:23:02.012149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pickle.dump(pipe_xgboost_7,open('model_7.sav','wb'))\nmodel_7=pickle.load(open('/kaggle/input/predict-student-performance/model_7.sav', 'rb'))\nmodels.append(model_7)\ny_predict = model_7.predict(X_test)\nprint('question 7 f1_score: ' + str(f1_score(y_test, y_predict)))\npickle.dump(model_7,open('model_7.sav','wb'))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:02.014365Z","iopub.execute_input":"2023-05-14T00:23:02.014823Z","iopub.status.idle":"2023-05-14T00:23:02.121146Z","shell.execute_reply.started":"2023-05-14T00:23:02.014789Z","shell.execute_reply":"2023-05-14T00:23:02.120449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#getting data for question 8\nX_tmp = df.loc[df['level_group'] == '5-12']\ny_tmp = targets.loc[targets['q'] == 8]\ny_tmp = y_tmp['correct']\nX_train, X_test, y_train, y_test = train_test_split(X_tmp, y_tmp, test_size=0.1, stratify=y_tmp, random_state=0)\nX_train.pop('level_group')\nX_test.pop('level_group')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:02.122233Z","iopub.execute_input":"2023-05-14T00:23:02.122645Z","iopub.status.idle":"2023-05-14T00:23:02.152487Z","shell.execute_reply.started":"2023-05-14T00:23:02.122618Z","shell.execute_reply":"2023-05-14T00:23:02.151836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pickle.dump(pipe_xgboost_8,open('model_8.sav','wb'))\nmodel_8=pickle.load(open('/kaggle/input/predict-student-performance/model_8.sav', 'rb'))\nmodels.append(model_8)\ny_predict = model_8.predict(X_test)\nprint('question 8 f1_score: ' + str(f1_score(y_test, y_predict)))\npickle.dump(model_8,open('model_8.sav','wb'))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:02.153622Z","iopub.execute_input":"2023-05-14T00:23:02.154038Z","iopub.status.idle":"2023-05-14T00:23:02.247803Z","shell.execute_reply.started":"2023-05-14T00:23:02.154013Z","shell.execute_reply":"2023-05-14T00:23:02.247162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#getting data for question 9\nX_tmp = df.loc[df['level_group'] == '5-12']\ny_tmp = targets.loc[targets['q'] == 9]\ny_tmp = y_tmp['correct']\nX_train, X_test, y_train, y_test = train_test_split(X_tmp, y_tmp, test_size=0.1, stratify=y_tmp, random_state=0)\nX_train.pop('level_group')\nX_test.pop('level_group')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:02.250743Z","iopub.execute_input":"2023-05-14T00:23:02.252231Z","iopub.status.idle":"2023-05-14T00:23:02.284990Z","shell.execute_reply.started":"2023-05-14T00:23:02.252199Z","shell.execute_reply":"2023-05-14T00:23:02.284306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pickle.dump(pipe_xgboost_9,open('model_9.sav','wb'))\nmodel_9=pickle.load(open('/kaggle/input/predict-student-performance/model_9.sav', 'rb'))\nmodels.append(model_9)\ny_predict = model_9.predict(X_test)\nprint('question 9 f1_score: ' + str(f1_score(y_test, y_predict)))\npickle.dump(model_9,open('model_9.sav','wb'))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:02.293209Z","iopub.execute_input":"2023-05-14T00:23:02.295029Z","iopub.status.idle":"2023-05-14T00:23:02.379968Z","shell.execute_reply.started":"2023-05-14T00:23:02.294996Z","shell.execute_reply":"2023-05-14T00:23:02.379289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#getting data for question 10\nX_tmp = df.loc[df['level_group'] == '5-12']\ny_tmp = targets.loc[targets['q'] == 10]\ny_tmp = y_tmp['correct']\nX_train, X_test, y_train, y_test = train_test_split(X_tmp, y_tmp, test_size=0.1, stratify=y_tmp, random_state=0)\nX_train.pop('level_group')\nX_test.pop('level_group')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:02.382559Z","iopub.execute_input":"2023-05-14T00:23:02.383278Z","iopub.status.idle":"2023-05-14T00:23:02.416304Z","shell.execute_reply.started":"2023-05-14T00:23:02.383221Z","shell.execute_reply":"2023-05-14T00:23:02.415494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pickle.dump(pipe_xgboost_10,open('model_10.sav','wb'))\nmodel_10=pickle.load(open('/kaggle/input/predict-student-performance/model_10.sav', 'rb'))\nmodels.append(model_10)\ny_predict = model_10.predict(X_test)\nprint('question 10 f1_score: ' + str(f1_score(y_test, y_predict)))\npickle.dump(model_10,open('model_10.sav','wb'))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:02.419242Z","iopub.execute_input":"2023-05-14T00:23:02.421263Z","iopub.status.idle":"2023-05-14T00:23:02.508342Z","shell.execute_reply.started":"2023-05-14T00:23:02.421230Z","shell.execute_reply":"2023-05-14T00:23:02.507652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#getting data for question 11\nX_tmp = df.loc[df['level_group'] == '5-12']\ny_tmp = targets.loc[targets['q'] == 11]\ny_tmp = y_tmp['correct']\nX_train, X_test, y_train, y_test = train_test_split(X_tmp, y_tmp, test_size=0.1, stratify=y_tmp, random_state=0)\nX_train.pop('level_group')\nX_test.pop('level_group')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:02.511287Z","iopub.execute_input":"2023-05-14T00:23:02.511967Z","iopub.status.idle":"2023-05-14T00:23:02.545522Z","shell.execute_reply.started":"2023-05-14T00:23:02.511939Z","shell.execute_reply":"2023-05-14T00:23:02.544445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pickle.dump(pipe_xgboost_11,open('model_11.sav','wb'))\nmodel_11=pickle.load(open('/kaggle/input/predict-student-performance/model_11.sav', 'rb'))\nmodels.append(model_11)\ny_predict = model_11.predict(X_test)\nprint('question 11 f1_score: ' + str(f1_score(y_test, y_predict)))\npickle.dump(model_11,open('model_11.sav','wb'))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:02.548891Z","iopub.execute_input":"2023-05-14T00:23:02.549755Z","iopub.status.idle":"2023-05-14T00:23:02.664684Z","shell.execute_reply.started":"2023-05-14T00:23:02.549695Z","shell.execute_reply":"2023-05-14T00:23:02.663756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#getting data for question 12\nX_tmp = df.loc[df['level_group'] == '5-12']\ny_tmp = targets.loc[targets['q'] == 12]\ny_tmp = y_tmp['correct']\nX_train, X_test, y_train, y_test = train_test_split(X_tmp, y_tmp, test_size=0.1, stratify=y_tmp, random_state=0)\nX_train.pop('level_group')\nX_test.pop('level_group')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:02.666025Z","iopub.execute_input":"2023-05-14T00:23:02.666279Z","iopub.status.idle":"2023-05-14T00:23:02.699507Z","shell.execute_reply.started":"2023-05-14T00:23:02.666254Z","shell.execute_reply":"2023-05-14T00:23:02.698867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pickle.dump(pipe_xgboost_12,open('model_12.sav','wb'))\nmodel_12=pickle.load(open('/kaggle/input/predict-student-performance/model_12.sav', 'rb'))\nmodels.append(model_12)\ny_predict = model_12.predict(X_test)\nprint('question 12 f1_score: ' + str(f1_score(y_test, y_predict)))\npickle.dump(model_12,open('model_12.sav','wb'))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:02.702370Z","iopub.execute_input":"2023-05-14T00:23:02.703814Z","iopub.status.idle":"2023-05-14T00:23:02.790514Z","shell.execute_reply.started":"2023-05-14T00:23:02.703788Z","shell.execute_reply":"2023-05-14T00:23:02.789420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#getting data for question 13\nX_tmp = df.loc[df['level_group'] == '5-12']\ny_tmp = targets.loc[targets['q'] == 13]\ny_tmp = y_tmp['correct']\nX_train, X_test, y_train, y_test = train_test_split(X_tmp, y_tmp, test_size=0.1, stratify=y_tmp, random_state=0)\nX_train.pop('level_group')\nX_test.pop('level_group')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:02.793947Z","iopub.execute_input":"2023-05-14T00:23:02.795658Z","iopub.status.idle":"2023-05-14T00:23:02.828559Z","shell.execute_reply.started":"2023-05-14T00:23:02.795623Z","shell.execute_reply":"2023-05-14T00:23:02.827899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pickle.dump(pipe_xgboost_13,open('model_13.sav','wb'))\nmodel_13=pickle.load(open('/kaggle/input/predict-student-performance/model_13.sav', 'rb'))\nmodels.append(model_13)\ny_predict = model_13.predict(X_test)\nprint('question 13 f1_score: ' + str(f1_score(y_test, y_predict)))\npickle.dump(model_13,open('model_13.sav','wb'))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:02.831553Z","iopub.execute_input":"2023-05-14T00:23:02.833109Z","iopub.status.idle":"2023-05-14T00:23:02.953821Z","shell.execute_reply.started":"2023-05-14T00:23:02.833079Z","shell.execute_reply":"2023-05-14T00:23:02.953181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#make model for question 14\nX_tmp = df.loc[df['level_group'] == '13-22']\ny_tmp = targets.loc[targets['q'] == 14]\ny_tmp = y_tmp['correct']\nX_train, X_test, y_train, y_test = train_test_split(X_tmp, y_tmp, test_size=0.1, stratify=y_tmp, random_state=0)\nX_train.pop('level_group')\nX_test.pop('level_group')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:02.955034Z","iopub.execute_input":"2023-05-14T00:23:02.955485Z","iopub.status.idle":"2023-05-14T00:23:02.984850Z","shell.execute_reply.started":"2023-05-14T00:23:02.955458Z","shell.execute_reply":"2023-05-14T00:23:02.983751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pickle.dump(pipe_xgboost_14,open('model_14.sav','wb'))\nmodel_14=pickle.load(open('/kaggle/input/predict-student-performance/model_14.sav', 'rb'))\nmodels.append(model_14)\ny_predict = model_14.predict(X_test)\nprint('question 14 f1_score: ' + str(f1_score(y_test, y_predict)))\npickle.dump(model_14,open('model_14.sav','wb'))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:02.985992Z","iopub.execute_input":"2023-05-14T00:23:02.986236Z","iopub.status.idle":"2023-05-14T00:23:03.074766Z","shell.execute_reply.started":"2023-05-14T00:23:02.986213Z","shell.execute_reply":"2023-05-14T00:23:03.074103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#make model for question 15\nX_tmp = df.loc[df['level_group'] == '13-22']\ny_tmp = targets.loc[targets['q'] == 15]\ny_tmp = y_tmp['correct']\nX_train, X_test, y_train, y_test = train_test_split(X_tmp, y_tmp, test_size=0.1, stratify=y_tmp, random_state=0)\nX_train.pop('level_group')\nX_test.pop('level_group')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.075576Z","iopub.execute_input":"2023-05-14T00:23:03.075905Z","iopub.status.idle":"2023-05-14T00:23:03.106690Z","shell.execute_reply.started":"2023-05-14T00:23:03.075881Z","shell.execute_reply":"2023-05-14T00:23:03.105470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_15=pickle.load(open('/kaggle/input/predict-student-performance/model_15.sav', 'rb'))\nmodels.append(model_15)\ny_predict = model_15.predict(X_test)\nprint('question 15 f1_score: ' + str(f1_score(y_test, y_predict)))\npickle.dump(model_15,open('model_15.sav','wb'))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.107777Z","iopub.execute_input":"2023-05-14T00:23:03.108020Z","iopub.status.idle":"2023-05-14T00:23:03.201417Z","shell.execute_reply.started":"2023-05-14T00:23:03.107995Z","shell.execute_reply":"2023-05-14T00:23:03.200398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#make model for question 16\nX_tmp = df.loc[df['level_group'] == '13-22']\ny_tmp = targets.loc[targets['q'] == 16]\ny_tmp = y_tmp['correct']\nX_train, X_test, y_train, y_test = train_test_split(X_tmp, y_tmp, test_size=0.1, stratify=y_tmp, random_state=0)\nX_train.pop('level_group')\nX_test.pop('level_group')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.204850Z","iopub.execute_input":"2023-05-14T00:23:03.206480Z","iopub.status.idle":"2023-05-14T00:23:03.239103Z","shell.execute_reply.started":"2023-05-14T00:23:03.206446Z","shell.execute_reply":"2023-05-14T00:23:03.238430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_16=pickle.load(open('/kaggle/input/predict-student-performance/model_16.sav', 'rb'))\nmodels.append(model_16)\ny_predict = model_16.predict(X_test)\nprint('question 16 f1_score: ' + str(f1_score(y_test, y_predict)))\npickle.dump(model_16,open('model_16.sav','wb'))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.242109Z","iopub.execute_input":"2023-05-14T00:23:03.243667Z","iopub.status.idle":"2023-05-14T00:23:03.314569Z","shell.execute_reply.started":"2023-05-14T00:23:03.243637Z","shell.execute_reply":"2023-05-14T00:23:03.313658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#make model for question 17\nX_tmp = df.loc[df['level_group'] == '13-22']\ny_tmp = targets.loc[targets['q'] == 17]\ny_tmp = y_tmp['correct']\nX_train, X_test, y_train, y_test = train_test_split(X_tmp, y_tmp, test_size=0.1, stratify=y_tmp, random_state=0)\nX_train.pop('level_group')\nX_test.pop('level_group')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.315553Z","iopub.execute_input":"2023-05-14T00:23:03.315796Z","iopub.status.idle":"2023-05-14T00:23:03.348918Z","shell.execute_reply.started":"2023-05-14T00:23:03.315770Z","shell.execute_reply":"2023-05-14T00:23:03.347499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_17=pickle.load(open('/kaggle/input/predict-student-performance/model_17.sav', 'rb'))\nmodels.append(model_17)\ny_predict = model_17.predict(X_test)\nprint('question 17 f1_score: ' + str(f1_score(y_test, y_predict)))\npickle.dump(model_17,open('model_17.sav','wb'))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.349937Z","iopub.execute_input":"2023-05-14T00:23:03.350160Z","iopub.status.idle":"2023-05-14T00:23:03.431449Z","shell.execute_reply.started":"2023-05-14T00:23:03.350136Z","shell.execute_reply":"2023-05-14T00:23:03.430765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#make model for question 18\nX_tmp = df.loc[df['level_group'] == '13-22']\ny_tmp = targets.loc[targets['q'] == 18]\ny_tmp = y_tmp['correct']\nX_train, X_test, y_train, y_test = train_test_split(X_tmp, y_tmp, test_size=0.1, stratify=y_tmp, random_state=0)\nX_train.pop('level_group')\nX_test.pop('level_group')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.434513Z","iopub.execute_input":"2023-05-14T00:23:03.436080Z","iopub.status.idle":"2023-05-14T00:23:03.467008Z","shell.execute_reply.started":"2023-05-14T00:23:03.436049Z","shell.execute_reply":"2023-05-14T00:23:03.465931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_18=pickle.load(open('/kaggle/input/predict-student-performance/model_18.sav', 'rb'))\nmodels.append(model_18)\ny_predict = model_18.predict(X_test)\nprint('question 18 f1_score: ' + str(f1_score(y_test, y_predict)))\npickle.dump(model_18,open('model_18.sav','wb'))","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.470958Z","iopub.execute_input":"2023-05-14T00:23:03.472731Z","iopub.status.idle":"2023-05-14T00:23:03.546009Z","shell.execute_reply.started":"2023-05-14T00:23:03.472687Z","shell.execute_reply":"2023-05-14T00:23:03.544437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#build function to predict test data. Need to preprocess data and add logic to load models and such for each question to create output.","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.547028Z","iopub.execute_input":"2023-05-14T00:23:03.547287Z","iopub.status.idle":"2023-05-14T00:23:03.555047Z","shell.execute_reply.started":"2023-05-14T00:23:03.547261Z","shell.execute_reply":"2023-05-14T00:23:03.553942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#making a list of strings for categories, numbers, and events\nCATS = ['event_name', 'fqid', 'room_fqid', 'text']\n\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y',\n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\n\nEVENTS = ['navigate_click','person_click','cutscene_click','object_click',\n          'map_hover','notification_click','map_click','observation_click',\n          'checkpoint']","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.556179Z","iopub.execute_input":"2023-05-14T00:23:03.556449Z","iopub.status.idle":"2023-05-14T00:23:03.564978Z","shell.execute_reply.started":"2023-05-14T00:23:03.556424Z","shell.execute_reply":"2023-05-14T00:23:03.564189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train):\n\n    dfs = []\n    for c in CATS: #go from 0 to (len(cats))-1 on and from start and each step including len(cats)-1 let c be the string at the given position\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique') #find how many times a given level group for a certain id has event_name, fqid, room_fqid, and text occur\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp) # add the dataframe\n    for c in NUMS: #same idea as c in Cats\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean') #find the mean elapsed time, level, page, ... for a given session_ids level group\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std') #find the standard deviation for elapsed_time, level, ... for a given session_ids level group\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    for c in EVENTS: #here we are separating all the events into their separate rows\n        train[c] = (train.event_name == c).astype('int8')\n    for c in EVENTS + ['elapsed_time']: #now we are getting the sum for each event for a unique id and elapsed_time\n        tmp = train.groupby(['session_id','level_group'])[c].agg('sum')\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n    train = train.drop(EVENTS,axis=1)\n\n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    # df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.566357Z","iopub.execute_input":"2023-05-14T00:23:03.566648Z","iopub.status.idle":"2023-05-14T00:23:03.577287Z","shell.execute_reply.started":"2023-05-14T00:23:03.566623Z","shell.execute_reply":"2023-05-14T00:23:03.576512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def get_input(file):\n#     tmp = pd.read_csv(file,usecols=[0]) #first we are just getting the user id\n#     tmp = tmp.groupby('session_id').session_id.agg('count') #then we are grouping all the same ids together and showing the total amount they appear. Example: id: 20090312431273200 appears 881 times\n#     PIECES = 10\n#     CHUNK = int( np.ceil(len(tmp)/PIECES) ) #there are 23,562 ids, and we want to divide it by 10. Thus, 1/10 is 2357\n#     reads = []\n#     skips = [0]\n#     for k in range(PIECES):\n#         a = k*CHUNK #start of current fraction\n#         b = (k+1)*CHUNK #end of current fraction\n#         if b>len(tmp): b=len(tmp) #if b is greater than len(tmp) == position + 1, then use len(tmp)\n#         r = tmp.iloc[a:b].sum() #get sum from start of fraction to (end of fraction) - 1\n#         reads.append(r)\n#         skips.append(skips[-1]+r)\n    \n#     # PROCESS TRAIN DATA IN PIECES\n#     all_pieces = []\n#     for k in range(PIECES):\n#         SKIPS = 0\n#         if k>0: SKIPS = range(1,skips[k]+1)\n#         train = pd.read_csv(file,\n#                             nrows=reads[k], skiprows=SKIPS)\n#         df = feature_engineer(train)\n#         all_pieces.append(df)\n#         # CONCATENATE ALL PIECES INTO ONE DATAFRAME\n#         import gc\n#         del train; gc.collect()\n#         df = pd.concat(all_pieces, axis=0)\n#         df = df.reset_index()\n#         df.pop('index')\n#     return df\n    \n# test_df = get_input('/kaggle/input/predict-student-performance-from-game-play/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.578532Z","iopub.execute_input":"2023-05-14T00:23:03.578990Z","iopub.status.idle":"2023-05-14T00:23:03.593303Z","shell.execute_reply.started":"2023-05-14T00:23:03.578963Z","shell.execute_reply":"2023-05-14T00:23:03.592424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.594504Z","iopub.execute_input":"2023-05-14T00:23:03.595036Z","iopub.status.idle":"2023-05-14T00:23:03.607094Z","shell.execute_reply.started":"2023-05-14T00:23:03.595009Z","shell.execute_reply":"2023-05-14T00:23:03.605896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predict test","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.607871Z","iopub.execute_input":"2023-05-14T00:23:03.608100Z","iopub.status.idle":"2023-05-14T00:23:03.616997Z","shell.execute_reply.started":"2023-05-14T00:23:03.608077Z","shell.execute_reply":"2023-05-14T00:23:03.616290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#level_groups '0-4' to train model for questions 1-3\n#level groups '5-12' to train questions 4 thru 13\n#level groups '13-22' to train questions 14 thru 18","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.618294Z","iopub.execute_input":"2023-05-14T00:23:03.618510Z","iopub.status.idle":"2023-05-14T00:23:03.627107Z","shell.execute_reply.started":"2023-05-14T00:23:03.618487Z","shell.execute_reply":"2023-05-14T00:23:03.626107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def predict(df, output):\n#     k = 0\n#     ids = df['session_id'].unique()\n    \n#     tmp = df.loc[df['level_group'] == '0-4']\n#     if not tmp.empty:\n#         tmp.pop('level_group')\n#         for j in range(3):\n#             output.loc[k, 'correct'] = int(str(models[j].predict(tmp)).strip('[]'))\n#             k = k + 1\n            \n#     tmp = df.loc[df['level_group'] == '5-12']\n#     if not tmp.empty:\n#         tmp.pop('level_group')\n#         for j in range(3,13):\n#             output.loc[k, f'correct'] = int(str(models[j].predict(tmp)).strip('[]'))\n#             k = k + 1\n            \n#     tmp = df.loc[df['level_group'] == '13-22']\n#     if not tmp.empty:\n#         tmp.pop('level_group')\n#         for j in range(13,18):\n#             output.loc[k, f'correct'] = int(str(models[j].predict(tmp)).strip('[]'))\n#             k = k + 1","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.627838Z","iopub.execute_input":"2023-05-14T00:23:03.628059Z","iopub.status.idle":"2023-05-14T00:23:03.638611Z","shell.execute_reply.started":"2023-05-14T00:23:03.628036Z","shell.execute_reply":"2023-05-14T00:23:03.637156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.639815Z","iopub.execute_input":"2023-05-14T00:23:03.640037Z","iopub.status.idle":"2023-05-14T00:23:03.658733Z","shell.execute_reply.started":"2023-05-14T00:23:03.640014Z","shell.execute_reply":"2023-05-14T00:23:03.657550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # The API will deliver two dataframes in this specific order,\n# # for every session+level grouping (one group per session for each checkpoint)\n# for (test, sample_submission) in iter_test:\n#     df = feature_engineer(test)\n#     predict(df,sample_submission)\n#     print(k)\n#     env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.660503Z","iopub.execute_input":"2023-05-14T00:23:03.660801Z","iopub.status.idle":"2023-05-14T00:23:03.666708Z","shell.execute_reply.started":"2023-05-14T00:23:03.660775Z","shell.execute_reply":"2023-05-14T00:23:03.665278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for (test, sample_submission) in iter_test:\n    sample_submission['question'] = [int(label.split('_')[1][1:]) for label in sample_submission['session_id']]\n    df = sample_submission\n    test = feature_engineer(test)\n    test.pop('level_group')\n    df.loc[df.question == 1, 'correct'] = int(str(models[0].predict(test)).strip('[]'))\n    df.loc[df.question == 2, 'correct'] = int(str(models[1].predict(test)).strip('[]'))\n    df.loc[df.question == 3, 'correct'] = int(str(models[2].predict(test)).strip('[]'))\n    df.loc[df.question == 4, 'correct'] = int(str(models[3].predict(test)).strip('[]'))\n    df.loc[df.question == 5, 'correct'] = int(str(models[4].predict(test)).strip('[]')) \n    df.loc[df.question == 6, 'correct'] = int(str(models[5].predict(test)).strip('[]'))\n    df.loc[df.question == 7, 'correct'] = int(str(models[6].predict(test)).strip('[]'))\n    df.loc[df.question == 8, 'correct'] = int(str(models[7].predict(test)).strip('[]'))\n    df.loc[df.question == 9, 'correct'] = int(str(models[8].predict(test)).strip('[]'))\n    df.loc[df.question == 10, 'correct'] = int(str(models[9].predict(test)).strip('[]'))\n    df.loc[df.question == 11, 'correct'] = int(str(models[10].predict(test)).strip('[]'))\n    df.loc[df.question == 12, 'correct'] = int(str(models[11].predict(test)).strip('[]'))\n    df.loc[df.question == 13, 'correct'] = int(str(models[12].predict(test)).strip('[]'))\n    df.loc[df.question == 14, 'correct'] = int(str(models[13].predict(test)).strip('[]')) \n    df.loc[df.question == 15, 'correct'] = int(str(models[14].predict(test)).strip('[]'))\n    df.loc[df.question == 16, 'correct'] = int(str(models[15].predict(test)).strip('[]'))\n    df.loc[df.question == 17, 'correct'] = int(str(models[16].predict(test)).strip('[]'))\n    df.loc[df.question == 18, 'correct'] = int(str(models[17].predict(test)).strip('[]'))\n    sample_submission = df[['session_id', 'correct']]\n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.668316Z","iopub.execute_input":"2023-05-14T00:23:03.668991Z","iopub.status.idle":"2023-05-14T00:23:03.680147Z","shell.execute_reply.started":"2023-05-14T00:23:03.668962Z","shell.execute_reply":"2023-05-14T00:23:03.677811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for (test, sample_submission) in iter_test:\n# #     print(test.columns)\n# #     print(test.shape)\n#     sample_submission['question'] = [int(label.split('_')[1][1:]) for label in sample_submission['session_id']]\n#     df = sample_submission\n#     df.loc[df.question == 1, 'correct'] = 0\n#     df.loc[df.question == 2, 'correct'] = 0  \n#     df.loc[df.question == 3, 'correct'] = 0\n#     df.loc[df.question == 4, 'correct'] = 0\n#     df.loc[df.question == 5, 'correct'] = 0 \n#     df.loc[df.question == 6, 'correct'] = 0 \n#     df.loc[df.question == 7, 'correct'] = 0 \n#     df.loc[df.question == 8, 'correct'] = 0 \n#     df.loc[df.question == 9, 'correct'] = 0 \n#     df.loc[df.question == 10, 'correct'] = 0\n#     df.loc[df.question == 11, 'correct'] = 0\n#     df.loc[df.question == 12, 'correct'] = 0 \n#     df.loc[df.question == 13, 'correct'] = 0 \n#     df.loc[df.question == 14, 'correct'] = 0 \n#     df.loc[df.question == 15, 'correct'] = 0\n#     df.loc[df.question == 16, 'correct'] = 0\n#     df.loc[df.question == 17, 'correct'] = 0\n#     df.loc[df.question == 18, 'correct'] = 0\n#     sample_submission = df[['session_id', 'correct']]\n#     print(sample_submission.tail(10))\n#     print('-'*20)\n#     env.predict(sample_submission)\n# #     print(sample_submission.tail(10))\n# #     print('-'*20)","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.681520Z","iopub.execute_input":"2023-05-14T00:23:03.681840Z","iopub.status.idle":"2023-05-14T00:23:03.817045Z","shell.execute_reply.started":"2023-05-14T00:23:03.681812Z","shell.execute_reply":"2023-05-14T00:23:03.816069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('submission.csv')\nprint( df.shape )\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.818145Z","iopub.execute_input":"2023-05-14T00:23:03.818554Z","iopub.status.idle":"2023-05-14T00:23:03.830859Z","shell.execute_reply.started":"2023-05-14T00:23:03.818528Z","shell.execute_reply":"2023-05-14T00:23:03.829407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.832694Z","iopub.execute_input":"2023-05-14T00:23:03.832968Z","iopub.status.idle":"2023-05-14T00:23:03.843722Z","shell.execute_reply.started":"2023-05-14T00:23:03.832942Z","shell.execute_reply":"2023-05-14T00:23:03.843036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-05-14T00:23:03.844679Z","iopub.execute_input":"2023-05-14T00:23:03.845073Z","iopub.status.idle":"2023-05-14T00:23:04.137177Z","shell.execute_reply.started":"2023-05-14T00:23:03.845049Z","shell.execute_reply":"2023-05-14T00:23:04.135878Z"},"trusted":true},"execution_count":null,"outputs":[]}]}