{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np, gc\nfrom sklearn.model_selection import KFold, GroupKFold\nimport xgboost\nfrom sklearn.metrics import f1_score, make_scorer\nfrom sklearn.model_selection import RandomizedSearchCV\nimport joblib","metadata":{"execution":{"iopub.status.busy":"2023-06-19T18:56:37.487595Z","iopub.execute_input":"2023-06-19T18:56:37.488277Z","iopub.status.idle":"2023-06-19T18:56:38.214305Z","shell.execute_reply.started":"2023-06-19T18:56:37.488244Z","shell.execute_reply":"2023-06-19T18:56:38.213382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv',usecols=[0])\ntmp = tmp.groupby('session_id').session_id.agg('count')\nPIECES=10\nCHUNK = int(np.ceil(len(tmp)/PIECES))\nreads = []\nskips = [0]\nfor k in range (PIECES):\n    a = k*CHUNK \n    b = (k+1)*CHUNK\n    if b>len(tmp) : b=len(tmp)\n    r = tmp.iloc[a:b].sum()\n    reads.append(r)\n    skips.append(skips[-1]+r)","metadata":{"execution":{"iopub.status.busy":"2023-06-19T18:56:38.216502Z","iopub.execute_input":"2023-06-19T18:56:38.216874Z","iopub.status.idle":"2023-06-19T18:57:48.873755Z","shell.execute_reply.started":"2023-06-19T18:56:38.216841Z","shell.execute_reply":"2023-06-19T18:57:48.872762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train =pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', nrows=reads[0])","metadata":{"execution":{"iopub.status.busy":"2023-06-19T18:57:48.875210Z","iopub.execute_input":"2023-06-19T18:57:48.875545Z","iopub.status.idle":"2023-06-19T18:57:56.185857Z","shell.execute_reply.started":"2023-06-19T18:57:48.875515Z","shell.execute_reply":"2023-06-19T18:57:56.184718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\nlabels['session'] = labels.session_id.apply(lambda x: int(x.split('_')[0]))\nlabels['q'] =labels.session_id.apply(lambda x :int(x.split('_')[-1][1:]))","metadata":{"execution":{"iopub.status.busy":"2023-06-19T18:57:56.188514Z","iopub.execute_input":"2023-06-19T18:57:56.189378Z","iopub.status.idle":"2023-06-19T18:57:57.442484Z","shell.execute_reply.started":"2023-06-19T18:57:56.189351Z","shell.execute_reply":"2023-06-19T18:57:57.441547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATS = ['event_name', 'fqid', 'room_fqid', 'text']\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\n\n# https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data\nEVENTS = ['navigate_click','person_click','cutscene_click','object_click',\n          'map_hover','notification_click','map_click','observation_click',\n          'checkpoint']","metadata":{"execution":{"iopub.status.busy":"2023-06-19T18:57:57.443901Z","iopub.execute_input":"2023-06-19T18:57:57.444381Z","iopub.status.idle":"2023-06-19T18:57:57.451211Z","shell.execute_reply.started":"2023-06-19T18:57:57.444347Z","shell.execute_reply":"2023-06-19T18:57:57.449226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train):\n    \n    dfs = []\n########categorical###############\n    for c in CATS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n###############nums###############\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id', 'level_group'])[c].agg(lambda x: x.max() - x.min())\n        tmp.name = tmp.name + '_range'\n        dfs.append(tmp)\n    for c in NUMS: \n        tmp = train.groupby(['session_id', 'level_group'])[c].agg('median')\n        tmp.name = tmp.name + '_median'\n        dfs.append(tmp)\n\n##############EVENTS list###################\"\"    \n    for c in EVENTS: \n        train[c] = (train.event_name == c).astype('int8') \n    for c in EVENTS + ['elapsed_time']: \n        tmp = train.groupby(['session_id','level_group'])[c].agg('sum') \n        tmp.name = tmp.name + '_sum' \n        dfs.append(tmp) \n    for c in EVENTS :\n        tmp = train.groupby(['session_id', 'level_group'])[c].agg(lambda x: x.mode().values[0])\n        tmp.name = tmp.name + '_mode'\n        dfs.append(tmp)\n   \n\n    \n    train = train.drop(EVENTS,axis=1)\n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-06-19T18:57:57.452602Z","iopub.execute_input":"2023-06-19T18:57:57.453495Z","iopub.status.idle":"2023-06-19T18:57:57.467188Z","shell.execute_reply.started":"2023-06-19T18:57:57.453416Z","shell.execute_reply":"2023-06-19T18:57:57.466118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_pieces = []\nprint(f'Processing train as {PIECES} pieces to avoid memory error... ')\nfor k in range(PIECES):\n    print(k,', ',end='')\n    SKIPS = 0\n    if k>0: SKIPS = range(1,skips[k]+1)\n    train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv',\n                        nrows=reads[k], skiprows=SKIPS)\n    df = feature_engineer(train)\n    all_pieces.append(df)\nprint('\\n')\ndel train; gc.collect()\ndf = pd.concat(all_pieces, axis=0)\nprint('Shape of all train data after feature engineering:', df.shape )\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-19T18:57:57.468551Z","iopub.execute_input":"2023-06-19T18:57:57.469055Z","iopub.status.idle":"2023-06-19T19:07:46.433639Z","shell.execute_reply.started":"2023-06-19T18:57:57.469024Z","shell.execute_reply":"2023-06-19T19:07:46.432725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\nALL_USERS = df.index.unique()","metadata":{"execution":{"iopub.status.busy":"2023-06-19T19:07:46.434903Z","iopub.execute_input":"2023-06-19T19:07:46.435990Z","iopub.status.idle":"2023-06-19T19:07:46.445798Z","shell.execute_reply.started":"2023-06-19T19:07:46.435953Z","shell.execute_reply":"2023-06-19T19:07:46.444885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_scorer = make_scorer(f1_score)\nclassifier = xgboost.XGBClassifier()\ngkf = GroupKFold(n_splits=5)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index = ALL_USERS)\nparams = {\n 'learning_rate' : [0.1,0.01,0.001],\n 'max_depth' : [3, 5, 7],\n 'gamma' : [0, 0.1,0.2],\n 'n_estimators': [100],\n 'colsample_bytree' : [0.8, 0.9 , 1.0 ],\n 'subsample' : [0.8, 0.9 , 1.0 ]\n}\n\nrs_model=RandomizedSearchCV(classifier,param_distributions=params,n_iter=5,scoring=f1_scorer,cv=5,verbose=1)\nmodels = {}\n","metadata":{"execution":{"iopub.status.busy":"2023-06-19T19:07:46.447262Z","iopub.execute_input":"2023-06-19T19:07:46.447611Z","iopub.status.idle":"2023-06-19T19:07:46.456475Z","shell.execute_reply.started":"2023-06-19T19:07:46.447581Z","shell.execute_reply":"2023-06-19T19:07:46.455396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25) \n    \n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index] \n        #rows = tmp.iloc[2:5, :]  # Retrieve rows 2 to 4, all columns\n        train_x = train_x.loc[train_x.level_group == grp]\n        #The .loc indexer in pandas is used to access a group of rows and columns by label or a boolean array. In this case, it is used to select rows where the value of the 'level_group' column is equal to grp.\n        train_users = train_x.index.values\n        train_y = labels.loc[labels.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        #valid_x is a DataFrame//valid_x.index returns the index labels of the DataFrame valid_x//valid_x.index.values retrieves the index values as a NumPy array.\n        valid_y = labels.loc[labels.q==t].set_index('session').loc[valid_users]\n        #labels.loc[labels.q==t] retrieves the rows from the DataFrame labels where the value in the column 'q' is equal to t.\\\\.set_index('session') sets the column named 'session' as the new index of the resulting DataFrame.//.loc[valid_users] then retrieves the rows from the DataFrame with the index values specified in the valid_users array.\n        \n        # TRAIN MODEL        \n        clf =  rs_model\n        clf.fit(train_x[FEATURES].astype('float32'), train_y['correct'],\n                eval_set=[ (valid_x[FEATURES].astype('float32'), valid_y['correct']) ],\n                verbose=0)\n        print(\"your actual params are : \\n\")  \n        print(f'{t}({clf.best_params_}), ',end='\\n')\n        # SAVE MODEL, PREDICT VALID OOF\n        models[f'{grp}_{t}'] = clf\n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n    print()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-19T19:07:46.460228Z","iopub.execute_input":"2023-06-19T19:07:46.460649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_model = clf.best_estimator_\nprint(best_model)\njoblib.dump(best_model, 'best_model.pkl')\n        ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predictions = loaded_model.predict(test_data)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}