{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # Import library NumPy (Kalkulasi dsb)\nimport pandas as pd # Import library Pandas (Baca data)\nimport polars as pl # Import library Polars (Baca data yang lebih efisien)\nfrom matplotlib import pyplot as plt # Visualisasi\nimport seaborn as sns # Visualisasi\nimport datatable as dt\nfrom warnings import filterwarnings, simplefilter\nimport joblib\nfilterwarnings('ignore')\nsimplefilter('ignore')\nplt.style.use('fivethirtyeight')\nplt.rcParams['figure.figsize'] = (30, 20)\nsns.set_style('darkgrid')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-03T02:46:35.028989Z","iopub.execute_input":"2023-05-03T02:46:35.029417Z","iopub.status.idle":"2023-05-03T02:46:36.671564Z","shell.execute_reply.started":"2023-05-03T02:46:35.029380Z","shell.execute_reply":"2023-05-03T02:46:36.670304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-03T02:46:36.677187Z","iopub.execute_input":"2023-05-03T02:46:36.677625Z","iopub.status.idle":"2023-05-03T02:46:36.698140Z","shell.execute_reply.started":"2023-05-03T02:46:36.677587Z","shell.execute_reply":"2023-05-03T02:46:36.697051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\",usecols=[0], low_memory = True)\ntmp = tmp.groupby('session_id').session_id.agg('count')\ndisplay(tmp)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T02:46:36.703562Z","iopub.execute_input":"2023-05-03T02:46:36.706429Z","iopub.status.idle":"2023-05-03T02:48:07.183513Z","shell.execute_reply.started":"2023-05-03T02:46:36.706369Z","shell.execute_reply":"2023-05-03T02:48:07.182248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\nlabels['session'] = labels['session_id'].str.split('_', expand = True)[0].astype(np.uint64)\nlabels['q'] = labels['session_id'].str.split('_q', expand = True)[1].astype(int)  \nlabels","metadata":{"execution":{"iopub.status.busy":"2023-05-03T02:48:07.185841Z","iopub.execute_input":"2023-05-03T02:48:07.186299Z","iopub.status.idle":"2023-05-03T02:48:10.316770Z","shell.execute_reply.started":"2023-05-03T02:48:07.186259Z","shell.execute_reply":"2023-05-03T02:48:10.315652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-05-03T02:48:10.318176Z","iopub.execute_input":"2023-05-03T02:48:10.318638Z","iopub.status.idle":"2023-05-03T02:48:10.326942Z","shell.execute_reply.started":"2023-05-03T02:48:10.318602Z","shell.execute_reply":"2023-05-03T02:48:10.326012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels['q'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T02:48:10.328178Z","iopub.execute_input":"2023-05-03T02:48:10.328708Z","iopub.status.idle":"2023-05-03T02:48:10.347588Z","shell.execute_reply.started":"2023-05-03T02:48:10.328673Z","shell.execute_reply":"2023-05-03T02:48:10.346597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.enable()\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T02:48:10.348921Z","iopub.execute_input":"2023-05-03T02:48:10.349485Z","iopub.status.idle":"2023-05-03T02:48:10.484950Z","shell.execute_reply.started":"2023-05-03T02:48:10.349450Z","shell.execute_reply":"2023-05-03T02:48:10.484052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ITER = 10\nPIECES = int(np.ceil(len(tmp) / ITER))\n\nreads = [] # -> Berapa banyak rows yang ingin kita baca\nskips = [0] # -> Kita Mulai dari rows berapa\n\nfor k in range(ITER) :\n    a = k*PIECES\n    b = (k+1)*PIECES\n    if b>len(tmp) : b=len(tmp)\n    r = tmp.iloc[a:b].sum()\n    reads.append(r)\n    skips.append(skips[-1] + r)\n\nprint(f'To avoid memory error, we will read train in {PIECES} pieces of sizes:')\nprint(reads)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T02:48:10.486205Z","iopub.execute_input":"2023-05-03T02:48:10.486812Z","iopub.status.idle":"2023-05-03T02:48:10.501418Z","shell.execute_reply.started":"2023-05-03T02:48:10.486776Z","shell.execute_reply":"2023-05-03T02:48:10.500421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtyping = {\n    'session_id' : np.uint64,\n    'index' : np.uint8,\n    'elapsed_time' : np.uint8,\n    'event_name' : 'category',\n    'name' : 'category',\n    'level' : np.uint8,\n    'page' : np.float32,\n    'room_coor_x' : np.float32,\n    'room_coor_y' : np.float32,\n    'screen_coor_x' : np.float32,\n    'screen_coor_y' : np.float32,\n    'hover_duration' : np.float32,\n    'text' : 'category',\n    'fqid' : 'category',\n    'room_fqid' : 'category',\n    'text_fqid' : 'category',\n    'fullscreen' : np.bool8,\n    'hq' : np.bool8,\n    'music' : np.bool8,\n    'level_group' : 'category'\n}","metadata":{"execution":{"iopub.status.busy":"2023-05-03T02:48:10.502811Z","iopub.execute_input":"2023-05-03T02:48:10.503823Z","iopub.status.idle":"2023-05-03T02:48:10.511234Z","shell.execute_reply.started":"2023-05-03T02:48:10.503782Z","shell.execute_reply":"2023-05-03T02:48:10.510058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf0 = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', nrows = reads[0], dtype = dtyping, low_memory = True)\nmem_usage = df0.memory_usage().sum() / 1024 ** 2\nprint(f'Memory Usage : {mem_usage} MB')","metadata":{"execution":{"iopub.status.busy":"2023-05-03T02:48:10.515872Z","iopub.execute_input":"2023-05-03T02:48:10.516548Z","iopub.status.idle":"2023-05-03T02:48:18.040504Z","shell.execute_reply.started":"2023-05-03T02:48:10.516505Z","shell.execute_reply":"2023-05-03T02:48:18.039446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATS = ['event_name', 'fqid', 'room_fqid', 'text', 'name']\nNUMS = ['elapsed_time', 'level', 'page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration', 'fullscreen', 'hq', 'music']\n\n# https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data\nEVENTS = ['navigate_click','person_click','cutscene_click','object_click',\n          'map_hover','notification_click','map_click','observation_click',\n          'checkpoint']","metadata":{"execution":{"iopub.status.busy":"2023-05-03T02:48:18.044823Z","iopub.execute_input":"2023-05-03T02:48:18.047135Z","iopub.status.idle":"2023-05-03T02:48:18.054645Z","shell.execute_reply.started":"2023-05-03T02:48:18.047080Z","shell.execute_reply":"2023-05-03T02:48:18.053467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df0","metadata":{"execution":{"iopub.status.busy":"2023-05-03T02:48:18.062612Z","iopub.execute_input":"2023-05-03T02:48:18.065360Z","iopub.status.idle":"2023-05-03T02:48:18.108667Z","shell.execute_reply.started":"2023-05-03T02:48:18.065286Z","shell.execute_reply":"2023-05-03T02:48:18.107638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df0.groupby(['session_id', 'level'])['elapsed_time'].sum()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T02:48:18.113077Z","iopub.execute_input":"2023-05-03T02:48:18.115381Z","iopub.status.idle":"2023-05-03T02:48:18.289508Z","shell.execute_reply.started":"2023-05-03T02:48:18.115333Z","shell.execute_reply":"2023-05-03T02:48:18.288583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_level(train) :\n    dfs = []\n    for c in CATS :\n        tmp = train.groupby(['session_id', 'level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS :\n        tmp = train.groupby(['session_id', 'level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in NUMS :\n        tmp = train.groupby(['session_id', 'level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    for c in EVENTS :\n        train[c] = (train.event_name == c).astype('int8')\n    for c in EVENTS + ['elapsed_time'] :\n        tmp = train.groupby(['session_id', 'level_group'])[c].agg('sum')\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n    train = train.drop(EVENTS, axis = 1)\n    \n    df = pd.concat(dfs, axis = 1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-03T02:48:18.293839Z","iopub.execute_input":"2023-05-03T02:48:18.296091Z","iopub.status.idle":"2023-05-03T02:48:18.306877Z","shell.execute_reply.started":"2023-05-03T02:48:18.296045Z","shell.execute_reply":"2023-05-03T02:48:18.305413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_pieces = []\nfor k in range(ITER) :\n    print(k, ',', end = ' ')\n    SKIPS = 0\n    if k>0 : SKIPS = range(1, skips[k] + 1)\n    train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', nrows = reads[k], skiprows = SKIPS, dtype = dtyping, low_memory = True)\n    df = process_level(train)\n    all_pieces.append(df)\n    del train; del df; gc.collect()\nprint('\\n')\ntrain_df = pd.concat(all_pieces, axis = 0).pipe(reduce_memory_usage)\nprint(f'Shape of Train DF : {train_df.shape}')\ndisplay(train_df.head())","metadata":{"execution":{"iopub.status.busy":"2023-05-03T02:48:18.312527Z","iopub.execute_input":"2023-05-03T02:48:18.315190Z","iopub.status.idle":"2023-05-03T02:52:38.284852Z","shell.execute_reply.started":"2023-05-03T02:48:18.315139Z","shell.execute_reply":"2023-05-03T02:52:38.283643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES = [c for c in train_df.columns if not c in ['level_group', 'level']]\nprint('We will train with', len(FEATURES) ,'features')\nALL_USERS = train_df.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"execution":{"iopub.status.busy":"2023-05-03T02:52:38.286588Z","iopub.execute_input":"2023-05-03T02:52:38.286950Z","iopub.status.idle":"2023-05-03T02:52:38.297384Z","shell.execute_reply.started":"2023-05-03T02:52:38.286909Z","shell.execute_reply":"2023-05-03T02:52:38.296039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import GroupKFold\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import f1_score\ngkf = GroupKFold(n_splits = 5)\noof = pd.DataFrame(data = np.zeros((len(ALL_USERS), 18)), index = ALL_USERS)\nscores = pd.DataFrame(\n    index = [f'FOLD_{i}' for i in range(5)],\n    columns = list(range(1, 19))\n)\nmodels = {}\n\nfor i, (t, v) in enumerate(gkf.split(X = train_df, groups = train_df.index)) :\n    print(f\"FOLD {i}\")\n    print('')\n    \n    for l in range(1, 19) :\n        if l <= 3 : grp = '0-4'\n        if l <= 13 : grp = '5-12'\n        if l <= 22 : grp = '13-22'\n        xtrain = train_df.iloc[t]\n        xtrain = xtrain.loc[xtrain.level_group == grp]\n        train_users = xtrain.index.values\n        ytrain = labels.loc[labels.q==l].set_index('session').loc[train_users]\n        \n        xval = train_df.iloc[v]\n        xval = xval.loc[xval.level_group == grp]\n        val_users = xval.index.values\n        yval = labels.loc[labels.q==l].set_index('session').loc[val_users]\n        \n        model = LogisticRegression(random_state = 0, solver = 'saga', max_iter = 1500, n_jobs = -1, C = 0.1)\n        model.fit(\n            xtrain[FEATURES].astype(np.float32), ytrain['correct'],\n        )\n        yhat = model.predict_proba(xval[FEATURES].astype(np.float32))[:, 1]\n        score = f1_score(yval['correct'], np.round(yhat).astype(int))\n        print(f'FOLD {i} LEVEL {l} : {score}')\n        models[f'fold{i}-level{l}'] = model\n        joblib.dump(model, f'model-fold{i}-level{l}.pkl')\n        oof.loc[val_users, l-1] = yhat\n        scores.loc[f'FOLD_{i}', l] = score\n        del xtrain, train_users, ytrain, xval, val_users, yval, model, yhat, score\n        gc.collect()\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T02:52:38.299204Z","iopub.execute_input":"2023-05-03T02:52:38.299624Z","iopub.status.idle":"2023-05-03T03:13:00.793584Z","shell.execute_reply.started":"2023-05-03T02:52:38.299585Z","shell.execute_reply":"2023-05-03T03:13:00.792403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores","metadata":{"execution":{"iopub.status.busy":"2023-05-03T03:13:00.798737Z","iopub.execute_input":"2023-05-03T03:13:00.799129Z","iopub.status.idle":"2023-05-03T03:13:00.825536Z","shell.execute_reply.started":"2023-05-03T03:13:00.799088Z","shell.execute_reply":"2023-05-03T03:13:00.824553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.mean(scores)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T03:13:00.827056Z","iopub.execute_input":"2023-05-03T03:13:00.828179Z","iopub.status.idle":"2023-05-03T03:13:00.839580Z","shell.execute_reply.started":"2023-05-03T03:13:00.828140Z","shell.execute_reply":"2023-05-03T03:13:00.838652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = labels.loc[labels.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-05-03T03:13:00.841121Z","iopub.execute_input":"2023-05-03T03:13:00.841968Z","iopub.status.idle":"2023-05-03T03:13:00.970651Z","shell.execute_reply.started":"2023-05-03T03:13:00.841924Z","shell.execute_reply":"2023-05-03T03:13:00.969168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.3, 0.95, 0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')\n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-05-03T03:13:00.972080Z","iopub.execute_input":"2023-05-03T03:13:00.972962Z","iopub.status.idle":"2023-05-03T03:13:10.918444Z","shell.execute_reply.started":"2023-05-03T03:13:00.972916Z","shell.execute_reply":"2023-05-03T03:13:10.917119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T03:13:10.919995Z","iopub.execute_input":"2023-05-03T03:13:10.920468Z","iopub.status.idle":"2023-05-03T03:13:11.279892Z","shell.execute_reply.started":"2023-05-03T03:13:10.920429Z","shell.execute_reply":"2023-05-03T03:13:11.278963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T03:13:11.280849Z","iopub.execute_input":"2023-05-03T03:13:11.281147Z","iopub.status.idle":"2023-05-03T03:13:11.579453Z","shell.execute_reply.started":"2023-05-03T03:13:11.281118Z","shell.execute_reply":"2023-05-03T03:13:11.578568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# IMPORT KAGGLE API\nimport jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n\n# CLEAR MEMORY\nimport gc\ndel labels, oof, true\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T03:13:11.581045Z","iopub.execute_input":"2023-05-03T03:13:11.581703Z","iopub.status.idle":"2023-05-03T03:13:11.746349Z","shell.execute_reply.started":"2023-05-03T03:13:11.581667Z","shell.execute_reply":"2023-05-03T03:13:11.745397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\nFOLD_TO_USE = 2\n\nfor (test, sample_submission) in iter_test:\n    \n    # FEATURE ENGINEER TEST DATA\n    df = process_level(test)\n    df = df.pivot_table(columns = 'level_group', values = [x for x in df.columns if not x == 'level_group'], index = 'session_id', aggfunc = 'sum', fill_value = 0)\n    \n    # INFER TEST DATA\n    grp = '0-4'\n    a,b = limits[grp]\n    for t in range(a,b):\n        clf = models[f'fold{FOLD_TO_USE}-level{t}']\n        p = clf.predict_proba(df[FEATURES].astype('float32'))[0,1]\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        sample_submission.loc[mask,'correct'] = int( p > best_threshold )\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T03:13:11.748118Z","iopub.execute_input":"2023-05-03T03:13:11.748822Z","iopub.status.idle":"2023-05-03T03:13:12.653679Z","shell.execute_reply.started":"2023-05-03T03:13:11.748783Z","shell.execute_reply":"2023-05-03T03:13:12.652717Z"},"trusted":true},"execution_count":null,"outputs":[]}]}