{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1>Predict Student Performance from Game Play</h1>","metadata":{}},{"cell_type":"markdown","source":"# Sommaire :\n\n**Part 1: Notebook configuration**\n\n - <a href=\"#C11\">P1.1 : Librairies loading </a>\n - <a href=\"#C12\">P1.2 : Fonctions </a>\n - <a href=\"#C13\">P1.3 : Data loading </a>\n \n**Part 2: Data Analysis**\n \n - <a href=\"#C21\">P2.1 : Data reducing </a>\n - <a href=\"#C22\">P2.2 : Data preprocessing </a>\n - <a href=\"#C23\">P2.3 : Target description </a>\n \n**Part 3: Data modelling**\n\n - <a href=\"#C31\">P3.1 : Logistic Regression </a>\n - <a href=\"#C32\">P3.2 : Random Forest </a>\n - <a href=\"#C33\">P3.3 : XGBoost </a>\n - <a href=\"#C34\">P3.4 : Keras </a>\n - <a href=\"#C35\">P3.5 :  Final model (optimization) </a>\n \n**Part 4: Submission**\n \n - <a href=\"#C41\">P4.1 : Submission </a>\n","metadata":{}},{"cell_type":"markdown","source":"# Part1: Notebook configuration","metadata":{}},{"cell_type":"markdown","source":"# <a name=\"C11\"> P1.1 : Librairies loading </a>","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\n\nimport jo_wilder\n\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import KFold, GroupKFold, GridSearchCV\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import f1_score\n\nfrom xgboost import XGBClassifier\n\nfrom tensorflow import keras\nimport tensorflow as tf\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom colorama import init as colorama_init\nfrom colorama import Fore\nfrom colorama import Style\npd.set_option(\"display.max_columns\", None)\npd.set_option(\"display.max_rows\", 400)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-06T14:43:36.494094Z","iopub.execute_input":"2023-03-06T14:43:36.494688Z","iopub.status.idle":"2023-03-06T14:43:46.560253Z","shell.execute_reply.started":"2023-03-06T14:43:36.494652Z","shell.execute_reply":"2023-03-06T14:43:46.559217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Libraries are loaded above","metadata":{}},{"cell_type":"markdown","source":"# <a name=\"C12\"> P1.2 : Functions </a>","metadata":{}},{"cell_type":"code","source":"#Reduce Memory Usage\ndef reduce_memory_usage(df):\n    '''\n    This function is used to reduce the memory size by adapting the used data types\n    param :\n    df: Pandas DataFrame \n    '''\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-03-06T14:43:46.561582Z","iopub.execute_input":"2023-03-06T14:43:46.562363Z","iopub.status.idle":"2023-03-06T14:43:46.575756Z","shell.execute_reply.started":"2023-03-06T14:43:46.562329Z","shell.execute_reply":"2023-03-06T14:43:46.574564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"the function \"reduce_memory_usage\" reduces the RAM memory size : 2Gb ==> 780 Mb. Here the original notebook : https://www.kaggle.com/code/mohammad2012191/reduce-memory-usage-2gb-780mb","metadata":{}},{"cell_type":"code","source":"def feature_engineer(train):\n    '''\n    This function preprocess the data \"train\" : nunique for categorical variables,\n    and mean\\std for numerical variables\n    param:\n    tarin : Pandas DataFrame\n    '''\n    dfs = []\n    for c in CATS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-03-06T14:43:46.577442Z","iopub.execute_input":"2023-03-06T14:43:46.577816Z","iopub.status.idle":"2023-03-06T14:43:46.593126Z","shell.execute_reply.started":"2023-03-06T14:43:46.577782Z","shell.execute_reply":"2023-03-06T14:43:46.592083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The function \"feature_engineer\" updates and preprocesses the initial data for medelling. Original notebook : https://www.kaggle.com/code/cdeotte/random-forest-baseline-0-664/notebook","metadata":{}},{"cell_type":"markdown","source":"# <a name=\"C13\"> P1.3 : Data loading </a>","metadata":{}},{"cell_type":"code","source":"%%time\ntrain_path = '/kaggle/input/predict-student-performance-from-game-play/train.csv'\ntrain = pd.read_csv(train_path)\nprint( train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T14:43:46.594440Z","iopub.execute_input":"2023-03-06T14:43:46.595279Z","iopub.status.idle":"2023-03-06T14:44:50.294201Z","shell.execute_reply.started":"2023-03-06T14:43:46.595239Z","shell.execute_reply":"2023-03-06T14:44:50.293233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data is loaded","metadata":{}},{"cell_type":"code","source":"%%time\ntrain_labels_path = '/kaggle/input/predict-student-performance-from-game-play/train_labels.csv'\ntargets = pd.read_csv(train_labels_path)\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]) )\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nprint( targets.shape )\ntargets.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T14:44:50.295713Z","iopub.execute_input":"2023-03-06T14:44:50.296330Z","iopub.status.idle":"2023-03-06T14:44:50.904225Z","shell.execute_reply.started":"2023-03-06T14:44:50.296297Z","shell.execute_reply":"2023-03-06T14:44:50.903281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Targets data is loaded","metadata":{}},{"cell_type":"markdown","source":"# Part 2: Data analysis","metadata":{}},{"cell_type":"markdown","source":"# <a name=\"C21\"> P2.1 : Data reducing </a>","metadata":{}},{"cell_type":"code","source":"train = reduce_memory_usage(train)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T14:45:23.984649Z","iopub.execute_input":"2023-03-06T14:45:23.985050Z","iopub.status.idle":"2023-03-06T14:45:38.492066Z","shell.execute_reply.started":"2023-03-06T14:45:23.985020Z","shell.execute_reply":"2023-03-06T14:45:38.490626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = reduce_memory_usage(targets)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T14:45:38.494233Z","iopub.execute_input":"2023-03-06T14:45:38.495070Z","iopub.status.idle":"2023-03-06T14:45:38.841150Z","shell.execute_reply.started":"2023-03-06T14:45:38.495018Z","shell.execute_reply":"2023-03-06T14:45:38.840131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data and targets RAM sizes are reduced","metadata":{}},{"cell_type":"markdown","source":"# <a name=\"C22\"> P2.2 : Data preprocessing </a>","metadata":{}},{"cell_type":"code","source":"CATS = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']","metadata":{"execution":{"iopub.status.busy":"2023-03-06T14:46:14.712476Z","iopub.execute_input":"2023-03-06T14:46:14.712896Z","iopub.status.idle":"2023-03-06T14:46:14.720044Z","shell.execute_reply.started":"2023-03-06T14:46:14.712862Z","shell.execute_reply":"2023-03-06T14:46:14.718460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf = feature_engineer(train)\nprint( df.shape )\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T14:46:15.581492Z","iopub.execute_input":"2023-03-06T14:46:15.581913Z","iopub.status.idle":"2023-03-06T14:46:32.896684Z","shell.execute_reply.started":"2023-03-06T14:46:15.581880Z","shell.execute_reply":"2023-03-06T14:46:32.895773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Preprocessed Data is generated","metadata":{}},{"cell_type":"markdown","source":"# <a name=\"C23\"> P2.3 : Target description </a>","metadata":{}},{"cell_type":"code","source":"#Groupby questions\nstudent_nb = targets['session'].nunique()\npass_rate = pd.DataFrame(targets.groupby('q').agg('correct').sum()/student_nb).reset_index()\npass_rate.columns = ['Question','Pass_Rate']\n\n#Seaborn barplot\nplt.rcParams['figure.figsize'] = (15,5)\nsns.barplot(x=\"Question\", y=\"Pass_Rate\", data=pass_rate)\nplt.title('Question difficulty', fontsize=16)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T09:44:22.421361Z","iopub.execute_input":"2023-03-06T09:44:22.421685Z","iopub.status.idle":"2023-03-06T09:44:22.740191Z","shell.execute_reply.started":"2023-03-06T09:44:22.421649Z","shell.execute_reply":"2023-03-06T09:44:22.739383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Questions pass_rate is given above","metadata":{}},{"cell_type":"code","source":"pass_rate = pass_rate.sort_values('Pass_Rate').reset_index(drop=True)\neasiest_q = pass_rate.loc[17,'Question']\nbest_pass_rate = pass_rate.loc[17,'Pass_Rate']\nhardest_q = pass_rate.loc[0,'Question']\nworst_pass_rate = pass_rate.loc[0,'Pass_Rate']\nmean_pass_rate = pass_rate['Pass_Rate'].mean()\nprint (f'Question {Fore.GREEN}{Style.BRIGHT}{easiest_q}{Style.RESET_ALL} is the easiest. {Fore.GREEN}{Style.BRIGHT}{best_pass_rate*100:.1f}%{Style.RESET_ALL} of the students got it right!')\nprint (f'Question {Fore.GREEN}{Style.BRIGHT}{hardest_q}{Style.RESET_ALL} is the hardest. Only {Fore.GREEN}{Style.BRIGHT}{worst_pass_rate*100:.1f}%{Style.RESET_ALL} of the students got it right!')\nprint (f'In average,{Fore.GREEN}{Style.BRIGHT}{mean_pass_rate*100:.1f}%{Style.RESET_ALL} of the answers are correct!')","metadata":{"execution":{"iopub.status.busy":"2023-03-06T09:44:23.850096Z","iopub.execute_input":"2023-03-06T09:44:23.850453Z","iopub.status.idle":"2023-03-06T09:44:23.860451Z","shell.execute_reply.started":"2023-03-06T09:44:23.850424Z","shell.execute_reply":"2023-03-06T09:44:23.859420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Groupby Student (session)\nstudent_performance = pd.DataFrame(targets.groupby('session').agg({'correct': 'sum'})).reset_index()\nstud_perf_distrib = pd.DataFrame(student_performance.groupby('correct').agg({'session': 'count'})).reset_index()\nstud_perf_distrib['session'] = stud_perf_distrib['session']/student_nb\nstud_perf_distrib.columns = ['Score','Density']\n\n#Seaborn barplot\nsns.barplot(x=\"Score\", y=\"Density\", data=stud_perf_distrib)\nplt.title('Student Performance Distribution', fontsize=16)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T09:44:24.701721Z","iopub.execute_input":"2023-03-06T09:44:24.702054Z","iopub.status.idle":"2023-03-06T09:44:24.936337Z","shell.execute_reply.started":"2023-03-06T09:44:24.702028Z","shell.execute_reply":"2023-03-06T09:44:24.935427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Score-Density distribution is given above","metadata":{}},{"cell_type":"markdown","source":"# Part 3: Data modelling","metadata":{}},{"cell_type":"markdown","source":"# <a name=\"C31\"> P3.1 : Logistic Regression </a>","metadata":{}},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\nALL_USERS = df.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"execution":{"iopub.status.busy":"2023-03-06T14:46:36.475384Z","iopub.execute_input":"2023-03-06T14:46:36.475964Z","iopub.status.idle":"2023-03-06T14:46:36.486177Z","shell.execute_reply.started":"2023-03-06T14:46:36.475918Z","shell.execute_reply":"2023-03-06T14:46:36.485110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gkf = GroupKFold(n_splits=5)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nLR_models = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        print(t,', ',end='')\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL\n        clf = LogisticRegression(max_iter=500)\n        clf.fit(train_x[FEATURES].astype('float32'), train_y['correct'])\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        LR_models[f'{grp}_{t}'] = clf\n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T14:46:51.976143Z","iopub.execute_input":"2023-03-06T14:46:51.976614Z","iopub.status.idle":"2023-03-06T14:47:01.280119Z","shell.execute_reply.started":"2023-03-06T14:46:51.976554Z","shell.execute_reply":"2023-03-06T14:47:01.269548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = targets.loc[targets.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-03-06T15:05:55.578153Z","iopub.execute_input":"2023-03-06T15:05:55.578609Z","iopub.status.idle":"2023-03-06T15:05:55.647169Z","shell.execute_reply.started":"2023-03-06T15:05:55.578570Z","shell.execute_reply":"2023-03-06T15:05:55.646205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nLR_best_score = 0; LR_best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>LR_best_score:\n        LR_best_score = m\n        LR_best_threshold = threshold\n        \nprint('\\n' ,'Best_score:', LR_best_score, '\\n','Best_threshold:', LR_best_threshold)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T15:05:56.820532Z","iopub.execute_input":"2023-03-06T15:05:56.821138Z","iopub.status.idle":"2023-03-06T15:06:00.101262Z","shell.execute_reply.started":"2023-03-06T15:05:56.821101Z","shell.execute_reply":"2023-03-06T15:06:00.100412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([LR_best_threshold], [LR_best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {LR_best_score:.3f} at Best Threshold = {LR_best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T15:06:00.861530Z","iopub.execute_input":"2023-03-06T15:06:00.862229Z","iopub.status.idle":"2023-03-06T15:06:01.172603Z","shell.execute_reply.started":"2023-03-06T15:06:00.862190Z","shell.execute_reply":"2023-03-06T15:06:01.171612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>LR_best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>LR_best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T09:44:59.106866Z","iopub.execute_input":"2023-03-06T09:44:59.107209Z","iopub.status.idle":"2023-03-06T09:44:59.229281Z","shell.execute_reply.started":"2023-03-06T09:44:59.107182Z","shell.execute_reply":"2023-03-06T09:44:59.228656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Logistic Regression model gets a F1_score = 0.645","metadata":{}},{"cell_type":"markdown","source":"# <a name=\"C32\"> P3.2 : Random Forest </a>","metadata":{}},{"cell_type":"markdown","source":"This model is created by : https://www.kaggle.com/code/cdeotte/random-forest-baseline-0-664/notebook","metadata":{}},{"cell_type":"code","source":"oof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nRFC_models = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        print(t,', ',end='')\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL\n        clf = RandomForestClassifier() \n        clf.fit(train_x[FEATURES].astype('float32'), train_y['correct'])\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        RFC_models[f'{grp}_{t}'] = clf\n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T09:45:02.059691Z","iopub.execute_input":"2023-03-06T09:45:02.060499Z","iopub.status.idle":"2023-03-06T09:49:56.612090Z","shell.execute_reply.started":"2023-03-06T09:45:02.060467Z","shell.execute_reply":"2023-03-06T09:49:56.611443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nRFC_best_score = 0; RFC_best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>RFC_best_score:\n        RFC_best_score = m\n        RFC_best_threshold = threshold\n        \nprint('\\n' ,'Best_score:', RFC_best_score, '\\n','Best_threshold:', RFC_best_threshold)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T09:49:56.613496Z","iopub.execute_input":"2023-03-06T09:49:56.613945Z","iopub.status.idle":"2023-03-06T09:49:59.182197Z","shell.execute_reply.started":"2023-03-06T09:49:56.613919Z","shell.execute_reply":"2023-03-06T09:49:59.180920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([RFC_best_threshold], [RFC_best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {RFC_best_score:.3f} at Best Threshold = {RFC_best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T09:50:15.834442Z","iopub.execute_input":"2023-03-06T09:50:15.834776Z","iopub.status.idle":"2023-03-06T09:50:16.030392Z","shell.execute_reply.started":"2023-03-06T09:50:15.834749Z","shell.execute_reply":"2023-03-06T09:50:16.029699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>RFC_best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>RFC_best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T09:50:20.566157Z","iopub.execute_input":"2023-03-06T09:50:20.566468Z","iopub.status.idle":"2023-03-06T09:50:20.698665Z","shell.execute_reply.started":"2023-03-06T09:50:20.566444Z","shell.execute_reply":"2023-03-06T09:50:20.698059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Random forest model gets a F1_score = 0.663","metadata":{}},{"cell_type":"markdown","source":"# <a name=\"C33\"> P3.3 : XGBoost </a>","metadata":{}},{"cell_type":"markdown","source":"Better model is given here : https://www.kaggle.com/code/cdeotte/xgboost-baseline-0-676/notebook?scriptVersionId=118599734","metadata":{}},{"cell_type":"code","source":"oof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nXGB_models = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        print(t,', ',end='')\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL\n        clf = XGBClassifier()\n        clf.fit(train_x[FEATURES].astype('float32'), train_y['correct'])\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        XGB_models[f'{grp}_{t}'] = clf\n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T09:50:28.565428Z","iopub.execute_input":"2023-03-06T09:50:28.565752Z","iopub.status.idle":"2023-03-06T09:52:52.162276Z","shell.execute_reply.started":"2023-03-06T09:50:28.565726Z","shell.execute_reply":"2023-03-06T09:52:52.161528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nXGB_best_score = 0; XGB_best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>XGB_best_score:\n        XGB_best_score = m\n        XGB_best_threshold = threshold\n        \nprint('\\n' ,'Best_score:', XGB_best_score, '\\n','Best_threshold:', XGB_best_threshold)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T09:52:54.977727Z","iopub.execute_input":"2023-03-06T09:52:54.978934Z","iopub.status.idle":"2023-03-06T09:52:57.429723Z","shell.execute_reply.started":"2023-03-06T09:52:54.978838Z","shell.execute_reply":"2023-03-06T09:52:57.428669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([XGB_best_threshold], [XGB_best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {XGB_best_score:.3f} at Best Threshold = {XGB_best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T09:52:58.470036Z","iopub.execute_input":"2023-03-06T09:52:58.470390Z","iopub.status.idle":"2023-03-06T09:52:58.657811Z","shell.execute_reply.started":"2023-03-06T09:52:58.470361Z","shell.execute_reply":"2023-03-06T09:52:58.656873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>XGB_best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>XGB_best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T09:53:05.251545Z","iopub.execute_input":"2023-03-06T09:53:05.251906Z","iopub.status.idle":"2023-03-06T09:53:05.385860Z","shell.execute_reply.started":"2023-03-06T09:53:05.251861Z","shell.execute_reply":"2023-03-06T09:53:05.384481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"XGBoost model gets a F1_score = 0.647","metadata":{}},{"cell_type":"markdown","source":"# <a name=\"C34\"> P3.4 : Keras </a>","metadata":{}},{"cell_type":"markdown","source":"This model is created by : https://www.kaggle.com/code/amaanansari09/neural-network-keras-for-jo-wilder-competition","metadata":{}},{"cell_type":"code","source":"oof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nkeras_models = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        print(t,', ',end='')\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q==t].set_index('session').loc[train_users]\n        train_y['uncorrect'] = (1-train_y['correct']).abs()\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets.q==t].set_index('session').loc[valid_users]\n        \n        # CLEARING Backend before every NN\n        keras.backend.clear_session()        \n        \n        clf = keras.models.Sequential([\n                    keras.layers.Dense(32, activation=\"elu\"),\n                    keras.layers.Dropout(0.5),\n                    keras.layers.Dense(16, activation=\"elu\"),\n                    keras.layers.Dropout(0.25),\n                    keras.layers.Dense(1, activation=\"sigmoid\")\n                ])\n\n        clf.compile(loss='binary_crossentropy',\n                    optimizer=keras.optimizers.Nadam(learning_rate=0.05), \n                    metrics=['accuracy']\n                    )\n\n        history = clf.fit(train_x[FEATURES].astype('float32'), train_y['correct'], epochs=10, verbose = 0)\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        keras_models[f'{grp}_{t}'] = clf\n        keras_p = clf.predict(valid_x[FEATURES].astype('float32'))\n        new_arr = np.hstack((np.zeros(keras_p.shape), keras_p))\n        oof.loc[valid_users, t-1] = new_arr[:,1]    \n    print()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T09:53:07.285899Z","iopub.execute_input":"2023-03-06T09:53:07.286232Z","iopub.status.idle":"2023-03-06T10:01:21.850378Z","shell.execute_reply.started":"2023-03-06T09:53:07.286206Z","shell.execute_reply":"2023-03-06T10:01:21.849535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nkeras_best_score = 0; keras_best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>keras_best_score:\n        keras_best_score = m\n        keras_best_threshold = threshold\n        \nprint('\\n' ,'Best_score:', keras_best_score, '\\n','Best_threshold:', keras_best_threshold)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T10:01:21.851943Z","iopub.execute_input":"2023-03-06T10:01:21.852384Z","iopub.status.idle":"2023-03-06T10:01:24.135920Z","shell.execute_reply.started":"2023-03-06T10:01:21.852357Z","shell.execute_reply":"2023-03-06T10:01:24.134942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([keras_best_threshold], [keras_best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {keras_best_score:.3f} at Best Threshold = {keras_best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T10:01:29.901746Z","iopub.execute_input":"2023-03-06T10:01:29.902085Z","iopub.status.idle":"2023-03-06T10:01:30.108125Z","shell.execute_reply.started":"2023-03-06T10:01:29.902061Z","shell.execute_reply":"2023-03-06T10:01:30.107072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>keras_best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>keras_best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T10:01:37.919730Z","iopub.execute_input":"2023-03-06T10:01:37.920908Z","iopub.status.idle":"2023-03-06T10:01:38.043026Z","shell.execute_reply.started":"2023-03-06T10:01:37.920823Z","shell.execute_reply":"2023-03-06T10:01:38.041934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Neural Network model gets a F1_score = 0.65","metadata":{}},{"cell_type":"markdown","source":"# <a name=\"C35\"> P3.5 : Final model (optimization) </a>","metadata":{}},{"cell_type":"code","source":"gkf = GroupKFold(n_splits=5)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nfinal_models = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        print(t,', ',end='')\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets.q==t].set_index('session').loc[valid_users]\n        \n        # Grid Search\n        \n        clf = RandomForestClassifier(n_estimators=1000, max_depth=4)\n        clf.fit(train_x[FEATURES].astype('float32'), train_y['correct'])\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        final_models[f'{grp}_{t}'] = clf\n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T14:47:11.982652Z","iopub.execute_input":"2023-03-06T14:47:11.983049Z","iopub.status.idle":"2023-03-06T15:05:25.714759Z","shell.execute_reply.started":"2023-03-06T14:47:11.983017Z","shell.execute_reply":"2023-03-06T15:05:25.713563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nfinal_best_score = 0; final_best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>final_best_score:\n        final_best_score = m\n        final_best_threshold = threshold\n        \nprint('\\n' ,'Best_score:', final_best_score, '\\n','Best_threshold:', final_best_threshold)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T15:06:30.059582Z","iopub.execute_input":"2023-03-06T15:06:30.059973Z","iopub.status.idle":"2023-03-06T15:06:33.296837Z","shell.execute_reply.started":"2023-03-06T15:06:30.059941Z","shell.execute_reply":"2023-03-06T15:06:33.295612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([final_best_threshold], [final_best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {final_best_score:.3f} at Best Threshold = {final_best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T15:06:34.361539Z","iopub.execute_input":"2023-03-06T15:06:34.361942Z","iopub.status.idle":"2023-03-06T15:06:34.605885Z","shell.execute_reply.started":"2023-03-06T15:06:34.361909Z","shell.execute_reply":"2023-03-06T15:06:34.604747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>final_best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>final_best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T15:06:35.729815Z","iopub.execute_input":"2023-03-06T15:06:35.730556Z","iopub.status.idle":"2023-03-06T15:06:35.906809Z","shell.execute_reply.started":"2023-03-06T15:06:35.730519Z","shell.execute_reply":"2023-03-06T15:06:35.905798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Final model gets F1_score = 0.669","metadata":{}},{"cell_type":"markdown","source":"# Part 4: Submission","metadata":{}},{"cell_type":"markdown","source":"# <a name=\"C41\"> P4.1 : Submission </a>","metadata":{}},{"cell_type":"code","source":"env = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T15:06:42.673729Z","iopub.execute_input":"2023-03-06T15:06:42.674124Z","iopub.status.idle":"2023-03-06T15:06:42.679031Z","shell.execute_reply.started":"2023-03-06T15:06:42.674091Z","shell.execute_reply":"2023-03-06T15:06:42.678148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\nmodels = final_models\nbest_threshold = final_best_threshold\n\nfor (sample_submission, test) in iter_test:\n    \n    df = feature_engineer(test)\n    grp = test.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        clf = models[f'{grp}_{t}']\n        p = clf.predict_proba(df[FEATURES].astype('float32'))[:,1]\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        sample_submission.loc[mask,'correct'] = int(p.item()>best_threshold)\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T15:06:44.454206Z","iopub.execute_input":"2023-03-06T15:06:44.454627Z","iopub.status.idle":"2023-03-06T15:06:50.239177Z","shell.execute_reply.started":"2023-03-06T15:06:44.454592Z","shell.execute_reply":"2023-03-06T15:06:50.238039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.read_csv('submission.csv')\nprint( submission_df.shape )\nsubmission_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T15:06:55.621275Z","iopub.execute_input":"2023-03-06T15:06:55.621702Z","iopub.status.idle":"2023-03-06T15:06:55.637603Z","shell.execute_reply.started":"2023-03-06T15:06:55.621666Z","shell.execute_reply":"2023-03-06T15:06:55.636102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Submission file","metadata":{}}]}