{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-28T15:49:43.915168Z","iopub.execute_input":"2023-04-28T15:49:43.916763Z","iopub.status.idle":"2023-04-28T15:49:43.999511Z","shell.execute_reply.started":"2023-04-28T15:49:43.916707Z","shell.execute_reply":"2023-04-28T15:49:43.997890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Importing required libraries\nimport numpy as np , gc\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as mpatches\nimport sklearn\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\nimport random\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import f1_score\nimport pandas as pd\nimport pyarrow.parquet as pq\nfrom xgboost import plot_importance","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:44.001530Z","iopub.execute_input":"2023-04-28T15:49:44.001964Z","iopub.status.idle":"2023-04-28T15:49:45.500699Z","shell.execute_reply.started":"2023-04-28T15:49:44.001924Z","shell.execute_reply":"2023-04-28T15:49:45.499464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# Load the parquet file into a PyArrow table\ntable = pq.read_table('/kaggle/input/how-to-get-32gb-ram/train.parquet')\n\n# Convert the PyArrow table to a Pandas dataframe\ntrain_df = table.to_pandas()\ntrain_df.describe()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:45.503477Z","iopub.execute_input":"2023-04-28T15:49:45.504261Z","iopub.status.idle":"2023-04-28T15:49:55.568909Z","shell.execute_reply.started":"2023-04-28T15:49:45.504202Z","shell.execute_reply":"2023-04-28T15:49:55.562819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dropping empty columns in df\ntrain_df = train_df.drop(columns=['fullscreen','hq', 'music'])","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.570191Z","iopub.status.idle":"2023-04-28T15:49:55.570890Z","shell.execute_reply.started":"2023-04-28T15:49:55.570682Z","shell.execute_reply":"2023-04-28T15:49:55.570707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#making a table for checkpoints\ntrain_df_checkpoint = train_df[(train_df.event_name == \"checkpoint\")]\ntrain_df_checkpoint","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.572099Z","iopub.status.idle":"2023-04-28T15:49:55.572723Z","shell.execute_reply.started":"2023-04-28T15:49:55.572526Z","shell.execute_reply":"2023-04-28T15:49:55.572549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we are looking at the checkpoints where people are supposed to answer Q's. There should be only 3 checkpoints","metadata":{}},{"cell_type":"markdown","source":"It seems there's a number of sessions with more than 3 checkpoints which is troubling.","metadata":{}},{"cell_type":"code","source":"#reading in training labels\ntable_labels = pq.read_table('/kaggle/input/how-to-get-32gb-ram/train_labels.parquet')\n\n# Convert the PyArrow table to a Pandas dataframe\ntrain_df_labels = table_labels.to_pandas()\ntrain_df_labels","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.573828Z","iopub.status.idle":"2023-04-28T15:49:55.574453Z","shell.execute_reply.started":"2023-04-28T15:49:55.574234Z","shell.execute_reply":"2023-04-28T15:49:55.574257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SEPARTING SESSION ID AND QUESTION FOR MODELLING\ntrain_df_labels['session'] = train_df_labels.session_id.apply(lambda x: int(x.split('_')[0]) )\ntrain_df_labels['q'] = train_df_labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.575555Z","iopub.status.idle":"2023-04-28T15:49:55.576172Z","shell.execute_reply.started":"2023-04-28T15:49:55.575971Z","shell.execute_reply":"2023-04-28T15:49:55.575997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#PRINTINT OUT Y TRAINS\nprint(train_df_labels)","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.577267Z","iopub.status.idle":"2023-04-28T15:49:55.577948Z","shell.execute_reply.started":"2023-04-28T15:49:55.577739Z","shell.execute_reply":"2023-04-28T15:49:55.577761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CREATING CATEGORIES AND SPLITING UP DATA TO AVOID MEMORY ERRORS\nCATS = ['event_name', 'fqid', 'room_fqid', 'text', 'event_comb',\n      \"event_index\" , \"name_index\" ]\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration', 'et_diff' , \"checkpoint_sum\"] #,'checkpoint_sum']\n\n# https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data\nEVENTS = ['navigate_click','person_click','cutscene_click','object_click',\n          'map_hover','notification_click','map_click','observation_click',\n          'checkpoint']\n\n\ntmp= train_df.iloc[:,0]\ntmp = pd.DataFrame(tmp)\ntmp = tmp.groupby('session_id').session_id.agg('count')\n\n# COMPUTE READS AND SKIPS\nPIECES = 10\nCHUNK = int( np.ceil(len(tmp)/PIECES) )\n\nreads = []\nskips = [0]\nfor k in range(PIECES):\n    a = k*CHUNK\n    b = (k+1)*CHUNK\n    if b>len(tmp): b=len(tmp)\n    r = tmp.iloc[a:b].sum()\n    reads.append(r)\n    skips.append(skips[-1]+r)\n    \nprint(f'To avoid memory error, we will read train in {PIECES} pieces of sizes:')\nprint(reads)\ntmp.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.579072Z","iopub.status.idle":"2023-04-28T15:49:55.579706Z","shell.execute_reply.started":"2023-04-28T15:49:55.579509Z","shell.execute_reply":"2023-04-28T15:49:55.579532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train_df):\n    # Create new columns\n    #CREATING EVENT COMBINATION FIELD\n    train_df[\"event_comb\"] = train_df[\"event_name\"] + \"_\" + train_df[\"name\"]\n    # CREATING EVENT INDEX FIELD\n    train_df[\"event_index\"] = train_df[\"event_name\"] + \"_\" + train_df[\"index\"].astype(str)\n    # CREATING EVENT NAME INDEX FIELD\n    train_df[\"name_index\"] = train_df[\"name\"] + \"_\" + train_df[\"index\"].astype(str)\n    # CREATING FIELD FOR ELAPSED TIME DIFFERENCE BETWEEN EVENTS\n    train_df[\"et_diff\"] = train_df.groupby(\"session_id\", sort=False).apply(lambda x: x[\"elapsed_time\"].diff().fillna(-1)).values\n    #CREATING FIELD FOR CHECKPOINTS REACHED BY PLAYER\n    train_df[\"checkpoint_sum\"] = train_df.groupby([\"session_id\", \"level_group\"]).apply(lambda x: (x[\"event_name\"] == \"checkpoint\").sum()).reset_index().iloc[:, -1]\n    # Create new features\n    dfs = []\n    for c in CATS:\n        tmp = train_df.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS + ['index']: \n        tmp = train_df.groupby(['session_id','level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train_df.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    for c in EVENTS: \n        train_df[c] = (train_df.event_name == c).astype('int8')\n    for c in EVENTS + ['elapsed_time','index']:\n        tmp = train_df.groupby(['session_id','level_group'])[c].agg('sum')\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n    train_df = train_df.drop(EVENTS,axis=1)\n        \n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.580835Z","iopub.status.idle":"2023-04-28T15:49:55.581481Z","shell.execute_reply.started":"2023-04-28T15:49:55.581264Z","shell.execute_reply":"2023-04-28T15:49:55.581302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# PROCESS TRAIN DATA IN PIECES\nall_pieces = []\nprint(f'Processing train as {PIECES} pieces to avoid memory error... ')\nfor k in range(PIECES):\n    print(k,', ',end='')\n    SKIPS = 0\n    if k>0: SKIPS = range(1,skips[k]+1)\n    df = feature_engineer(train_df)\n    all_pieces.append(df)\n    \n# CONCATENATE ALL PIECES\nprint('\\n')\ndel train_df; gc.collect()\ndf = pd.concat(all_pieces, axis=0)\nprint('Shape of all train data after feature engineering:', df.shape )\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.582605Z","iopub.status.idle":"2023-04-28T15:49:55.583207Z","shell.execute_reply.started":"2023-04-28T15:49:55.583008Z","shell.execute_reply":"2023-04-28T15:49:55.583031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PRINTING OUT HOW MANY FEATURES AND USERS WE HAVE\nFEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\nALL_USERS = df.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.584698Z","iopub.status.idle":"2023-04-28T15:49:55.585424Z","shell.execute_reply.started":"2023-04-28T15:49:55.585163Z","shell.execute_reply":"2023-04-28T15:49:55.585198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gkf = GroupKFold(n_splits=5)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nmodels = {}\n# https://www.kaggle.com/code/cdeotte/xgboost-baseline-0-680 \n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    xgb_params = {\n    'objective' : 'binary:logistic',\n    'eval_metric':'logloss',\n    'learning_rate': 0.05,\n    'max_depth': 4,\n    'n_estimators': 1000,\n    'early_stopping_rounds': 50,\n    'tree_method':'hist',\n    'subsample':0.8,\n    'colsample_bytree': 0.4,\n    'use_label_encoder' : False}\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = train_df_labels.loc[train_df_labels.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = train_df_labels.loc[train_df_labels.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL        \n        clf =  XGBClassifier(**xgb_params)\n        clf.fit(train_x[FEATURES].astype('float32'), train_y['correct'],\n                eval_set=[ (valid_x[FEATURES].astype('float32'), valid_y['correct']) ],\n                verbose=0)\n        print(f'{t}({clf.best_ntree_limit}), ',end='')\n         # SAVE MODEL, PREDICT VALID OOF\n        clf.save_model(f'XGB_question{t}.xgb')\n        \n        # feature importance\n        plot_importance(clf)\n        plt.show()\n        \n        # predicting validation oof\n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n        \n    print()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-04-28T15:49:55.614687Z","iopub.status.idle":"2023-04-28T15:49:55.615740Z","shell.execute_reply.started":"2023-04-28T15:49:55.615499Z","shell.execute_reply":"2023-04-28T15:49:55.615531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = train_df_labels.loc[train_df_labels.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.617155Z","iopub.status.idle":"2023-04-28T15:49:55.617807Z","shell.execute_reply.started":"2023-04-28T15:49:55.617609Z","shell.execute_reply":"2023-04-28T15:49:55.617632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(FEATURES)","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.618922Z","iopub.status.idle":"2023-04-28T15:49:55.619474Z","shell.execute_reply.started":"2023-04-28T15:49:55.619304Z","shell.execute_reply":"2023-04-28T15:49:55.619325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0,1,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.620535Z","iopub.status.idle":"2023-04-28T15:49:55.621081Z","shell.execute_reply.started":"2023-04-28T15:49:55.620911Z","shell.execute_reply":"2023-04-28T15:49:55.620931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.622114Z","iopub.status.idle":"2023-04-28T15:49:55.622648Z","shell.execute_reply.started":"2023-04-28T15:49:55.622435Z","shell.execute_reply":"2023-04-28T15:49:55.622461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.623700Z","iopub.status.idle":"2023-04-28T15:49:55.624206Z","shell.execute_reply.started":"2023-04-28T15:49:55.624041Z","shell.execute_reply":"2023-04-28T15:49:55.624061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import log_loss\nfrom joblib import dump\n\ngkf = GroupKFold(n_splits=5)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nlogmodels = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    lr_params = {\n        'penalty': 'l2',\n        'C': 1.0,\n        'solver': 'lbfgs',\n        'max_iter': 1000\n    }\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = train_df_labels.loc[train_df_labels.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = train_df_labels.loc[train_df_labels.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL        \n        log = LogisticRegression(**lr_params)\n        log.fit(train_x[FEATURES].astype('float32'), train_y['correct'])\n            \n        # SAVE MODEL, PREDICT VALID OOF\n        dump(log, f'logistic_regression_model_question{t}.joblib')\n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n        \n        # EVALUATING LOG LOSS FOR DATASET\n        ll = log_loss(valid_y['correct'], clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1])\n        print(f'{t}({ll:.4f}), ',end='')\n        \n    print()","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.625144Z","iopub.status.idle":"2023-04-28T15:49:55.625655Z","shell.execute_reply.started":"2023-04-28T15:49:55.625491Z","shell.execute_reply":"2023-04-28T15:49:55.625511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = train_df_labels.loc[train_df_labels.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.626576Z","iopub.status.idle":"2023-04-28T15:49:55.627086Z","shell.execute_reply.started":"2023-04-28T15:49:55.626914Z","shell.execute_reply":"2023-04-28T15:49:55.626942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.01,1,.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.628019Z","iopub.status.idle":"2023-04-28T15:49:55.628543Z","shell.execute_reply.started":"2023-04-28T15:49:55.628380Z","shell.execute_reply":"2023-04-28T15:49:55.628400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.629465Z","iopub.status.idle":"2023-04-28T15:49:55.629964Z","shell.execute_reply.started":"2023-04-28T15:49:55.629799Z","shell.execute_reply":"2023-04-28T15:49:55.629819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.630875Z","iopub.status.idle":"2023-04-28T15:49:55.631394Z","shell.execute_reply.started":"2023-04-28T15:49:55.631206Z","shell.execute_reply":"2023-04-28T15:49:55.631226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import log_loss\n\ngkf = GroupKFold(n_splits=2)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nrfmodels = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    rf_params = {\n        'n_estimators': 1000,\n        'max_depth': 6,\n        'min_samples_split': 5,\n        'min_samples_leaf': 5,\n        'max_features': 'sqrt',\n        'bootstrap': True,\n        'n_jobs': -1,\n        'random_state': 42\n    }\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = train_df_labels.loc[train_df_labels.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = train_df_labels.loc[train_df_labels.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL        \n        rf =  RandomForestClassifier(**rf_params)\n        rf.fit(train_x[FEATURES].astype('float32'), train_y['correct'])\n        # Saving Models\n        dump(rf, f'random_forest_question{t}.joblib')\n\n        #predicting validation oof\n        oof.loc[valid_users, t-1] = rf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n        \n        # Evaluate Log Loss for validation set\n        bet = log_loss(valid_y['correct'], rf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1])\n        print(f'{t}({bet:.4f}), ',end='')\n        \n    print()","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.632451Z","iopub.status.idle":"2023-04-28T15:49:55.632765Z","shell.execute_reply.started":"2023-04-28T15:49:55.632606Z","shell.execute_reply":"2023-04-28T15:49:55.632627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CALLING OUT HOW MANY ROWS ARE IN THE DATAFRAME\nlen(df)","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.633896Z","iopub.status.idle":"2023-04-28T15:49:55.634204Z","shell.execute_reply.started":"2023-04-28T15:49:55.634051Z","shell.execute_reply":"2023-04-28T15:49:55.634069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = train_df_labels.loc[train_df_labels.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.635434Z","iopub.status.idle":"2023-04-28T15:49:55.636169Z","shell.execute_reply.started":"2023-04-28T15:49:55.635989Z","shell.execute_reply":"2023-04-28T15:49:55.636014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.01,1,.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.637264Z","iopub.status.idle":"2023-04-28T15:49:55.637622Z","shell.execute_reply.started":"2023-04-28T15:49:55.637465Z","shell.execute_reply":"2023-04-28T15:49:55.637483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.638685Z","iopub.status.idle":"2023-04-28T15:49:55.638997Z","shell.execute_reply.started":"2023-04-28T15:49:55.638844Z","shell.execute_reply":"2023-04-28T15:49:55.638861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.640463Z","iopub.status.idle":"2023-04-28T15:49:55.641229Z","shell.execute_reply.started":"2023-04-28T15:49:55.641066Z","shell.execute_reply":"2023-04-28T15:49:55.641085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Balanced Classes Models**\n","metadata":{}},{"cell_type":"code","source":"from sklearn.utils import class_weight\nfrom sklearn.preprocessing import LabelEncoder\ngkf = GroupKFold(n_splits=5)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nmodels = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    xgb_params = {\n    'objective' : 'binary:logistic',\n    'eval_metric':'logloss',\n    'learning_rate': 0.05,\n    'max_depth': 4,\n    'n_estimators': 1000,\n    'early_stopping_rounds': 50,\n    'tree_method':'hist',\n    'subsample':0.8,\n    'colsample_bytree': 0.4,\n    'use_label_encoder' : False}\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = train_df_labels.loc[train_df_labels.q==t].set_index('session').loc[train_users]\n        le = LabelEncoder()\n        train_y = le.fit_transform(train_y['correct'])\n        # Calculate class weights\n        class_weights = class_weight.compute_class_weight('balanced', classes=np.unique(train_y), y=train_y)\n        class_weights = dict(enumerate(class_weights))\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = train_df_labels.loc[train_df_labels.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL        \n        clf =  XGBClassifier(**xgb_params, class_weight=class_weights)\n        clf.fit(train_x[FEATURES].astype('float32'), train_y,\n                eval_set=[ (valid_x[FEATURES].astype('float32'), valid_y['correct']) ],\n                verbose=0)\n        print(f'{t}({clf.best_ntree_limit}), ',end='')\n        clf.save_model(f'XGB_question_balanced{t}.xgb')\n        \n        # feature importance\n        plot_importance(clf)\n        plt.show()\n            \n        # PREDICT VALID OOF\n        \n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n        \n    print()","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.642065Z","iopub.status.idle":"2023-04-28T15:49:55.642884Z","shell.execute_reply.started":"2023-04-28T15:49:55.642694Z","shell.execute_reply":"2023-04-28T15:49:55.642716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = train_df_labels.loc[train_df_labels.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.643822Z","iopub.status.idle":"2023-04-28T15:49:55.644578Z","shell.execute_reply.started":"2023-04-28T15:49:55.644410Z","shell.execute_reply":"2023-04-28T15:49:55.644432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0,1,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.645589Z","iopub.status.idle":"2023-04-28T15:49:55.645916Z","shell.execute_reply.started":"2023-04-28T15:49:55.645742Z","shell.execute_reply":"2023-04-28T15:49:55.645764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.647032Z","iopub.status.idle":"2023-04-28T15:49:55.647391Z","shell.execute_reply.started":"2023-04-28T15:49:55.647195Z","shell.execute_reply":"2023-04-28T15:49:55.647214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.648237Z","iopub.status.idle":"2023-04-28T15:49:55.648577Z","shell.execute_reply.started":"2023-04-28T15:49:55.648423Z","shell.execute_reply":"2023-04-28T15:49:55.648441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import log_loss\nfrom joblib import dump\nfrom imblearn.over_sampling import RandomOverSampler\nfrom sklearn.utils import class_weight\nfrom sklearn.preprocessing import LabelEncoder\n\ngkf = GroupKFold(n_splits=5)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nlogmodels = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    lr_params = {\n        'penalty': 'l2',\n        'C': 1.0,\n        'solver': 'lbfgs',\n        'max_iter': 1000\n    }\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = train_df_labels.loc[train_df_labels.q==t].set_index('session').loc[train_users]\n        le = LabelEncoder()\n        train_y = le.fit_transform(train_y['correct'])\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = train_df_labels.loc[train_df_labels.q==t].set_index('session').loc[valid_users]\n        \n        # Handle class imbalance in training data\n        oversample = RandomOverSampler(sampling_strategy='minority')\n        train_x, train_y = oversample.fit_resample(train_x[FEATURES].astype('float32'), train_y)\n        \n        # TRAIN MODEL        \n        log = LogisticRegression(**lr_params)\n        log.fit(train_x[FEATURES].astype('float32'), train_y)\n            \n        # SAVE MODEL FOR INFERENCE\n        dump(log, f'logistic_regression_model_balanced_question{t}.joblib')\n        \n        # PREDICT VALID OOF\n        oof.loc[valid_users, t-1] = log.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n        \n        # Evaluate Log Loss for validation set\n        ll = log_loss(valid_y['correct'], log.predict_proba(valid_x[FEATURES].astype('float32'))[:,1])\n        print(f'{t}({ll:.4f}), ',end='')\n        \n    print()","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.649488Z","iopub.status.idle":"2023-04-28T15:49:55.649794Z","shell.execute_reply.started":"2023-04-28T15:49:55.649630Z","shell.execute_reply":"2023-04-28T15:49:55.649647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = train_df_labels.loc[train_df_labels.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.650716Z","iopub.status.idle":"2023-04-28T15:49:55.651019Z","shell.execute_reply.started":"2023-04-28T15:49:55.650862Z","shell.execute_reply":"2023-04-28T15:49:55.650879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.01,1,.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.651810Z","iopub.status.idle":"2023-04-28T15:49:55.652096Z","shell.execute_reply.started":"2023-04-28T15:49:55.651948Z","shell.execute_reply":"2023-04-28T15:49:55.651965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.653390Z","iopub.status.idle":"2023-04-28T15:49:55.653701Z","shell.execute_reply.started":"2023-04-28T15:49:55.653546Z","shell.execute_reply":"2023-04-28T15:49:55.653564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.654479Z","iopub.status.idle":"2023-04-28T15:49:55.654774Z","shell.execute_reply.started":"2023-04-28T15:49:55.654621Z","shell.execute_reply":"2023-04-28T15:49:55.654638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import log_loss\nfrom joblib import dump\n\ngkf = GroupKFold(n_splits=2)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nrfmodels = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    rf_params = {\n        'n_estimators': 1000,\n        'max_depth': 6,\n        'min_samples_split': 5,\n        'min_samples_leaf': 5,\n        'max_features': 'sqrt',\n        'bootstrap': True,\n        'n_jobs': -1,\n        'random_state': 42, \n        'class_weight': 'balanced'\n    }\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = train_df_labels.loc[train_df_labels.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = train_df_labels.loc[train_df_labels.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL        \n        rf =  RandomForestClassifier(**rf_params)\n        rf.fit(train_x[FEATURES].astype('float32'), train_y['correct'])\n        \n        # SAVE MODEL FOR INFERENCE\n        dump(rf, f'random_forest_balanced_question{t}.joblib')\n        \n        # PREDICT VALID OOF\n        oof.loc[valid_users, t-1] = rf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n        \n        # Evaluate Log Loss for validation set\n        bet = log_loss(valid_y['correct'], rf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1])\n        print(f'{t}({bet:.4f}), ',end='')\n        \n    print()","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.655737Z","iopub.status.idle":"2023-04-28T15:49:55.656030Z","shell.execute_reply.started":"2023-04-28T15:49:55.655884Z","shell.execute_reply":"2023-04-28T15:49:55.655900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = train_df_labels.loc[train_df_labels.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.657895Z","iopub.status.idle":"2023-04-28T15:49:55.658194Z","shell.execute_reply.started":"2023-04-28T15:49:55.658043Z","shell.execute_reply":"2023-04-28T15:49:55.658060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.01,1,.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.659171Z","iopub.status.idle":"2023-04-28T15:49:55.659499Z","shell.execute_reply.started":"2023-04-28T15:49:55.659344Z","shell.execute_reply":"2023-04-28T15:49:55.659362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.660566Z","iopub.status.idle":"2023-04-28T15:49:55.660863Z","shell.execute_reply.started":"2023-04-28T15:49:55.660710Z","shell.execute_reply":"2023-04-28T15:49:55.660727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-04-28T15:49:55.662968Z","iopub.status.idle":"2023-04-28T15:49:55.663526Z","shell.execute_reply.started":"2023-04-28T15:49:55.663248Z","shell.execute_reply":"2023-04-28T15:49:55.663276Z"},"trusted":true},"execution_count":null,"outputs":[]}]}