{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# WE R (Kajol, Manu, Mili) Pace University \n\n\n<https://www.kaggle.com/mannubhardwaj1022/we-r-xgb-boost-model>","metadata":{}},{"cell_type":"code","source":"# Import necessary libraries\nimport pandas as pd\nimport numpy as np\nimport gc\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import f1_score","metadata":{"execution":{"iopub.status.busy":"2023-05-05T17:56:05.482944Z","iopub.execute_input":"2023-05-05T17:56:05.483358Z","iopub.status.idle":"2023-05-05T17:56:06.382015Z","shell.execute_reply.started":"2023-05-05T17:56:05.483327Z","shell.execute_reply":"2023-05-05T17:56:06.381188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install memory_profiler\n%load_ext memory_profiler","metadata":{"papermill":{"duration":31.476109,"end_time":"2023-04-25T11:12:23.801378","exception":false,"start_time":"2023-04-25T11:11:52.325269","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T17:56:06.386781Z","iopub.execute_input":"2023-05-05T17:56:06.388629Z","iopub.status.idle":"2023-05-05T17:56:39.606654Z","shell.execute_reply.started":"2023-05-05T17:56:06.388596Z","shell.execute_reply":"2023-05-05T17:56:39.605553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Train Data and Labels\n","metadata":{"papermill":{"duration":0.008793,"end_time":"2023-04-25T11:12:23.819173","exception":false,"start_time":"2023-04-25T11:12:23.810380","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\ndtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\n# Read user ID only to get the row count for creating chunk sizes\ndata = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\", usecols=[0], dtype=dtypes)\nrow_count = data.groupby('session_id').session_id.agg('count')\n\n# Compute reads and skips\npieces = 10\nchunk_size = int(np.ceil(len(row_count) / pieces))\n\n# Create data pieces\nreads = []\nskips = [0]\nfor k in range(pieces):\n    start_index = k * chunk_size\n    end_index = (k + 1) * chunk_size\n    read_sum = row_count.iloc[start_index:end_index].sum()\n    reads.append(read_sum)\n    skips.append(skips[-1] + read_sum)\n\n# Print the chunk sizes to avoid memory errors\nprint(f'To avoid memory errors, we will read \"train.csv\" in {pieces} pieces of the following sizes:')\nprint(reads)\n","metadata":{"_kg_hide-input":true,"papermill":{"duration":75.537684,"end_time":"2023-04-25T11:13:39.365740","exception":false,"start_time":"2023-04-25T11:12:23.828056","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T17:56:39.608369Z","iopub.execute_input":"2023-05-05T17:56:39.608918Z","iopub.status.idle":"2023-05-05T17:58:11.957234Z","shell.execute_reply.started":"2023-05-05T17:56:39.608882Z","shell.execute_reply":"2023-05-05T17:58:11.956307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\ndef reduce_memory_usage(df):\n    # Calculate the initial memory usage of the DataFrame\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of the DataFrame is {:.2f} MB'.format(start_mem))\n    \n    # Iterate over each column in the DataFrame\n    for col in df.columns:\n        # Get the data type of the column\n        col_type = df[col].dtype.name\n        \n        # Skip datetime and category columns\n        if col_type != 'datetime64[ns]' and col_type != 'category':\n            # Skip object columns and reduce memory usage of numeric columns\n            if col_type != 'object':\n                # Get the minimum and maximum values of the column\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                # Check if the column is of integer type\n                if str(col_type)[:3] == 'int':\n                    # Iterate over possible integer data types and convert if necessary\n                    for dtype in [np.int8, np.int16, np.int32, np.int64]:\n                        if c_min > np.iinfo(dtype).min and c_max < np.iinfo(dtype).max:\n                            df[col] = df[col].astype(dtype)\n                            break\n\n                # Check if the column is of floating-point type\n                else:\n                    # Iterate over possible floating-point data types and convert if necessary\n                    for dtype in [np.float16, np.float32]:\n                        if c_min > np.finfo(dtype).min and c_max < np.finfo(dtype).max:\n                            df[col] = df[col].astype(dtype)\n                            break\n            else:\n                # Convert object columns to the category type\n                df[col] = df[col].astype('category')\n    \n    # Calculate and display the reduced memory usage of the DataFrame\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"Memory usage reduced to: {:.2f} MB\".format(mem_usg))\n    \n    return df\n","metadata":{"papermill":{"duration":0.029454,"end_time":"2023-04-25T11:13:39.405054","exception":false,"start_time":"2023-04-25T11:13:39.375600","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T17:58:11.959599Z","iopub.execute_input":"2023-05-05T17:58:11.960175Z","iopub.status.idle":"2023-05-05T17:58:11.971407Z","shell.execute_reply.started":"2023-05-05T17:58:11.960140Z","shell.execute_reply":"2023-05-05T17:58:11.969305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LOAD DATA IN PIECES, THEN REDUCE MEMORY USAGE\nall_pieces = []\nprint(f'Loading train as {pieces} pieces to avoid memory error... ')\nfor k in range(pieces):\n    print(k,', ',end='')\n    SKIPS = 0\n    if k>0: SKIPS = range(1,skips[k]+1)\n    train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv',\n                        nrows=reads[k], skiprows=SKIPS)\n    df = reduce_memory_usage(train)\n    all_pieces.append(df)","metadata":{"papermill":{"duration":129.657579,"end_time":"2023-04-25T11:15:49.072443","exception":false,"start_time":"2023-04-25T11:13:39.414864","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T17:58:11.974850Z","iopub.execute_input":"2023-05-05T17:58:11.975598Z","iopub.status.idle":"2023-05-05T18:04:18.876506Z","shell.execute_reply.started":"2023-05-05T17:58:11.975558Z","shell.execute_reply":"2023-05-05T18:04:18.873872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Concatenate all the data pieces to create the full train data\nprint('\\n')\ndel train  # Remove the temporary train variable to free up memory\ngc.collect()  # Perform garbage collection to release memory\n\npre_df = pd.concat(all_pieces, axis=0)  # Concatenate the data pieces along the row axis\nprint('Shape of all train data:', pre_df.shape)\npre_df.head()\n","metadata":{"papermill":{"duration":3.24057,"end_time":"2023-04-25T11:15:52.322843","exception":false,"start_time":"2023-04-25T11:15:49.082273","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:04:18.881529Z","iopub.execute_input":"2023-05-05T18:04:18.881945Z","iopub.status.idle":"2023-05-05T18:04:22.365831Z","shell.execute_reply.started":"2023-05-05T18:04:18.881908Z","shell.execute_reply":"2023-05-05T18:04:22.364031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the label data\ntargets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\n\n# Extract the session id and question information into separate columns\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]))\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]))\n\nprint(targets.shape)  # Print the shape of the label data\ntargets.head()  # Display the first few rows of the label data\n","metadata":{"papermill":{"duration":1.138864,"end_time":"2023-04-25T11:15:53.473572","exception":false,"start_time":"2023-04-25T11:15:52.334708","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:04:22.367376Z","iopub.execute_input":"2023-05-05T18:04:22.367726Z","iopub.status.idle":"2023-05-05T18:04:23.423288Z","shell.execute_reply.started":"2023-05-05T18:04:22.367698Z","shell.execute_reply":"2023-05-05T18:04:23.422523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets.tail()","metadata":{"papermill":{"duration":0.025685,"end_time":"2023-04-25T11:15:53.509633","exception":false,"start_time":"2023-04-25T11:15:53.483948","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:04:23.424591Z","iopub.execute_input":"2023-05-05T18:04:23.425670Z","iopub.status.idle":"2023-05-05T18:04:23.436164Z","shell.execute_reply.started":"2023-05-05T18:04:23.425637Z","shell.execute_reply":"2023-05-05T18:04:23.435041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory Data Analysis\n\n","metadata":{"papermill":{"duration":0.009896,"end_time":"2023-04-25T11:15:53.530025","exception":false,"start_time":"2023-04-25T11:15:53.520129","status":"completed"},"tags":[]}},{"cell_type":"code","source":"pre_df.info()","metadata":{"papermill":{"duration":0.066915,"end_time":"2023-04-25T11:15:53.607358","exception":false,"start_time":"2023-04-25T11:15:53.540443","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:04:23.437588Z","iopub.execute_input":"2023-05-05T18:04:23.437990Z","iopub.status.idle":"2023-05-05T18:04:23.469406Z","shell.execute_reply.started":"2023-05-05T18:04:23.437962Z","shell.execute_reply":"2023-05-05T18:04:23.467917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def summary(df):\n    # Print the shape of the DataFrame\n    print(f'data shape: {df.shape}')\n    \n    # Create a summary DataFrame\n    summ = pd.DataFrame(df.dtypes, columns=['data type'])\n    \n    # Calculate the number and percentage of missing values for each column\n    summ['#missing'] = df.isnull().sum().values\n    summ['%missing'] = df.isnull().sum().values / len(df)\n    \n    # Calculate the number of unique values for each column\n    summ['#unique'] = df.nunique().values\n    \n    # Calculate the minimum and maximum values for each column\n    desc = pd.DataFrame(df.describe(include='all').transpose())\n    summ['min'] = desc['min'].values\n    summ['max'] = desc['max'].values\n    \n    return summ\n","metadata":{"papermill":{"duration":0.057242,"end_time":"2023-04-25T11:15:53.691962","exception":false,"start_time":"2023-04-25T11:15:53.634720","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:04:23.476565Z","iopub.execute_input":"2023-05-05T18:04:23.476982Z","iopub.status.idle":"2023-05-05T18:04:23.487775Z","shell.execute_reply.started":"2023-05-05T18:04:23.476950Z","shell.execute_reply":"2023-05-05T18:04:23.486526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display a summary of the data properties (missing values, unique values, min, max)\nsummary_table = summary(pre_df)\nsummary_table","metadata":{"papermill":{"duration":45.747863,"end_time":"2023-04-25T11:16:39.462944","exception":false,"start_time":"2023-04-25T11:15:53.715081","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:04:23.489802Z","iopub.execute_input":"2023-05-05T18:04:23.490206Z","iopub.status.idle":"2023-05-05T18:05:31.390721Z","shell.execute_reply.started":"2023-05-05T18:04:23.490172Z","shell.execute_reply":"2023-05-05T18:05:31.389281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pre_df['event_name'].value_counts()","metadata":{"papermill":{"duration":0.228919,"end_time":"2023-04-25T11:16:39.702609","exception":false,"start_time":"2023-04-25T11:16:39.473690","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:05:31.392530Z","iopub.execute_input":"2023-05-05T18:05:31.392825Z","iopub.status.idle":"2023-05-05T18:05:31.584072Z","shell.execute_reply.started":"2023-05-05T18:05:31.392799Z","shell.execute_reply":"2023-05-05T18:05:31.582892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineer\nWe create basic aggregate features. \n","metadata":{"papermill":{"duration":0.012287,"end_time":"2023-04-25T11:16:39.751183","exception":false,"start_time":"2023-04-25T11:16:39.738896","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Define the list of categorical columns\nCATS = ['event_name', 'fqid', 'room_fqid', 'text']\n\n# Define the list of numerical columns\nNUMS = ['elapsed_time', 'level', 'page', 'room_coor_x', 'room_coor_y',\n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\n\n# Define the list of event columns\nEVENTS = ['navigate_click', 'person_click', 'cutscene_click', 'object_click',\n          'map_hover', 'notification_click', 'map_click', 'observation_click',\n          'checkpoint']\n","metadata":{"papermill":{"duration":0.021401,"end_time":"2023-04-25T11:16:39.784778","exception":false,"start_time":"2023-04-25T11:16:39.763377","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:05:31.585708Z","iopub.execute_input":"2023-05-05T18:05:31.586006Z","iopub.status.idle":"2023-05-05T18:05:31.591685Z","shell.execute_reply.started":"2023-05-05T18:05:31.585979Z","shell.execute_reply":"2023-05-05T18:05:31.590391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train):\n    dfs = []\n    \n    # Calculate the number of unique values per categorical feature per level group within each session\n    for c in CATS:\n        tmp = train.groupby(['session_id', 'level_group'])[c].nunique()\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    \n    # Calculate the mean of the numerical features per level group within each session\n    for c in NUMS:\n        tmp = train.groupby(['session_id', 'level_group'])[c].mean()\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    \n    # Calculate the standard deviation of the numerical features per level group within each session\n    for c in NUMS:\n        tmp = train.groupby(['session_id', 'level_group'])[c].std()\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    \n    # Convert event occurrences into binary features\n    for c in EVENTS:\n        train[c] = (train.event_name == c).astype('int8')\n    \n    # Calculate the sum of event occurrences per level group within each session\n    for c in EVENTS + ['elapsed_time']:\n        tmp = train.groupby(['session_id', 'level_group'])[c].sum()\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n    \n    # Remove the original event columns\n    train = train.drop(EVENTS, axis=1)\n    \n    # Concatenate all the engineered features into a single DataFrame\n    df = pd.concat(dfs, axis=1)\n    \n    # Fill missing values with -1\n    df = df.fillna(-1)\n    \n    # Reset the index and set the session_id as the new index\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    \n    return df\n","metadata":{"papermill":{"duration":0.025867,"end_time":"2023-04-25T11:16:39.821665","exception":false,"start_time":"2023-04-25T11:16:39.795798","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:05:31.593403Z","iopub.execute_input":"2023-05-05T18:05:31.593731Z","iopub.status.idle":"2023-05-05T18:05:31.605527Z","shell.execute_reply.started":"2023-05-05T18:05:31.593704Z","shell.execute_reply":"2023-05-05T18:05:31.604493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create the features per piece of training data and display the head\ndf = feature_engineer(pre_df)\ndf.head()","metadata":{"papermill":{"duration":59.802417,"end_time":"2023-04-25T11:17:39.634970","exception":false,"start_time":"2023-04-25T11:16:39.832553","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:05:31.607288Z","iopub.execute_input":"2023-05-05T18:05:31.607605Z","iopub.status.idle":"2023-05-05T18:06:39.647965Z","shell.execute_reply.started":"2023-05-05T18:05:31.607578Z","shell.execute_reply":"2023-05-05T18:06:39.647148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train XGBoost Model\nWe train one model for each of 18 questions. Furthermore, we use data from `level_groups = '0-4'` to train model for questions 1-3, and `level groups '5-12'` to train questions 4 thru 13 and `level groups '13-22'` to train questions 14 thru 18. Because this is the data we get (to predict corresponding questions) from Kaggle's inference API during test inference. We can improve our model by saving a user's previous data from earlier `level_groups` and using that to predict future `level_groups`.","metadata":{"papermill":{"duration":0.011745,"end_time":"2023-04-25T11:17:39.658063","exception":false,"start_time":"2023-04-25T11:17:39.646318","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Get the list of features excluding the 'level_group' column\nFEATURES = [c for c in df.columns if c != 'level_group']\nprint('Number of features:', len(FEATURES))\n\n# Get the unique session_ids (users) in the data\nALL_USERS = df.index.unique()\nprint('Number of unique users:', len(ALL_USERS))\n","metadata":{"papermill":{"duration":0.025419,"end_time":"2023-04-25T11:17:39.694972","exception":false,"start_time":"2023-04-25T11:17:39.669553","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:06:39.649029Z","iopub.execute_input":"2023-05-05T18:06:39.649491Z","iopub.status.idle":"2023-05-05T18:06:39.657907Z","shell.execute_reply.started":"2023-05-05T18:06:39.649445Z","shell.execute_reply":"2023-05-05T18:06:39.657196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"``GroupKFold`` **is a variation of k-fold which ensures that the same group is not represented in both testing/validation and training sets. For example if the data is obtained from different subjects with several samples per-subject and if the model is flexible enough to learn from highly person specific features it could fail to generalize to new subjects.** ``GroupKFold`` **makes it possible to detect this kind of overfitting situations.**\n","metadata":{"papermill":{"duration":0.011794,"end_time":"2023-04-25T11:17:39.718530","exception":false,"start_time":"2023-04-25T11:17:39.706736","status":"completed"},"tags":[]}},{"cell_type":"code","source":"gkf = GroupKFold(n_splits=10)\n\n# Create an empty DataFrame to collect out-of-fold predictions\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS), 18)), index=ALL_USERS)\nmodels = {}\n\n# Perform cross-validation with 5 GroupKFold splits\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#' * 25)\n    print('### Fold', i + 1)\n    print('#' * 25)\n    \n    # Define XGBoost hyperparameters\n    xgb_params = {\n        'objective': 'binary:logistic',\n        'eval_metric': 'logloss',\n        'learning_rate': 0.05,\n        'max_depth': 4,\n        'n_estimators': 1000,\n        'early_stopping_rounds': 50,\n        'tree_method': 'hist',\n        'subsample': 0.8,\n        'colsample_bytree': 0.4,\n        'use_label_encoder': False\n    }\n    \n    # Iterate through questions 1 to 18\n    for t in range(1, 19):\n        # Determine the level group for the current question\n        if t <= 3:\n            grp = '0-4'\n        elif t <= 13:\n            grp = '5-12'\n        elif t <= 22:\n            grp = '13-22'\n            \n        # Get the train and validation data for the current question and level group\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q == t].set_index('session').loc[train_users]\n        \n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets.q == t].set_index('session').loc[valid_users]\n        \n        # Train an XGBoost model for the current question and level group\n        clf = XGBClassifier(**xgb_params)\n        clf.fit(\n            train_x[FEATURES].astype('float32'), train_y['correct'],\n            eval_set=[(valid_x[FEATURES].astype('float32'), valid_y['correct'])],\n            verbose=0\n        )\n        print(f'{t}({clf.best_ntree_limit}), ', end='')\n        \n        # Save the trained model and predict on the validation data to collect out-of-fold predictions\n        models[f'{grp}_{t}'] = clf\n        oof.loc[valid_users, t - 1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:, 1]\n        \n    print()\n","metadata":{"papermill":{"duration":110.50811,"end_time":"2023-04-25T11:19:30.238796","exception":false,"start_time":"2023-04-25T11:17:39.730686","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:06:39.658958Z","iopub.execute_input":"2023-05-05T18:06:39.659700Z","iopub.status.idle":"2023-05-05T18:11:08.914380Z","shell.execute_reply.started":"2023-05-05T18:06:39.659668Z","shell.execute_reply":"2023-05-05T18:11:08.913437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compute CV Score\nWe need to convert prediction probabilities into `1s` and `0s`. The competition metric is F1 Score which is the harmonic mean of precision and recall. Let's find the optimal threshold for `p > threshold` when to predict `1` and when to predict `0` to maximize F1 Score.","metadata":{"papermill":{"duration":0.016276,"end_time":"2023-04-25T11:19:30.274909","exception":false,"start_time":"2023-04-25T11:19:30.258633","status":"completed"},"tags":[]}},{"cell_type":"code","source":"true = oof.copy()\n\n# Iterate through the 18 questions\nfor k in range(18):\n    # Get the true labels for the current question\n    tmp = targets.loc[targets.q == k + 1].set_index('session').loc[ALL_USERS]\n    \n    # Update the corresponding column in the 'true' DataFrame with the true labels\n    true[k] = tmp.correct.values\n","metadata":{"papermill":{"duration":0.13751,"end_time":"2023-04-25T11:19:30.429065","exception":false,"start_time":"2023-04-25T11:19:30.291555","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:11:08.918301Z","iopub.execute_input":"2023-05-05T18:11:08.919060Z","iopub.status.idle":"2023-05-05T18:11:09.077842Z","shell.execute_reply.started":"2023-05-05T18:11:08.919003Z","shell.execute_reply":"2023-05-05T18:11:09.076656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = []  # List to store F1 scores for each threshold\nthresholds = []  # List to store the threshold values\nbest_score = 0  # Variable to store the best F1 score\nbest_threshold = 0  # Variable to store the best threshold value\n\n# Iterate through a range of threshold values\nfor threshold in np.arange(0.4, 0.81, 0.01):\n    print(f'{threshold:.02f}, ', end='')\n    \n    # Convert probabilities to binary predictions using the current threshold\n    preds = (oof.values.reshape((-1)) > threshold).astype('int')\n    \n    # Calculate the F1 score for the predictions\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')\n    \n    # Store the F1 score and threshold in the respective lists\n    scores.append(m)\n    thresholds.append(threshold)\n    \n    # Check if the current score is better than the previous best score\n    if m > best_score:\n        best_score = m\n        best_threshold = threshold\n","metadata":{"papermill":{"duration":5.933285,"end_time":"2023-04-25T11:19:36.379488","exception":false,"start_time":"2023-04-25T11:19:30.446203","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:11:09.079555Z","iopub.execute_input":"2023-05-05T18:11:09.079891Z","iopub.status.idle":"2023-05-05T18:11:15.733599Z","shell.execute_reply.started":"2023-05-05T18:11:09.079862Z","shell.execute_reply":"2023-05-05T18:11:15.732214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Plotting threshold vs. F1 score\nplt.figure(figsize=(20, 5))\nplt.plot(thresholds, scores, '-o', color='blue')  # Plotting threshold vs. F1 score as a line plot\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)  # Marking the best threshold and F1 score with a scatter point\nplt.xlabel('Threshold', size=14)\nplt.ylabel('Validation F1 Score', size=14)\nplt.title(f'Threshold vs. F1 Score with Best F1 Score = {best_score:.3f} at Best Threshold = {best_threshold:.3f}', size=18)\nplt.show()\n","metadata":{"papermill":{"duration":0.333893,"end_time":"2023-04-25T11:19:36.731681","exception":false,"start_time":"2023-04-25T11:19:36.397788","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:11:15.735267Z","iopub.execute_input":"2023-05-05T18:11:15.735901Z","iopub.status.idle":"2023-05-05T18:11:16.088665Z","shell.execute_reply.started":"2023-05-05T18:11:15.735865Z","shell.execute_reply":"2023-05-05T18:11:16.087717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting histogram of F1 scores\nplt.figure(figsize=(10, 5))\nplt.hist(scores, bins=20, color='blue', alpha=0.5)\nplt.axvline(x=best_score, color='red', linestyle='--', label='Best F1 Score')\nplt.xlabel('F1 Score', size=14)\nplt.ylabel('Frequency', size=14)\nplt.title('Distribution of F1 Scores', size=18)\nplt.legend()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-05T18:11:16.089698Z","iopub.execute_input":"2023-05-05T18:11:16.089992Z","iopub.status.idle":"2023-05-05T18:11:16.341282Z","shell.execute_reply.started":"2023-05-05T18:11:16.089965Z","shell.execute_reply":"2023-05-05T18:11:16.340517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## F1-score Macro == Competition metric\n","metadata":{"papermill":{"duration":0.018712,"end_time":"2023-04-25T11:19:36.769375","exception":false,"start_time":"2023-04-25T11:19:36.750663","status":"completed"},"tags":[]}},{"cell_type":"code","source":"for k in range(18):\n    # COMPUTE F1 SCORE PER QUESTION\n    predicted_labels = (oof[k].values > best_threshold).astype('int')\n    f1 = f1_score(true[k].values, predicted_labels, average='macro')\n    print(f'Q{k}: F1 = {f1:.3f}')\n    \n# COMPUTE F1 SCORE OVERALL\npredicted_labels_all = (oof.values.reshape((-1)) > best_threshold).astype('int')\nf1_overall = f1_score(true.values.reshape((-1)), predicted_labels_all, average='macro')\nprint('==> Overall F1 =', f1_overall)\n","metadata":{"papermill":{"duration":0.326172,"end_time":"2023-04-25T11:19:37.114380","exception":false,"start_time":"2023-04-25T11:19:36.788208","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:11:16.342609Z","iopub.execute_input":"2023-05-05T18:11:16.343079Z","iopub.status.idle":"2023-05-05T18:11:16.679280Z","shell.execute_reply.started":"2023-05-05T18:11:16.343049Z","shell.execute_reply":"2023-05-05T18:11:16.678524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Infer Test Data","metadata":{"papermill":{"duration":0.017201,"end_time":"2023-04-25T11:19:37.150914","exception":false,"start_time":"2023-04-25T11:19:37.133713","status":"completed"},"tags":[]}},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/test.csv', dtype=dtypes)","metadata":{"execution":{"iopub.status.busy":"2023-05-05T18:11:16.680589Z","iopub.execute_input":"2023-05-05T18:11:16.681064Z","iopub.status.idle":"2023-05-05T18:11:16.725434Z","shell.execute_reply.started":"2023-05-05T18:11:16.681036Z","shell.execute_reply":"2023-05-05T18:11:16.724569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Engineer features for the test data\ntest_df = feature_engineer(test)\n\n# Initialize an empty DataFrame to store test predictions\ntest_preds = pd.DataFrame(data=np.zeros((len(test_df.index.unique()), 18)), index=test_df.index.unique())\n\n# Iterate through questions 1 to 18\nfor t in range(1, 19):\n    # Determine the level group for the current question\n    if t <= 3:\n        grp = '0-4'\n    elif t <= 13:\n        grp = '5-12'\n    elif t <= 22:\n        grp = '13-22'\n    \n    # Get the test data for the current question and level group\n    test_x = test_df.loc[test_df.level_group == grp]","metadata":{"execution":{"iopub.status.busy":"2023-05-05T18:11:24.620933Z","iopub.execute_input":"2023-05-05T18:11:24.621380Z","iopub.status.idle":"2023-05-05T18:11:24.703651Z","shell.execute_reply.started":"2023-05-05T18:11:24.621346Z","shell.execute_reply":"2023-05-05T18:11:24.702313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = models[f'{grp}_{t}']\n\n# Predict on the test data and store the predictions in the 'test_preds' DataFrame\ntest_preds.loc[test_x.index, t - 1] = clf.predict_proba(test_x[FEATURES].astype('float32'))[:, 1]\n","metadata":{"execution":{"iopub.status.busy":"2023-05-05T18:11:27.625219Z","iopub.execute_input":"2023-05-05T18:11:27.625663Z","iopub.status.idle":"2023-05-05T18:11:27.641857Z","shell.execute_reply.started":"2023-05-05T18:11:27.625629Z","shell.execute_reply":"2023-05-05T18:11:27.641018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds_binary = (test_preds.values > best_threshold).astype('int')","metadata":{"execution":{"iopub.status.busy":"2023-05-05T18:11:16.830604Z","iopub.execute_input":"2023-05-05T18:11:16.834880Z","iopub.status.idle":"2023-05-05T18:11:16.839640Z","shell.execute_reply.started":"2023-05-05T18:11:16.834846Z","shell.execute_reply":"2023-05-05T18:11:16.838512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(test_preds_binary, columns=[f'Q{i + 1}' for i in range(18)], index=test_preds.index)\nsubmission.index.name = 'session_id'","metadata":{"execution":{"iopub.status.busy":"2023-05-05T18:11:16.841438Z","iopub.execute_input":"2023-05-05T18:11:16.841745Z","iopub.status.idle":"2023-05-05T18:11:16.850531Z","shell.execute_reply.started":"2023-05-05T18:11:16.841719Z","shell.execute_reply":"2023-05-05T18:11:16.849513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('/kaggle/working/xgb_submission.csv', index=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-05T18:11:16.855530Z","iopub.execute_input":"2023-05-05T18:11:16.855842Z","iopub.status.idle":"2023-05-05T18:11:16.869161Z","shell.execute_reply.started":"2023-05-05T18:11:16.855815Z","shell.execute_reply":"2023-05-05T18:11:16.867811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# IMPORT KAGGLE API\n#import jo_wilder\n#env = jo_wilder.make_env()\n#iter_test = env.iter_test()\n\n# CLEAR MEMORY\n#import gc\n#del targets, df, oof, true\n# _ = gc.collect()","metadata":{"papermill":{"duration":0.204506,"end_time":"2023-04-25T11:19:37.373185","exception":false,"start_time":"2023-04-25T11:19:37.168679","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:11:16.870804Z","iopub.execute_input":"2023-05-05T18:11:16.871584Z","iopub.status.idle":"2023-05-05T18:11:16.881998Z","shell.execute_reply.started":"2023-05-05T18:11:16.871545Z","shell.execute_reply":"2023-05-05T18:11:16.880385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\n#for (test, sample_submission) in iter_test:\n    \n    # FEATURE ENGINEER TEST DATA\n   # df = feature_engineer(test)\n    \n    # INFER TEST DATA\n  #  grp = test.level_group.values[0]\n   # a, b = limits[grp]\n   # mask = sample_submission.session_id.str.contains(f'q')\n    \n    #for t in range(a, b):\n     #   p = clf.predict_proba(df[FEATURES].astype('float32')).iloc[0, 1]\n      #  sample_submission.loc[mask & sample_submission.session_id.str.contains(f'q{t}'), 'correct'] = int(p > best_threshold)\n    \n   # env.predict(sample_submission)\n","metadata":{"papermill":{"duration":0.806962,"end_time":"2023-04-25T11:19:38.198370","exception":false,"start_time":"2023-04-25T11:19:37.391408","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:11:16.883769Z","iopub.execute_input":"2023-05-05T18:11:16.884121Z","iopub.status.idle":"2023-05-05T18:11:16.893852Z","shell.execute_reply.started":"2023-05-05T18:11:16.884090Z","shell.execute_reply":"2023-05-05T18:11:16.892500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA submission.csv","metadata":{"papermill":{"duration":0.017434,"end_time":"2023-04-25T11:19:38.235758","exception":false,"start_time":"2023-04-25T11:19:38.218324","status":"completed"},"tags":[]}},{"cell_type":"code","source":"#df = pd.read_csv('submission.csv')\n#print( df.shape )\n#df.head()","metadata":{"papermill":{"duration":0.038489,"end_time":"2023-04-25T11:19:38.292714","exception":false,"start_time":"2023-04-25T11:19:38.254225","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:11:16.895485Z","iopub.execute_input":"2023-05-05T18:11:16.895838Z","iopub.status.idle":"2023-05-05T18:11:16.905853Z","shell.execute_reply.started":"2023-05-05T18:11:16.895799Z","shell.execute_reply":"2023-05-05T18:11:16.904531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(df.correct.mean())","metadata":{"papermill":{"duration":0.030069,"end_time":"2023-04-25T11:19:38.342207","exception":false,"start_time":"2023-04-25T11:19:38.312138","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-05T18:11:16.907273Z","iopub.execute_input":"2023-05-05T18:11:16.908368Z","iopub.status.idle":"2023-05-05T18:11:16.922729Z","shell.execute_reply.started":"2023-05-05T18:11:16.908336Z","shell.execute_reply":"2023-05-05T18:11:16.921428Z"},"trusted":true},"execution_count":null,"outputs":[]}]}