{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Predict Student Performance from Game Play\n\n## Is it reasonable to treat session information as sequential data?\n\nThe intuition of this notebook is to encode all the rows in a session as sequential data, and then use a Recurrent Neural Networks to predict whether the user for this particular session will answer this question correctly.\n\nIf the output is potential, this could tremendously reduce the effort of future engineering, or can become a reliable support for encoding useful features, which can combine with features from statistical analysis to produce a better classifier.","metadata":{}},{"cell_type":"markdown","source":"# Read the DataFrame","metadata":{}},{"cell_type":"code","source":"# Import required libraries\nimport pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn","metadata":{"execution":{"iopub.status.busy":"2023-04-24T01:57:38.625063Z","iopub.execute_input":"2023-04-24T01:57:38.626079Z","iopub.status.idle":"2023-04-24T01:57:43.432844Z","shell.execute_reply.started":"2023-04-24T01:57:38.626034Z","shell.execute_reply":"2023-04-24T01:57:43.431755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the dataset\ndtypes = {\n    'elapsed_time': np.int32,\n    'event_name': 'category', \n    'name': 'category',\n    'level': 'category',\n    'room_coor_x': np.float32,\n    'room_coor_y': np.float32,\n    'screen_coor_x': np.float32,\n    'screen_coor_y': np.float32,\n    'hover_duration': np.float32,\n    'text': 'category',\n    'fqid': 'category',\n    'room_fqid': 'category',\n    'text_fqid': 'category',\n    'fullscreen': 'category',\n    'hq': 'category',\n    'music': 'category',\n    'level_group': 'category'\n}\n\ndf = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\n\n# Print the first 5 rows\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-24T01:57:43.435124Z","iopub.execute_input":"2023-04-24T01:57:43.435974Z","iopub.status.idle":"2023-04-24T01:59:40.569663Z","shell.execute_reply.started":"2023-04-24T01:57:43.435927Z","shell.execute_reply":"2023-04-24T01:59:40.568354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-24T01:59:40.571783Z","iopub.execute_input":"2023-04-24T01:59:40.572223Z","iopub.status.idle":"2023-04-24T01:59:40.580760Z","shell.execute_reply.started":"2023-04-24T01:59:40.572179Z","shell.execute_reply":"2023-04-24T01:59:40.579578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Intuition of using LSTM\n\nIn specific, taking a `session_id`...","metadata":{}},{"cell_type":"code","source":"session_1_df = df[df['session_id'] == 20090312431273200]\nsession_1_df","metadata":{"execution":{"iopub.status.busy":"2023-04-24T01:59:40.584553Z","iopub.execute_input":"2023-04-24T01:59:40.585773Z","iopub.status.idle":"2023-04-24T01:59:40.720817Z","shell.execute_reply.started":"2023-04-24T01:59:40.585733Z","shell.execute_reply":"2023-04-24T01:59:40.719606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"...which contains 881 actions recorded. We consider it as a report document of 881 words to process and see whether the prediction made from this document is reliable.\n\n**The strategy for encoding a record into numeric format:**\n\n- Employed columns: `event_name`, `name`, `level`, `room_coor_x`, `room_coor_y`, `screen_coor_x`, `screen_coor_y`, `hover_duration`.\n- Set all the null values to 0 since there exists a identification, `event_name`, that shows the reason why these values are zeros.\n- Encode all categorical columns (using one-hot encoding).","metadata":{}},{"cell_type":"code","source":"df.set_index(['session_id', 'index'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-24T01:59:40.722724Z","iopub.execute_input":"2023-04-24T01:59:40.723450Z","iopub.status.idle":"2023-04-24T01:59:42.160667Z","shell.execute_reply.started":"2023-04-24T01:59:40.723397Z","shell.execute_reply":"2023-04-24T01:59:42.159531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df[['event_name', 'name', 'level', 'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'hover_duration']]\nfor col in ['room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'hover_duration']:\n    # Scaling the coordinates and durations\n    df[col] = (df[col] - df[col].min()) / (df[col].max() - df[col].min())\n    df[col] = df[col].fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-04-24T01:59:42.162562Z","iopub.execute_input":"2023-04-24T01:59:42.163002Z","iopub.status.idle":"2023-04-24T01:59:45.698000Z","shell.execute_reply.started":"2023-04-24T01:59:42.162952Z","shell.execute_reply":"2023-04-24T01:59:45.696948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# One-Hot Encoding and Aggregation","metadata":{}},{"cell_type":"markdown","source":"We are using a custom `GetDummies` class for one-hot encoding for 2 reasons:\n\n1. `OneHotEncoder` runs excessively slower than `pd.get_dummies` in encoding large data.\n2. `pd.get_dummies` transformations may encounter inconsistent amount of columns in transformed data if 2 datasets contain different number of unique categorical values.","metadata":{}},{"cell_type":"code","source":"import sklearn\n\n\nclass GetDummies(sklearn.base.TransformerMixin):\n    \"\"\"Fast one-hot-encoder that makes use of pandas.get_dummies() safely\n    on train/test splits.\n    \"\"\"\n    def __init__(self, dtypes=None):\n        self.input_columns = None\n        self.final_columns = None\n        if dtypes is None:\n            dtypes = [object, 'category']\n        self.dtypes = dtypes\n\n    def fit(self, X, y=None, **kwargs):\n        self.input_columns = list(X.select_dtypes(self.dtypes).columns)\n        X = pd.get_dummies(X, columns=self.input_columns)\n        self.final_columns = X.columns\n        return self\n        \n    def transform(self, X, y=None, **kwargs):\n        X = pd.get_dummies(X, columns=self.input_columns)\n        X_columns = X.columns\n        # if columns in X had values not in the data set used during\n        # fit add them and set to 0\n        missing = set(self.final_columns) - set(X_columns)\n        for c in missing:\n            X[c] = 0\n        # remove any new columns that may have resulted from values in\n        # X that were not in the data set when fit\n        return X[self.final_columns]\n    \n    def get_feature_names(self):\n        return tuple(self.final_columns)","metadata":{"execution":{"iopub.status.busy":"2023-04-24T01:59:45.699495Z","iopub.execute_input":"2023-04-24T01:59:45.699855Z","iopub.status.idle":"2023-04-24T01:59:46.573802Z","shell.execute_reply.started":"2023-04-24T01:59:45.699816Z","shell.execute_reply":"2023-04-24T01:59:46.572616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_dummies = GetDummies()\ndf = get_dummies.fit_transform(df)\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-24T01:59:46.575437Z","iopub.execute_input":"2023-04-24T01:59:46.576846Z","iopub.status.idle":"2023-04-24T02:00:04.578428Z","shell.execute_reply.started":"2023-04-24T01:59:46.576803Z","shell.execute_reply":"2023-04-24T02:00:04.577327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Aggregate data in each session into a 2D numpy array\ngrouped_data = df.groupby('session_id').apply(lambda x: np.array(x))\ngrouped_data","metadata":{"execution":{"iopub.status.busy":"2023-04-24T02:00:04.580239Z","iopub.execute_input":"2023-04-24T02:00:04.581015Z","iopub.status.idle":"2023-04-24T02:00:16.130590Z","shell.execute_reply.started":"2023-04-24T02:00:04.580967Z","shell.execute_reply":"2023-04-24T02:00:16.129474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Convert to PyTorch Dataloader","metadata":{}},{"cell_type":"code","source":"from torch.utils.data import Dataset, DataLoader\n\nclass MyDataset(Dataset):\n    def __init__(self, data):\n        self.data = data\n        \n    def __len__(self):\n        return len(self.data)\n    \n    def __getitem__(self, idx):\n        # Get the numpy array at the given index\n        return torch.from_numpy(self.data[idx]).float()","metadata":{"execution":{"iopub.status.busy":"2023-04-24T02:00:16.135878Z","iopub.execute_input":"2023-04-24T02:00:16.136298Z","iopub.status.idle":"2023-04-24T02:00:16.144522Z","shell.execute_reply.started":"2023-04-24T02:00:16.136258Z","shell.execute_reply":"2023-04-24T02:00:16.143140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since the length of each sequence are different, batch transformation is required. For each batch, we will take the session with highest number of actions recorded, let say A, and perform **padding** to other elements by adding zeros to have the shapes of those elements are identical to A.","metadata":{}},{"cell_type":"code","source":"def collate_fn_padd(batch):\n    \"\"\"\n    Padds batch of variable length\n\n    Note: it converts things ToTensor manually here since the ToTensor transform\n    assume it takes in images rather than arbitrary tensors.\n    \"\"\"\n    ## Get sequence lengths\n    lengths = [t.shape[0] for t in batch]\n    try:\n        n_features = batch[0].shape[1]\n    except:\n        n_features = 1\n    max_length = max(lengths)\n    if max_length == 0:\n        max_length += 1\n    batch_size = len(lengths)\n\n    padded_tensor = torch.zeros(batch_size, max_length, n_features, dtype=torch.float32)\n    for i, val in enumerate(batch):\n        l = lengths[i]\n        if n_features == 1:\n            padded_tensor[i, :l] = val.reshape(-1, 1)\n        else:\n            padded_tensor[i, :l] = val\n    \n    return padded_tensor","metadata":{"execution":{"iopub.status.busy":"2023-04-24T02:00:16.146398Z","iopub.execute_input":"2023-04-24T02:00:16.147127Z","iopub.status.idle":"2023-04-24T02:00:16.157301Z","shell.execute_reply.started":"2023-04-24T02:00:16.147085Z","shell.execute_reply":"2023-04-24T02:00:16.156077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create an instance of the custom dataset\ndataset = MyDataset(grouped_data.values)\n\n# Create a PyTorch DataLoader\ndataloader = DataLoader(dataset, batch_size=32, shuffle=True, collate_fn=collate_fn_padd)","metadata":{"execution":{"iopub.status.busy":"2023-04-24T02:00:16.158930Z","iopub.execute_input":"2023-04-24T02:00:16.159306Z","iopub.status.idle":"2023-04-24T02:00:16.174504Z","shell.execute_reply.started":"2023-04-24T02:00:16.159269Z","shell.execute_reply":"2023-04-24T02:00:16.173544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Processing the labels\n\nNow we collect and process the labels...","metadata":{}},{"cell_type":"code","source":"# Collect and process the label\nlabel_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\n\n# session_id format: <session>_q<idx> ==> session = <session>\nlabel_df['session'] = label_df.session_id.apply(lambda x: int(x.split('_')[0]) )\n\n# session_id format: <session>_q<idx> ==> question_idx = <idx>\nlabel_df['question_idx'] = label_df.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nlabel_df.drop(\"session_id\", axis=1, inplace=True)\n\n# Pivot the table so that the columns are question indices (from 1 to 18),\n# indices are session_ids, and value is 0-1 (correct or not)\npivoted_questions = label_df.pivot(columns='question_idx', values='correct', index='session')\n\n# We have a total_score column here just for analysis if needed\npivoted_questions['total_score'] = pivoted_questions.iloc[:, 0:18].sum(axis=1)\n\n# Rename the columns\npivoted_questions.columns = [f'q_{i}' for i in range(1, 19)] + ['total_score']\npivoted_questions","metadata":{"execution":{"iopub.status.busy":"2023-04-24T02:00:16.175902Z","iopub.execute_input":"2023-04-24T02:00:16.176621Z","iopub.status.idle":"2023-04-24T02:00:17.326679Z","shell.execute_reply.started":"2023-04-24T02:00:16.176581Z","shell.execute_reply":"2023-04-24T02:00:17.325485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LSTM model and Training","metadata":{}},{"cell_type":"code","source":"# Define the LSTM model\nclass StackedLSTM(nn.Module):\n    def __init__(self, n_layers, n_hidden, n_features, n_embeddings):\n        super(StackedLSTM, self).__init__()\n        self.embedding = nn.Linear(n_features, n_embeddings)\n        self.lstm = nn.LSTM(n_embeddings, n_hidden, n_layers, batch_first=True)\n        self.linear = nn.Linear(n_hidden, 18)\n        \n    def forward(self, x):\n        # Pass the input through the Embedding layer\n        embed_out = self.embedding(x)\n\n        # Pass the input through the LSTM layers\n        lstm_out, _ = self.lstm(embed_out)\n\n        # Get only the last output of the LSTM layer\n        out = lstm_out[:, -1, :]\n        \n        # Flatten the LSTM output and pass it through the linear layer\n        out = self.linear(out)\n        \n        # Apply sigmoid activation function to the output\n        out = torch.sigmoid(out)\n        \n        return out\n\n# Create an instance of the model\nn_layers = 3  # Number of LSTM layers\nn_hidden = 16  # Number of LSTM units\nn_embeddings = 16 # Number of dimension in embedding layer\nn_features = 45  # Number of features in each sequence\n\ndevice = 'cuda' if torch.cuda.is_available() else 'cpu'\nmodel = StackedLSTM(n_layers, n_hidden, n_features, n_embeddings).to(device)","metadata":{"execution":{"iopub.status.busy":"2023-04-24T02:00:17.328506Z","iopub.execute_input":"2023-04-24T02:00:17.329579Z","iopub.status.idle":"2023-04-24T02:00:23.944103Z","shell.execute_reply.started":"2023-04-24T02:00:17.329529Z","shell.execute_reply":"2023-04-24T02:00:23.943075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\n\n# Define number of output labels (number of questions)\nn_out = 18\n\n# Define the batch size\nbatch_size = 32\n\n# Define the number of epochs\nn_epochs = 3\n\n# Data size\nn_samples = len(grouped_data)\n\n# Define the loss function and optimizer\ncriterion = nn.BCELoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=0.01)\n\n# Train the model\nmodel.train()\nfor epoch in range(n_epochs):\n    for i, sample in tqdm(enumerate(dataloader)):\n        model.zero_grad()\n        \n        # Get label\n        labels = torch.from_numpy(pivoted_questions.iloc[i*batch_size:(i+1)*batch_size, :18].values).float()\n        \n        sample = sample.to(device)\n        labels = labels.to(device)\n        \n        # Forward pass\n        outputs = model(sample)\n\n        # Compute the loss\n        loss = criterion(outputs, labels)\n        \n        # Backward pass and optimization\n        loss.backward()\n        optimizer.step()\n\n        sample = sample.to('cpu')\n        labels = labels.to('cpu')\n        \n    # Print the loss after every epoch\n    print(f'Epoch {epoch+1}/{n_epochs}, Loss: {loss.item():.4f}')","metadata":{"execution":{"iopub.status.busy":"2023-04-24T02:00:23.945437Z","iopub.execute_input":"2023-04-24T02:00:23.946082Z","iopub.status.idle":"2023-04-24T02:05:56.708277Z","shell.execute_reply.started":"2023-04-24T02:00:23.946043Z","shell.execute_reply":"2023-04-24T02:05:56.707144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model\npred_list = []\ntrue_list = []\n\nmodel.eval()\nfor i, sample in tqdm(enumerate(dataloader)):\n    model.zero_grad()\n        \n    # Get label\n    labels = torch.from_numpy(pivoted_questions.iloc[i*batch_size:(i+1)*batch_size, :18].values).float()\n    \n    sample = sample.to(device)\n    labels = labels.to(device)\n\n    # Forward pass\n    outputs = model(sample)\n\n    sample = sample.to('cpu')\n    labels = labels.to('cpu')\n\n    pred_list.append(outputs.data.cpu().numpy())\n    true_list.append(labels.data.cpu().numpy())","metadata":{"execution":{"iopub.status.busy":"2023-04-24T02:05:56.710198Z","iopub.execute_input":"2023-04-24T02:05:56.711071Z","iopub.status.idle":"2023-04-24T02:06:46.893920Z","shell.execute_reply.started":"2023-04-24T02:05:56.711025Z","shell.execute_reply":"2023-04-24T02:06:46.892699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Flatten the result dict\ntest_pred_flattened = np.concatenate(pred_list).ravel()\ntest_true_flattened = np.concatenate(true_list).ravel()","metadata":{"execution":{"iopub.status.busy":"2023-04-24T02:06:46.895958Z","iopub.execute_input":"2023-04-24T02:06:46.896806Z","iopub.status.idle":"2023-04-24T02:06:46.905144Z","shell.execute_reply.started":"2023-04-24T02:06:46.896759Z","shell.execute_reply":"2023-04-24T02:06:46.903739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluation","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, precision_score, recall_score\n\nprint(accuracy_score(test_true_flattened, np.round(test_pred_flattened)))\nprint(precision_score(test_true_flattened, np.round(test_pred_flattened)))\nprint(recall_score(test_true_flattened, np.round(test_pred_flattened)))","metadata":{"execution":{"iopub.status.busy":"2023-04-24T02:06:46.907187Z","iopub.execute_input":"2023-04-24T02:06:46.907636Z","iopub.status.idle":"2023-04-24T02:06:47.538297Z","shell.execute_reply.started":"2023-04-24T02:06:46.907588Z","shell.execute_reply":"2023-04-24T02:06:47.537056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Generally, black-box RNN can somehow infer the predictions based on action sequence recorded from the user (with 73.2% accuracy on training set). However, there is something I am still wondering is that the loss did not converge (still at a rate of 0.518x), I hope to get any comments for improvement or spotting whether I have made a mistake in this notebook. Thanks for reading!**","metadata":{}},{"cell_type":"code","source":"# For test set\n\n# Remove the training set to save RAM\ndel(df)\ndel(grouped_data)\ndel(dataloader)\ndel(dataset)","metadata":{"execution":{"iopub.status.busy":"2023-04-24T02:06:47.539757Z","iopub.execute_input":"2023-04-24T02:06:47.540775Z","iopub.status.idle":"2023-04-24T02:06:47.599855Z","shell.execute_reply.started":"2023-04-24T02:06:47.540724Z","shell.execute_reply":"2023-04-24T02:06:47.598942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Workflow on test set","metadata":{}},{"cell_type":"code","source":"# PROCESSING THE TEST DATASET\n\n# Reading the dataset\ntest_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/test.csv', dtype=dtypes)\n\n# Preprocess the data with feature selection\ntest_df.set_index(['session_id', 'index'], inplace=True)\ntest_df = test_df[['event_name', 'name', 'level', 'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'hover_duration']]\nfor col in ['room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'hover_duration']:\n    # Scaling the coordinates and durations\n    test_df[col] = (test_df[col] - test_df[col].min()) / (test_df[col].max() - test_df[col].min())\n    test_df[col] = test_df[col].fillna(0)\n\n# Perform one-hot encoding\ntest_df = get_dummies.transform(test_df)\ngrouped_data = test_df.groupby('session_id').apply(lambda x: np.array(x))\n\ndataset = MyDataset(grouped_data.values)\ndataloader = DataLoader(dataset, batch_size=3, shuffle=True, collate_fn=collate_fn_padd)\n\n# Make predictions\npred_list = []\n\nmodel.eval()\nfor i, sample in tqdm(enumerate(dataloader)):\n    model.zero_grad()\n    sample = sample.to(device)\n    # Forward pass\n    outputs = model(sample)\n    sample = sample.to('cpu')\n    pred_list.append(outputs.data.cpu().numpy())\n    \npred_flattened = np.concatenate(pred_list).ravel()\nsession_ids = test_df.index.get_level_values('session_id').unique().tolist()\n\nfrom functools import reduce\nsession_ids = reduce(lambda x, y: x + [f'{y}_q{i}' for i in range(1, 19)], session_ids, [])\n\ntest_result = pd.DataFrame({\n    'session_id': session_ids,\n    'correct': (pred_flattened > 0.6).astype('int')\n})\ntest_result.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-24T02:06:47.601578Z","iopub.execute_input":"2023-04-24T02:06:47.602421Z","iopub.status.idle":"2023-04-24T02:06:47.720227Z","shell.execute_reply.started":"2023-04-24T02:06:47.602382Z","shell.execute_reply":"2023-04-24T02:06:47.719087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"import jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-04-24T02:06:47.722138Z","iopub.execute_input":"2023-04-24T02:06:47.722545Z","iopub.status.idle":"2023-04-24T02:06:47.754943Z","shell.execute_reply.started":"2023-04-24T02:06:47.722509Z","shell.execute_reply":"2023-04-24T02:06:47.753790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for (test, sample_submission) in iter_test:\n    \n    test_df = test\n    test_df.set_index(['session_id', 'index'], inplace=True)\n\n    test_df = test_df[['event_name', 'name', 'level', 'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'hover_duration']]\n    for col in ['room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'hover_duration']:\n        # Scaling the coordinates and durations\n        test_df[col] = (test_df[col] - test_df[col].min()) / (test_df[col].max() - test_df[col].min())\n        test_df[col] = test_df[col].fillna(0)\n\n    test_df = get_dummies.transform(test_df)\n    grouped_data = test_df.groupby('session_id').apply(lambda x: np.array(x))\n\n    dataset = MyDataset(grouped_data.values)\n    dataloader = DataLoader(dataset, batch_size=3, shuffle=True, collate_fn=collate_fn_padd)\n\n    # Make predictions\n    pred_list = []\n\n    model.eval()\n    for i, sample in tqdm(enumerate(dataloader)):\n        model.zero_grad()\n        sample = sample.to(device)\n        # Forward pass\n        outputs = model(sample)\n        sample = sample.to('cpu')\n        pred_list.append(outputs.data.cpu().numpy())\n\n    pred_flattened = np.concatenate(pred_list).ravel()\n    session_ids = test_df.index.get_level_values('session_id').unique().tolist()\n\n    from functools import reduce\n    session_ids = reduce(lambda x, y: x + [f'{y}_q{i}' for i in range(1, 19)], session_ids, [])\n\n    test_result = pd.DataFrame({\n        'session_id': session_ids,\n        'correct': (pred_flattened > 0.6).astype('int')\n    })\n    test_result.head()\n    \n    env.predict(test_result)","metadata":{"execution":{"iopub.status.busy":"2023-04-24T02:06:47.756463Z","iopub.execute_input":"2023-04-24T02:06:47.756811Z","iopub.status.idle":"2023-04-24T02:06:48.128345Z","shell.execute_reply.started":"2023-04-24T02:06:47.756776Z","shell.execute_reply":"2023-04-24T02:06:48.127192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-04-24T02:20:01.882311Z","iopub.execute_input":"2023-04-24T02:20:01.883092Z","iopub.status.idle":"2023-04-24T02:20:03.096773Z","shell.execute_reply.started":"2023-04-24T02:20:01.883060Z","shell.execute_reply":"2023-04-24T02:20:03.095171Z"},"trusted":true},"execution_count":null,"outputs":[]}]}