{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-24T11:00:26.463522Z","iopub.execute_input":"2023-02-24T11:00:26.464538Z","iopub.status.idle":"2023-02-24T11:00:26.507362Z","shell.execute_reply.started":"2023-02-24T11:00:26.464502Z","shell.execute_reply":"2023-02-24T11:00:26.506437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1) 데이터 분석의 목표 : 산타토익을 사용한 학습자의 문제 풀이 기록을 바탕으로 새로운 문제에 대한 정답률을 예측하고자 함\n2) 활용 방법 및 이유 : DKT를 활용하고자 함\n - KT 방법론 중에서 상대적으로 최신의 알고리즘을 활용해보고자 함.\n - 뤼이드 데이터의 sequential한 특성을 살리고자 함.\n3) 결과 해석 방법 : AUG 지표를 확인함. 다만, 이 알고리즘 자체를 테스트하는 것이고 비교군이 없기 때문에 큰 의미는 없음.\n4) 사용 데이터\n - X : 일정 길이(여기서는 100)의 문제 시퀀스\n - Y : 해당 문제의 답","metadata":{}},{"cell_type":"code","source":"ex_test = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/example_test.csv')\nex_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T11:00:29.453262Z","iopub.execute_input":"2023-02-24T11:00:29.454041Z","iopub.status.idle":"2023-02-24T11:00:29.496662Z","shell.execute_reply.started":"2023-02-24T11:00:29.453985Z","shell.execute_reply":"2023-02-24T11:00:29.495406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"questions = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/questions.csv')\nquestions.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T11:00:31.671935Z","iopub.execute_input":"2023-02-24T11:00:31.672746Z","iopub.status.idle":"2023-02-24T11:00:31.716218Z","shell.execute_reply.started":"2023-02-24T11:00:31.672706Z","shell.execute_reply":"2023-02-24T11:00:31.71503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lectures = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/lectures.csv')\nlectures.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T11:00:33.415464Z","iopub.execute_input":"2023-02-24T11:00:33.416611Z","iopub.status.idle":"2023-02-24T11:00:33.438382Z","shell.execute_reply.started":"2023-02-24T11:00:33.416557Z","shell.execute_reply":"2023-02-24T11:00:33.437523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/train.csv')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T11:00:35.728278Z","iopub.execute_input":"2023-02-24T11:00:35.72904Z","iopub.status.idle":"2023-02-24T11:03:51.995864Z","shell.execute_reply.started":"2023-02-24T11:00:35.729003Z","shell.execute_reply":"2023-02-24T11:03:51.99412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['user_answer'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T11:03:51.999267Z","iopub.execute_input":"2023-02-24T11:03:51.999847Z","iopub.status.idle":"2023-02-24T11:03:52.582283Z","shell.execute_reply.started":"2023-02-24T11:03:51.999801Z","shell.execute_reply":"2023-02-24T11:03:52.580924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.loc[train['user_answer'] != -1,:]\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T11:03:52.583867Z","iopub.execute_input":"2023-02-24T11:03:52.585086Z","iopub.status.idle":"2023-02-24T11:04:03.204214Z","shell.execute_reply.started":"2023-02-24T11:03:52.585027Z","shell.execute_reply":"2023-02-24T11:04:03.203012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['user_answer'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T11:04:03.208653Z","iopub.execute_input":"2023-02-24T11:04:03.209012Z","iopub.status.idle":"2023-02-24T11:04:03.773275Z","shell.execute_reply.started":"2023-02-24T11:04:03.20898Z","shell.execute_reply":"2023-02-24T11:04:03.772483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_columns = ['user_id','content_id', 'task_container_id', 'answered_correctly']","metadata":{"execution":{"iopub.status.busy":"2023-02-24T11:04:03.774737Z","iopub.execute_input":"2023-02-24T11:04:03.775279Z","iopub.status.idle":"2023-02-24T11:04:03.780457Z","shell.execute_reply.started":"2023-02-24T11:04:03.775239Z","shell.execute_reply":"2023-02-24T11:04:03.77914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train[train_columns]\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T12:41:03.849286Z","iopub.execute_input":"2023-02-24T12:41:03.849757Z","iopub.status.idle":"2023-02-24T12:41:14.890277Z","shell.execute_reply.started":"2023-02-24T12:41:03.849721Z","shell.execute_reply":"2023-02-24T12:41:14.889416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = train","metadata":{"execution":{"iopub.status.busy":"2023-02-24T12:41:19.715598Z","iopub.execute_input":"2023-02-24T12:41:19.716875Z","iopub.status.idle":"2023-02-24T12:41:19.721933Z","shell.execute_reply.started":"2023-02-24T12:41:19.716825Z","shell.execute_reply":"2023-02-24T12:41:19.721108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = data.iloc[:int(len(data)/50),:]\nlen(data)","metadata":{"execution":{"iopub.status.busy":"2023-02-24T12:41:22.230739Z","iopub.execute_input":"2023-02-24T12:41:22.231502Z","iopub.status.idle":"2023-02-24T12:41:22.240708Z","shell.execute_reply.started":"2023-02-24T12:41:22.231445Z","shell.execute_reply":"2023-02-24T12:41:22.239331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T12:41:27.257638Z","iopub.execute_input":"2023-02-24T12:41:27.258118Z","iopub.status.idle":"2023-02-24T12:41:27.332232Z","shell.execute_reply.started":"2023-02-24T12:41:27.25808Z","shell.execute_reply":"2023-02-24T12:41:27.331112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2023-02-24T12:41:28.169533Z","iopub.execute_input":"2023-02-24T12:41:28.169976Z","iopub.status.idle":"2023-02-24T12:41:28.175808Z","shell.execute_reply.started":"2023-02-24T12:41:28.16993Z","shell.execute_reply":"2023-02-24T12:41:28.174371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# Define the maximum sequence length and number of skills\nsequence_length = 100\nnum_skills = data['content_id'].nunique()\n\n# Create a dictionary to map skill names to IDs\nskill_id_map = {}\nfor i, skill in enumerate(data['content_id'].unique()): #아이디가 숫자가 아닐 경우를 대비\n    skill_id_map[skill] = i","metadata":{"execution":{"iopub.status.busy":"2023-02-24T12:41:34.909561Z","iopub.execute_input":"2023-02-24T12:41:34.910118Z","iopub.status.idle":"2023-02-24T12:41:34.962823Z","shell.execute_reply.started":"2023-02-24T12:41:34.910072Z","shell.execute_reply":"2023-02-24T12:41:34.961763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert the data to the required input format\nnum_sequences = len(data) // sequence_length  # number of sequences of length sequence_length\ninput_data = np.zeros((num_sequences, sequence_length, 2), dtype=np.float32) #99개?\n#output_data = np.zeros((num_sequences, sequence_length, 1), dtype=np.float32) #이거 있어야함?\nfor i in range(num_sequences):\n    start_index = i * sequence_length\n    end_index = start_index + sequence_length\n    sequence_data = data[start_index:end_index]\n    skill_ids = [skill_id_map[skill] for skill in sequence_data['content_id']]\n    correctness = sequence_data['answered_correctly'].astype(np.float32).to_numpy()\n    input_data[i, :, 0] = skill_ids[:] \n    input_data[i, :, 1] = correctness[:] \n    #output_data[i, :, 0] = correctness[1:] #이게 무슨 의미가 있는거지?","metadata":{"execution":{"iopub.status.busy":"2023-02-24T12:41:36.649961Z","iopub.execute_input":"2023-02-24T12:41:36.650394Z","iopub.status.idle":"2023-02-24T12:41:43.947777Z","shell.execute_reply.started":"2023-02-24T12:41:36.650356Z","shell.execute_reply":"2023-02-24T12:41:43.946512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_data.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-24T11:12:01.952809Z","iopub.execute_input":"2023-02-24T11:12:01.953302Z","iopub.status.idle":"2023-02-24T11:12:01.960777Z","shell.execute_reply.started":"2023-02-24T11:12:01.953263Z","shell.execute_reply":"2023-02-24T11:12:01.959593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install torchsummary","metadata":{"execution":{"iopub.status.busy":"2023-02-24T11:12:07.689047Z","iopub.execute_input":"2023-02-24T11:12:07.689989Z","iopub.status.idle":"2023-02-24T11:12:22.586367Z","shell.execute_reply.started":"2023-02-24T11:12:07.68994Z","shell.execute_reply":"2023-02-24T11:12:22.584822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#모델을 개같이 추천해서 잘 될지 모르겠\n\nimport torch\nimport torch.nn as nn\nfrom torchsummary import summary\n\n# Define the DKT model\nclass DKTModel(nn.Module): ##nn.Module의 subclass DKTModel.. 이 안에서 nn.Module에 있는 function들을 활용할 수 있음.\n    def __init__(self, num_skills, hidden_dim):\n        super(DKTModel, self).__init__() #상위 클래스 nn.Module의 function을 쓰기 위한 것임\n        self.num_skills = num_skills\n        self.hidden_dim = hidden_dim\n\n        self.skill_embeddings = nn.Embedding(num_skills, hidden_dim)\n        self.lstm = nn.LSTM(input_size=hidden_dim+1, hidden_size=hidden_dim)\n        self.fc = nn.Sequential(\n            nn.Linear(hidden_dim, 1), # Linear 했을 떄 100% 정확도가 나왔음\n            nn.Sigmoid()\n        )\n\n    def forward(self, input_seq):\n        skill_ids = input_seq[:, :, 0].long() #1. [batch_size, sequence_length, 2] -> [batch_size, sequence_length, 1](skill_id)\n        #print('skill_ids: ',skill_ids.size())\n        correctness = input_seq[:, :, 1].long() #2. [batch_size, sequence_length, 2] -> [batch_size, sequence_length, 1](correctness)\n        #print('correctness: ',correctness.size())\n        skill_embeddings = self.skill_embeddings(skill_ids) #3. output : [batch_size, sequence_length, hidden_dim]\n        #print('skill_embeddings: ', skill_embeddings.size())\n        lstm_input = torch.cat((skill_embeddings, correctness.unsqueeze(-1)), dim=-1) #5. torch.cat : [batch_size, sequence_length, hidden_dim+1] \n        #print('lstm_input: ', lstm_input.size())\n        lstm_output, _ = self.lstm(lstm_input) # [batch_size, sequence_length, hidden_dim]\n        #print('lstm_output: ', lstm_output.size())\n        final_output = self.fc(lstm_output[:, -1, :]) # [batch_size, 1]\n        #print('final_output: ', final_output.size())\n        return final_output.squeeze(-1) # [batch_size]\n        ","metadata":{"execution":{"iopub.status.busy":"2023-02-24T12:36:28.045319Z","iopub.execute_input":"2023-02-24T12:36:28.046429Z","iopub.status.idle":"2023-02-24T12:36:28.056667Z","shell.execute_reply.started":"2023-02-24T12:36:28.046388Z","shell.execute_reply":"2023-02-24T12:36:28.055695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\nfrom tqdm import tqdm\n\n# Split the data into train and test sets\ntrain_data, test_data = train_test_split(input_data, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-02-24T12:41:45.559785Z","iopub.execute_input":"2023-02-24T12:41:45.560217Z","iopub.status.idle":"2023-02-24T12:41:45.576068Z","shell.execute_reply.started":"2023-02-24T12:41:45.560179Z","shell.execute_reply":"2023-02-24T12:41:45.574799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the hyperparameters\nnum_skills = data['content_id'].nunique()\nhidden_dim = 64\nlearning_rate = 0.001\nbatch_size = 64\n\n# Define the model\nmodel = DKTModel(num_skills=num_skills, hidden_dim=hidden_dim)\n\n# Define the loss function and optimizer\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = optim.Adam(model.parameters(), lr=learning_rate)","metadata":{"execution":{"iopub.status.busy":"2023-02-24T12:41:48.622702Z","iopub.execute_input":"2023-02-24T12:41:48.623102Z","iopub.status.idle":"2023-02-24T12:41:48.657709Z","shell.execute_reply.started":"2023-02-24T12:41:48.623071Z","shell.execute_reply":"2023-02-24T12:41:48.656472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in train_loader:\n    a = i\n    break\n\nprint(len(a[0][:][-1]))","metadata":{"execution":{"iopub.status.busy":"2023-02-24T12:18:13.323946Z","iopub.execute_input":"2023-02-24T12:18:13.325394Z","iopub.status.idle":"2023-02-24T12:18:13.333897Z","shell.execute_reply.started":"2023-02-24T12:18:13.325343Z","shell.execute_reply":"2023-02-24T12:18:13.332484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create data loaders for the train and test sets\ntrain_loader = DataLoader(TensorDataset(torch.from_numpy(train_data)), batch_size=batch_size, shuffle=True)\ntest_loader = DataLoader(TensorDataset(torch.from_numpy(test_data)), batch_size=batch_size, shuffle=False)\n\n# Train the model\nnum_epochs = 10\nfor epoch in range(num_epochs):\n    total_loss = 0\n    for batch in tqdm(train_loader, desc=f\"Epoch {epoch+1}\", leave=False): # batch (1,64,100,2)\n        optimizer.zero_grad()\n        input_seq = batch[0] \n        target = batch[0][:, -1, 1]\n        output = model(input_seq)\n        loss = criterion(output, target)\n        loss.backward()\n        optimizer.step()\n        total_loss += loss.item()\n    print(f\"Epoch {epoch+1} - Loss: {total_loss/len(train_loader):.4f}\")","metadata":{"execution":{"iopub.status.busy":"2023-02-24T12:41:51.331988Z","iopub.execute_input":"2023-02-24T12:41:51.332447Z","iopub.status.idle":"2023-02-24T12:44:03.379055Z","shell.execute_reply.started":"2023-02-24T12:41:51.332407Z","shell.execute_reply":"2023-02-24T12:44:03.377938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score, roc_curve\nimport matplotlib.pyplot as plt\n\n# Test the model\ntotal_correct = 0\ntotal_samples = 0\nwith torch.no_grad():\n    for batch in tqdm(test_loader, desc=\"Testing\", leave=False):\n        input_seq = batch[0]\n        target = batch[0][:, -1, 1]\n        output = model(input_seq)\n        #print(target)\n        #print(output)\n        #break\n        predictions = (output > 0.5).float()\n        total_correct += (predictions == target).sum().item()\n        total_samples += target.size(0)\naccuracy = total_correct / total_samples\nprint(f\"Accuracy: {accuracy:.4f}\")\n\nauc = roc_auc_score(predictions, target)\nprint(f\"AUC_ROC: {auc:.4f}\")\n\n# Visualize ROC curve\nfpr, tpr, thresholds = roc_curve(predictions, target)\nplt.plot(fpr, tpr)\nplt.title(\"ROC Curve\")\nplt.xlabel(\"False Positive Rate\")\nplt.ylabel(\"True Positive Rate\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T12:44:35.144759Z","iopub.execute_input":"2023-02-24T12:44:35.145196Z","iopub.status.idle":"2023-02-24T12:44:36.338398Z","shell.execute_reply.started":"2023-02-24T12:44:35.145163Z","shell.execute_reply":"2023-02-24T12:44:36.337006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ex_test = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/example_test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-02-24T12:49:26.946453Z","iopub.execute_input":"2023-02-24T12:49:26.947171Z","iopub.status.idle":"2023-02-24T12:49:26.955384Z","shell.execute_reply.started":"2023-02-24T12:49:26.947134Z","shell.execute_reply":"2023-02-24T12:49:26.954386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ex_test = ex_test.loc[ex_test['user_answer'] != -1,:]\nex_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T12:49:28.195894Z","iopub.execute_input":"2023-02-24T12:49:28.196647Z","iopub.status.idle":"2023-02-24T12:49:28.281285Z","shell.execute_reply.started":"2023-02-24T12:49:28.196604Z","shell.execute_reply":"2023-02-24T12:49:28.279697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = ['user_id','content_id', 'task_container_id', 'answered_correctly']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ex_test = ex_test[columns]\nex_test.head()","metadata":{},"execution_count":null,"outputs":[]}]}