{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. EDA\n\nTổng cộng, được cung cấp 7 tệp.\n\n>Điều chỉnh giáo dục phù hợp với trình độ khả năng của học sinh là một trong nhiều điều có giá trị mà một gia sư AI có thể làm. Thử thách của bạn trong cuộc thi này là một phiên bản của nhiệm vụ tổng thể đó; bạn sẽ dự đoán liệu học sinh có thể trả lời chính xác các câu hỏi tiếp theo của họ hay không. Bạn sẽ được cung cấp cùng một loại thông tin mà một ứng dụng giáo dục hoàn chỉnh sẽ có: thành tích lịch sử của học sinh đó, thành tích của các học sinh khác trong cùng một câu hỏi, siêu dữ liệu về chính câu hỏi đó, v.v.\n\n>Đây là cuộc thi viết mã theo chuỗi thời gian, bạn sẽ nhận được dữ liệu bộ thử nghiệm và đưa ra dự đoán với API chuỗi thời gian của Kaggle. Hãy đảm bảo xem xét kỹ phần Chi tiết API chuỗi thời gian.","metadata":{}},{"cell_type":"markdown","source":"Tại đây cài đặt và tải các thư viện của mình và đặt một số cài đặt mặc định để sử dụng trong tương lai.","metadata":{}},{"cell_type":"code","source":"!pip install ../input/python-datatable/datatable-0.11.0-cp37-cp37m-manylinux2010_x86_64.whl > /dev/null","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-06-03T16:43:46.655060Z","iopub.execute_input":"2022-06-03T16:43:46.655670Z","iopub.status.idle":"2022-06-03T16:44:18.681137Z","shell.execute_reply.started":"2022-06-03T16:43:46.655605Z","shell.execute_reply":"2022-06-03T16:44:18.679973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Nhập các thư viện cơ bản cho EDA:\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n#\n\nimport datatable as dt # để tải .csv nhanh hơn\n\n#\n\nimport gc # để xóa bộ nhớ\nimport warnings # để ẩn cảnh báo\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-03T16:44:41.401194Z","iopub.execute_input":"2022-06-03T16:44:41.401612Z","iopub.status.idle":"2022-06-03T16:44:42.484540Z","shell.execute_reply.started":"2022-06-03T16:44:41.401571Z","shell.execute_reply":"2022-06-03T16:44:42.483192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cài đặt tạo kiểu:\n\nplt.rcParams['figure.figsize'] = [18,10]\nplt.style.use('ggplot')","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:44:51.575001Z","iopub.execute_input":"2022-06-03T16:44:51.575418Z","iopub.status.idle":"2022-06-03T16:44:51.581612Z","shell.execute_reply.started":"2022-06-03T16:44:51.575379Z","shell.execute_reply":"2022-06-03T16:44:51.580317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tải dữ liệu\n\nDữ liệu đưa ra là rất lớn, không dễ dàng để tải tất cả chúng vào RAM mà không bị hết bộ nhớ. Cần đặt các kiểu dtypes cho mỗi cột để giảm mức sử dụng bộ nhớ. Theo mặc định, chúng là 32/64 cho các cột số, nhưng có thể chọn các loại theo cách thủ công dựa trên số lượng tối đa của chúng trong cột. Các lựa chọn kiểu loại này dựa trên:\n\n> int8 / uint8: sử dụng 1 byte bộ nhớ, phạm vi từ -128/127 hoặc 0/255,\n\n> bool: sử dụng 1 byte, true hoặc false,\n\n> float16 / int16 / uint16: sử dụng 2 byte bộ nhớ, phạm vi từ -32768 đến 32767 hoặc 0/65535,\n\n> float32 / int32 / uint32: sử dụng 4 byte bộ nhớ, phạm vi từ -2147483648 đến 2147483647,\n\n> float64 / int64 / uint64: sử dụng 8 byte bộ nhớ.\n\n[Source](https://medium.com/@vincentteyssier/optimizing-the-size-of-a-pandas-dataframe-for-low-memory-environment-5f07db3d72e)\n\nSử dụng ** datatable ** để tải nhanh hơn. Bạn có thể tìm hiểu giải thích sâu hơn ở [here](https://www.kaggle.com/rohanrao/tutorial-on-reading-large-datasets).\n\n","metadata":{}},{"cell_type":"markdown","source":"# Exploring Train","metadata":{}},{"cell_type":"code","source":"# Dict for dtypes:\n\ndata_types = {\n    'row_id': 'int32',\n    'timestamp': 'int64',\n    'user_id': 'int64',\n    'content_id': 'int16',\n    'content_type_id': 'int8',\n    'task_container_id': 'int16',\n    'user_answer': 'int8',\n    'answered_correctly': 'int8',\n    'prior_question_elapsed_time': 'float32',\n    'prior_question_had_explanation': 'boolean'\n}","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:44:56.216367Z","iopub.execute_input":"2022-06-03T16:44:56.216722Z","iopub.status.idle":"2022-06-03T16:44:56.223573Z","shell.execute_reply.started":"2022-06-03T16:44:56.216691Z","shell.execute_reply":"2022-06-03T16:44:56.222174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tải dữ liệu với datatable và chuyển đổi nó thành pandas df:\n\ntrain_df = dt.fread('../input/riiid-test-answer-prediction/train.csv').to_pandas()\n\n# Chọn ngẫu nhiên một phần dữ liệu để xử lý nhanh hơn.\n\ntrain_df = train_df.sample(len(train_df)//5,random_state=42)\n\n\n# Đặt loại dtypes cho mỗi cột\n\nfor column, d_type in data_types.items():\n    train_df[column] = train_df[column].astype(d_type) ","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2022-06-03T16:45:00.495522Z","iopub.execute_input":"2022-06-03T16:45:00.495880Z","iopub.status.idle":"2022-06-03T16:47:25.271561Z","shell.execute_reply.started":"2022-06-03T16:45:00.495846Z","shell.execute_reply":"2022-06-03T16:47:25.270104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tải các tệp dữ liệu khác:\n\nquestions_df = pd.read_csv('../input/riiid-test-answer-prediction/questions.csv')\nlectures_df = pd.read_csv('../input/riiid-test-answer-prediction/lectures.csv')\ntest_df = pd.read_csv('../input/riiid-test-answer-prediction/example_test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:47:45.409520Z","iopub.execute_input":"2022-06-03T16:47:45.409989Z","iopub.status.idle":"2022-06-03T16:47:45.473107Z","shell.execute_reply.started":"2022-06-03T16:47:45.409928Z","shell.execute_reply":"2022-06-03T16:47:45.472101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hiển thị các loại và sử dụng bộ nhớ.\n\ntrain_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:47:48.123995Z","iopub.execute_input":"2022-06-03T16:47:48.124601Z","iopub.status.idle":"2022-06-03T16:47:48.151851Z","shell.execute_reply.started":"2022-06-03T16:47:48.124564Z","shell.execute_reply":"2022-06-03T16:47:48.150845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Timestamp\n\n#### \"timestamp\": Thời gian tính bằng mili giây giữa lần tương tác của người dùng này đến khi hoàn thành sự kiện đầu tiên từ người dùng đó.\n\n#### Tại đây, có thể thấy các phần trước đó của dòng thời gian hoạt động nhiều hơn so với các phiên dài hơn, điều này được mong đợi. Ngoài ra, có thể nhận thấy rằng có một số người dùng dành thời gian khá dài!","metadata":{}},{"cell_type":"code","source":"# Vẽ đồ thị liên quan đến timestamp\n\nfig, ax = plt.subplots(ncols=2, nrows=1, figsize=(32,14))\n\nsns.distplot(train_df.timestamp, kde=False,hist_kws={\n                 'rwidth': 0.85,\n                 'edgecolor': 'black',\n                 'alpha': 0.8}, bins=100, ax=ax[0])\n\nax[0].set_xlabel('Time in Miliseconds')\nax[0].set_ylabel('Count')\nax[0].set_title('Timestamp Distribution', weight='bold')\n\n\nsns.distplot(train_df.groupby('user_id').agg({'timestamp': 'mean'}), kde=False, hist_kws={\n                 'rwidth': 0.85,\n                 'edgecolor': 'black',\n                 'alpha': 0.8}, bins=50,ax=ax[1])\n\nax[1].set_xlabel('Time in Miliseconds*')\nax[1].set_ylabel('Count')\nax[1].set_title('Mean Timestamp Per User Distribution', weight='bold')\n\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-06-03T16:47:50.646431Z","iopub.execute_input":"2022-06-03T16:47:50.647170Z","iopub.status.idle":"2022-06-03T16:47:53.817867Z","shell.execute_reply.started":"2022-06-03T16:47:50.647124Z","shell.execute_reply":"2022-06-03T16:47:53.816878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Người dùng và Nội dung\n\n##### Ở đây có ID duy nhất cho mỗi người dùng, nếu tính tất cả các mối quan tâm do người dùng thực hiện, có thể thấy có một số người dùng khá tích cực, trong số ~ 300 nghìn người dùng duy nhất, chúng tôi thấy 25 người hàng đầu gần như kiếm được hơn 3 nghìn các tương tác.\n\n#### Nội dung khá giống với ID người dùng. Tại đây, có thể thấy hầu hết các nội dung phổ biến, có vẻ như nội dung # 6116 thực sự được yêu thích nhất, tiếp theo là # 6173 và # 4120 với khoảng 400 nghìn lượt tương tác.","metadata":{}},{"cell_type":"code","source":"# Vẽ biểu đồ liên quan đến người dùng và nội dung.\n\nfig, ax = plt.subplots(ncols=2, nrows=1, figsize=(32,14))\n\n# Distplot:\n\nsns.countplot(y='user_id', data=train_df, order=train_df.user_id.value_counts().index[:25], palette='autumn',ax = ax[0])\nax[0].set_title('Top 25 Active Users', weight='bold')\n\n# Countplot:\n\nsns.countplot(y='content_id', data=train_df, order=train_df.content_id.value_counts().index[:25], palette='autumn',ax = ax[1])\nax[1].set_title('Top 25 Content', weight='bold')\n\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:48:00.729372Z","iopub.execute_input":"2022-06-03T16:48:00.730022Z","iopub.status.idle":"2022-06-03T16:48:14.649501Z","shell.execute_reply.started":"2022-06-03T16:48:00.729954Z","shell.execute_reply":"2022-06-03T16:48:14.648409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loại nội dung\n\n#### \"content_type_id\": 0 nếu sự kiện là một câu hỏi được đặt ra cho người dùng, 1 nếu sự kiện là người dùng đang xem một bài giảng.\n\n#### Ở đây chúng ta có thể thấy rằng 98% mẫu của chúng ta là câu hỏi và ~ 2% là bài giảng.","metadata":{}},{"cell_type":"code","source":"# Các loại nội dung vẽ sơ đồ\n\ng=sns.countplot(train_df.content_type_id, palette='autumn')\n\n# Thêm phần trăm\n\ntotal = float(len(train_df['content_type_id']))\n\nfor p in g.patches:\n    height = p.get_height()\n    g.text(p.get_x() + p.get_width() / 2.,\n            height + 2,\n            '{:1.2f}%'.format((height / total) * 100),\n            ha='center')\n\nplt.ylabel('Count*10^7')    \nplt.title('Content Types - 0: Question, 1: Lecture', weight='bold')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:48:28.659244Z","iopub.execute_input":"2022-06-03T16:48:28.659716Z","iopub.status.idle":"2022-06-03T16:48:30.033519Z","shell.execute_reply.started":"2022-06-03T16:48:28.659670Z","shell.execute_reply":"2022-06-03T16:48:30.032383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Vùng chứa Tác vụ\n\n#### \"task_container_id\": Mã id cho loạt câu hỏi hoặc bài giảng. Ví dụ: một người dùng có thể thấy ba câu hỏi liên tiếp trước khi xem giải thích cho bất kỳ câu hỏi nào trong số đó. Tất cả ba câu đó sẽ chia sẻ một task_container_id.\n\n#### Có thể thấy rằng các tác vụ có ID nhỏ hơn phổ biến hơn nhiều so với các số lớn hơn, trong khi đó, nhiệm vụ phổ biến nhất là # 14","metadata":{}},{"cell_type":"code","source":"# Lập đồ thị vùng chứa nhiệm vụ:\n\nfig, ax = plt.subplots(ncols=2, nrows=1, figsize=(32,14))\n\n# Phân bổ:\n\nsns.distplot(train_df.task_container_id, kde=False,hist_kws={\n                 'rwidth': 0.85,\n                 'edgecolor': 'black',\n                 'alpha': 0.8}, ax=ax[0])\n\n\nax[0].set_ylabel('Frequency')\nax[0].set_title('Task Container ID Distribution', weight='bold')\n\n# Số lượng:\n\nsns.countplot(y='task_container_id', data=train_df, order=train_df.task_container_id.value_counts().index[:25], palette='autumn', ax=ax[1])\nax[1].set_title('Top 25 Tasks', weight='bold')\n\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:49:04.014639Z","iopub.execute_input":"2022-06-03T16:49:04.015115Z","iopub.status.idle":"2022-06-03T16:49:09.925572Z","shell.execute_reply.started":"2022-06-03T16:49:04.015077Z","shell.execute_reply":"2022-06-03T16:49:09.924260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Câu trả lời của người dùng\n\n#### \"user_answer\": Câu trả lời của người dùng cho câu hỏi, nếu có. Đọc -1 là null cho các bài giảng.\n\n#### Ở đây có thể thấy tùy chọn trả lời số 2 ít phổ biến hơn so với các câu trả lời còn lại trong ba câu trả lời. Có vẻ như người dùng / người hướng dẫn không thích tùy chọn số 2 cho lắm :)","metadata":{}},{"cell_type":"code","source":"# Lập đồ thị câu trả lời của người dùng:\n\ng=sns.countplot(train_df.user_answer, hue=train_df.answered_correctly, palette='autumn', order=train_df.user_answer.value_counts().index)\n\n# Thêm phần trăm:\n\ntotal = float(len(train_df['user_answer']))\n\nfor p in g.patches:\n    height = p.get_height()\n    g.text(p.get_x() + p.get_width() / 2.,\n            height + 2,\n            '{:1.2f}%'.format((height / total) * 100),\n            ha='center')\n\nplt.title('False/Correct per User Answer  (-1 for Lectures)', weight='bold')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:49:13.397122Z","iopub.execute_input":"2022-06-03T16:49:13.398276Z","iopub.status.idle":"2022-06-03T16:49:16.413003Z","shell.execute_reply.started":"2022-06-03T16:49:13.398205Z","shell.execute_reply":"2022-06-03T16:49:16.412002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Câu hỏi trước_Prior Questions\n\n\n#### before_question_elapsed_time: Thời gian trung bình tính bằng mili giây người dùng trả lời từng câu hỏi trong gói câu hỏi trước đó, bỏ qua bất kỳ bài giảng nào ở giữa. Không có giá trị cho gói câu hỏi hoặc bài giảng đầu tiên của người dùng. Lưu ý rằng thời gian là thời gian trung bình mà người dùng dành để giải quyết từng câu hỏi trong nhóm trước đó.\n\n#### before_question_had_explanation: Người dùng có thấy giải thích và (các) câu trả lời chính xác hay không sau khi trả lời gói câu hỏi trước đó, bỏ qua bất kỳ bài giảng nào ở giữa. Giá trị được chia sẻ trên một gói câu hỏi và không có giá trị đối với gói câu hỏi hoặc bài giảng đầu tiên của người dùng. Thông thường, một số câu hỏi đầu tiên mà người dùng nhìn thấy là một phần của kiểm tra chẩn đoán giới thiệu mà họ không nhận được bất kỳ phản hồi nào.","metadata":{}},{"cell_type":"code","source":"# Vẽ đồ thị cho các câu hỏi trước những nội dung liên quan:\n\nfig, ax = plt.subplots(ncols=2, nrows=1, figsize=(32,14))\n\n\nsns.distplot(train_df.prior_question_elapsed_time.dropna(), kde=False, hist_kws={\n                 'rwidth': 0.85,\n                 'edgecolor': 'black',\n                 'alpha': 0.8}, ax=ax[0])\n\nax[0].set_ylabel('Count')\nax[0].set_title('Prior Question Elapsed Time Distribution', weight='bold')\n\n\ng=sns.countplot(train_df.prior_question_had_explanation.dropna(), palette='autumn', ax=ax[1])\n\n# thêm phần trăm\n\ntotal = float(len(train_df.prior_question_had_explanation.dropna()))\nfor p in g.patches:\n    height = p.get_height()\n    g.text(p.get_x() + p.get_width() / 2.,\n            height + 2,\n            '{:1.2f}%'.format((height / total) * 100),\n            ha='center')\n\nax[1].set_title('Prior Question Elapsed had Explanation?', weight='bold')\n    \nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-06-03T16:49:19.048442Z","iopub.execute_input":"2022-06-03T16:49:19.048907Z","iopub.status.idle":"2022-06-03T16:49:33.510420Z","shell.execute_reply.started":"2022-06-03T16:49:19.048861Z","shell.execute_reply":"2022-06-03T16:49:33.509212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Bỏ các bài giảng khỏi khung dữ liệu\n\ntrain_df = train_df.loc[train_df['answered_correctly'] != -1].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:50:27.087996Z","iopub.execute_input":"2022-06-03T16:50:27.088404Z","iopub.status.idle":"2022-06-03T16:50:28.990101Z","shell.execute_reply.started":"2022-06-03T16:50:27.088362Z","shell.execute_reply":"2022-06-03T16:50:28.988623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Câu hỏi\n\n#### Tại đây, chúng tôi thêm dữ liệu bổ sung mà đã cung cấp. Vì vậy, có thể tìm thấy một số gợi ý sâu sắc ...","metadata":{}},{"cell_type":"code","source":"# Hợp nhất dữ liệu câu hỏi với dữ liệu train:\n\ntrain_df = pd.merge(train_df,questions_df[['question_id','part']], how='left', left_on='content_id', right_on='question_id').sort_values('row_id')\ntrain_df['part'] = train_df['part'].astype('int8')\n","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:50:31.338208Z","iopub.execute_input":"2022-06-03T16:50:31.338625Z","iopub.status.idle":"2022-06-03T16:50:40.863728Z","shell.execute_reply.started":"2022-06-03T16:50:31.338591Z","shell.execute_reply":"2022-06-03T16:50:40.862780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Phần Câu hỏi\n\n#### part: Phần liên quan của bài thi TOEIC.\n\n#### Hợp nhất các phần câu hỏi dựa trên ID câu hỏi cụ thể của chúng. Những  phần liên quan của bài thi TOEIC được giải thích ở đây:\n\n![](https://i.imgur.com/2wqNAJ1.png)\n![](https://i.imgur.com/4B3AQyL.png)\n\n### Ở đây chúng tôi nhận thấy rằng các câu hỏi Phần 5 là câu hỏi phổ biến nhất, trong đó phải hoàn thành các câu bằng bốn lựa chọn cho sẵn.","metadata":{}},{"cell_type":"code","source":"g=sns.countplot(train_df.part, hue=train_df.answered_correctly, palette='autumn')\n\n# thêm phần trăm\n\ntotal = float(len(train_df.part))\nfor p in g.patches:\n    height = p.get_height()\n    g.text(p.get_x() + p.get_width() / 2.,\n            height + 2,\n            '{:1.2f}%'.format((height / total) * 100),\n            ha='center')\n\nplt.title('False/Correct per Question Part', weight='bold')\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-06-03T16:50:45.451059Z","iopub.execute_input":"2022-06-03T16:50:45.451522Z","iopub.status.idle":"2022-06-03T16:50:48.441777Z","shell.execute_reply.started":"2022-06-03T16:50:45.451479Z","shell.execute_reply":"2022-06-03T16:50:48.440497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# nhóm user id và lấy trung bình, tổng, số\n\nusr_ans = train_df.groupby('user_id').agg({ 'answered_correctly': ['mean','sum', 'count']})\nusr_ans.columns = ['avg_correct_answer','num_of_correct', 'total_answers']\n\n# thay đổi loại để giảm bộ nhớ (mặc định = 64)\n\nusr_ans['num_of_correct'] = usr_ans['num_of_correct'].astype('int16')\nusr_ans['total_answers'] = usr_ans['total_answers'].astype('int16')\n\n\ntrain_df = pd.merge(train_df, usr_ans, how='left', on = 'user_id')","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:50:54.249440Z","iopub.execute_input":"2022-06-03T16:50:54.249915Z","iopub.status.idle":"2022-06-03T16:51:03.595532Z","shell.execute_reply.started":"2022-06-03T16:50:54.249878Z","shell.execute_reply":"2022-06-03T16:51:03.593998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Độ chính xác của câu trả lời đúng\n\n#### lọc những người dùng đã trả lời hơn 100 câu hỏi và sắp xếp chúng dựa trên tỷ lệ câu trả lời đúng của họ, có một số người dùng khá chính xác!","metadata":{}},{"cell_type":"code","source":"#  Biểu đồ 25 người dùng chính xác hàng đầu\n\nsns.barplot(x='avg_correct_answer',y='user_id', orient='h', data=usr_ans[usr_ans['total_answers']>100].sort_values('avg_correct_answer', ascending=False).reset_index().iloc[:25],\n            palette='autumn', order=usr_ans[usr_ans['total_answers']>100].sort_values('avg_correct_answer', ascending=False).reset_index().user_id.iloc[:25])\n\nplt.title('Top 25 Accurate Users', weight='bold')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:51:09.823398Z","iopub.execute_input":"2022-06-03T16:51:09.823785Z","iopub.status.idle":"2022-06-03T16:51:10.211438Z","shell.execute_reply.started":"2022-06-03T16:51:09.823750Z","shell.execute_reply":"2022-06-03T16:51:10.210137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Độ chính xác của câu trả lời so với Tổng số câu hỏi đã được trả lời\n\n#### Có thể quan sát thấy tỷ lệ câu trả lời đúng ngày càng tăng trên tổng số câu hỏi mà người dùng trả lời. ","metadata":{}},{"cell_type":"code","source":"# vẽ tổng số câu trả lời so với độ chính xác trung bình\n\nsns.regplot(data=usr_ans[usr_ans['total_answers']> 100], y='avg_correct_answer', x='total_answers', ci=False, scatter_kws={'alpha':0.5}, line_kws={\"color\": \"orange\"})\nplt.axhline(train_df.avg_correct_answer.mean(), color='k', linestyle='dashed', linewidth=3)\nplt.axvline(train_df.total_answers.mean(), color='k', linestyle='dashed', linewidth=3)\n\nmin_ylim, max_ylim = plt.ylim()\nplt.text(train_df.total_answers.mean()+25, max_ylim*0.20, 'Average Questions Solved {:.2f}'.format(train_df.total_answers.mean()))\nplt.text(train_df.total_answers.mean()+2400, max_ylim*0.6, 'Average Correct Answer: {:.2f}'.format(train_df.avg_correct_answer.mean()))\n\nplt.title('Average Correct Answer Ratio vs. Total Questions Answered per User', weight='bold')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:51:25.352875Z","iopub.execute_input":"2022-06-03T16:51:25.353311Z","iopub.status.idle":"2022-06-03T16:51:27.759725Z","shell.execute_reply.started":"2022-06-03T16:51:25.353276Z","shell.execute_reply":"2022-06-03T16:51:27.756214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Độ chính xác của câu trả lời - Mối quan hệ về thời gian\n\n#### Lấy thời gian tối đa của người dùng và lọc ra những người dùng nếu họ có ít hơn một giờ. Nhận được sự gia tăng nhẹ về độ chính xác của người dùng bằng cách tăng thời gian sử dụng nhưng có vẻ như không đáng kể.","metadata":{}},{"cell_type":"code","source":"# Biểu đồ answer accuracy vs time\n\ntotal_time = train_df.groupby('user_id')[\"timestamp\"].max()\ntotal_time = pd.merge(total_time.reset_index(), usr_ans.reset_index(), how='left', on = 'user_id')\n\nsns.regplot(data=total_time[total_time['timestamp']> 3.6e+6], y='avg_correct_answer', x='timestamp', ci=False, scatter_kws={'alpha':0.5}, line_kws={\"color\": \"orange\"})\n\nplt.axvline(total_time.timestamp.mean(), color='k', linestyle='dashed', linewidth=3)\nplt.text(total_time.timestamp.mean()+total_time.timestamp.mean()*0.1, max_ylim*0.03, 'Average Time {:.2f}'.format(total_time.timestamp.mean()))\n\nplt.title('Average Correct Answer Ratio vs. Time Spent', weight='bold')\nplt.show()\n\n# xóa một số biến để tiết kiệm bộ nhớ:\n\ndel total_time\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:51:32.908950Z","iopub.execute_input":"2022-06-03T16:51:32.909423Z","iopub.status.idle":"2022-06-03T16:51:47.471375Z","shell.execute_reply.started":"2022-06-03T16:51:32.909380Z","shell.execute_reply":"2022-06-03T16:51:47.470299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Độ chính xác của câu trả lời - Mối quan hệ nội dung\n\n#### Khi tính độ chính xác của câu trả lời theo nội dung, có thể thấy rằng nội dung phổ biến / chung chung hơn có tỷ lệ câu trả lời đúng thấp hơn. Trong khi đó các câu hỏi ít phổ biến / cụ thể hơn có tỷ lệ đúng cao hơn.","metadata":{}},{"cell_type":"code","source":"# tạo tính năng mới dựa trên id nội dung\n\ncnt_ans = train_df.groupby('content_id').agg({ 'answered_correctly': ['mean','sum', 'count']})\ncnt_ans.columns = ['avg_correct_answer_c','num_of_correct_c', 'total_answers_c']\n\n# thay đổi loại để giảm bộ nhớ (mặc định = 64)\n\ncnt_ans['num_of_correct_c'] = cnt_ans['num_of_correct_c'].astype('int32')\ncnt_ans['total_answers_c'] = cnt_ans['total_answers_c'].astype('int32')\n\ntrain_df = pd.merge(train_df, cnt_ans, how='left', on = 'content_id')","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:52:06.596671Z","iopub.execute_input":"2022-06-03T16:52:06.597304Z","iopub.status.idle":"2022-06-03T16:52:15.532649Z","shell.execute_reply.started":"2022-06-03T16:52:06.597260Z","shell.execute_reply":"2022-06-03T16:52:15.531298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Biểu đồ contents vs. answer accuracies\n\nsns.regplot(data=cnt_ans[cnt_ans['total_answers_c']> 100], y='avg_correct_answer_c', x='total_answers_c', ci=False, scatter_kws={'alpha':0.5}, line_kws={\"color\": \"orange\"})\n\n# Biểu đồ mean lines\n\nplt.axhline(train_df.avg_correct_answer_c.mean(), color='k', linestyle='dashed', linewidth=3)\nplt.axvline(train_df.total_answers_c.mean(), color='k', linestyle='dashed', linewidth=3)\n\n\nmin_ylim, max_ylim = plt.ylim()\nplt.text(35000, max_ylim*0.65, 'Average Correct Answer: {:.2f}'.format(train_df.avg_correct_answer_c.mean()))\nplt.text(5500, max_ylim*0.10, 'Average Questions Solved per Content: {:.2f}'.format(train_df.total_answers_c.mean()))\n\nplt.title('Average Correct Answer Ratio vs. Total Questions Answered per Content', weight='bold')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:52:26.789102Z","iopub.execute_input":"2022-06-03T16:52:26.789529Z","iopub.status.idle":"2022-06-03T16:52:27.767535Z","shell.execute_reply.started":"2022-06-03T16:52:26.789492Z","shell.execute_reply":"2022-06-03T16:52:27.766393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:52:49.122788Z","iopub.execute_input":"2022-06-03T16:52:49.123318Z","iopub.status.idle":"2022-06-03T16:52:49.162370Z","shell.execute_reply.started":"2022-06-03T16:52:49.123201Z","shell.execute_reply":"2022-06-03T16:52:49.161350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Baseline Model\n\n#### Đây là phần sẽ thực hiện một số mô hình cơ sở đơn giản để đo điểm chuẩn cho các mô hình tương lai của chúng tôi. Bắt đầu bằng cách điền vào một số giá trị còn thiếu trong dữ liệu và sau đó chia nó thành X và y để lập mô hình. Cần dự đoán xem người dùng đã trả lời câu hỏi cụ thể có đúng hay không.","metadata":{}},{"cell_type":"code","source":"#Tạo biến X để training\n\nX = train_df.copy()","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:52:54.657733Z","iopub.execute_input":"2022-06-03T16:52:54.658535Z","iopub.status.idle":"2022-06-03T16:52:55.627324Z","shell.execute_reply.started":"2022-06-03T16:52:54.658472Z","shell.execute_reply":"2022-06-03T16:52:55.626406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# điền giá trị N/A\n\nX['prior_question_elapsed_time'].fillna(0,  inplace=True)\nX['prior_question_had_explanation'] = X['prior_question_had_explanation'].fillna(value = False).astype(bool)","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:52:58.023062Z","iopub.execute_input":"2022-06-03T16:52:58.024025Z","iopub.status.idle":"2022-06-03T16:52:58.131905Z","shell.execute_reply.started":"2022-06-03T16:52:58.023924Z","shell.execute_reply":"2022-06-03T16:52:58.130347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Xóa df train để làm sạch bộ nhớ:\n\ndel train_df\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:53:01.168081Z","iopub.execute_input":"2022-06-03T16:53:01.168511Z","iopub.status.idle":"2022-06-03T16:53:01.286611Z","shell.execute_reply.started":"2022-06-03T16:53:01.168472Z","shell.execute_reply":"2022-06-03T16:53:01.285335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Đặt X và Y để đào tạo:\n\nX=X.sort_values(['user_id'])\ny = X[[\"answered_correctly\"]]\nX = X.drop([\"answered_correctly\"], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:53:05.124392Z","iopub.execute_input":"2022-06-03T16:53:05.124789Z","iopub.status.idle":"2022-06-03T16:53:10.392842Z","shell.execute_reply.started":"2022-06-03T16:53:05.124754Z","shell.execute_reply":"2022-06-03T16:53:10.391790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# Nhận nhãn số cho dữ liệu phân loại:\n\nlb_make = LabelEncoder()\nX[\"prior_question_had_explanation_enc\"] = lb_make.fit_transform(X[\"prior_question_had_explanation\"])\nX['prior_question_had_explanation_enc'] = X['prior_question_had_explanation_enc'].astype('int8')\nX.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:53:12.573349Z","iopub.execute_input":"2022-06-03T16:53:12.573755Z","iopub.status.idle":"2022-06-03T16:53:13.716297Z","shell.execute_reply.started":"2022-06-03T16:53:12.573721Z","shell.execute_reply":"2022-06-03T16:53:13.714788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lựa chọn các tính năng để đào tạo:\n\nX = X[['avg_correct_answer','num_of_correct','total_answers', 'avg_correct_answer_c', 'prior_question_elapsed_time','prior_question_had_explanation_enc','part']] ","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:53:17.158595Z","iopub.execute_input":"2022-06-03T16:53:17.159205Z","iopub.status.idle":"2022-06-03T16:53:17.536626Z","shell.execute_reply.started":"2022-06-03T16:53:17.159156Z","shell.execute_reply":"2022-06-03T16:53:17.535308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Kiểm tra hình dạng của dữ liệu đào tạo để chắc chắn:\nX.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:53:21.234534Z","iopub.execute_input":"2022-06-03T16:53:21.235016Z","iopub.status.idle":"2022-06-03T16:53:21.242170Z","shell.execute_reply.started":"2022-06-03T16:53:21.234947Z","shell.execute_reply":"2022-06-03T16:53:21.240807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dùng thuật toán phân loại Lightgbm tham số mặc định để training.","metadata":{}},{"cell_type":"code","source":"# Tải (các) mô hình để thử nghiệm:\nfrom sklearn.model_selection import StratifiedKFold, cross_validate\nimport lightgbm as lgb\n\nlight = lgb.LGBMClassifier(\n)","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:53:24.032856Z","iopub.execute_input":"2022-06-03T16:53:24.033662Z","iopub.status.idle":"2022-06-03T16:53:24.146546Z","shell.execute_reply.started":"2022-06-03T16:53:24.033599Z","shell.execute_reply":"2022-06-03T16:53:24.145391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Validation\n\n#### Phân tầng và xáo trộn mục tiêu  và xác thực nó bằng cách sử dụng 3 lần.","metadata":{}},{"cell_type":"code","source":"# Đặt kfold phân tầng để xác thực:\n\nkf = StratifiedKFold(3, shuffle=True, random_state=42)\nclassifiers = [light]","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:53:27.467135Z","iopub.execute_input":"2022-06-03T16:53:27.467558Z","iopub.status.idle":"2022-06-03T16:53:27.473293Z","shell.execute_reply.started":"2022-06-03T16:53:27.467521Z","shell.execute_reply":"2022-06-03T16:53:27.471975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def model_check(X, y, classifiers, cv):\n    \n    ''' Một chức năng để kiểm tra nhiều bộ phân loại và trả về một số số liệu. '''\n    \n    model_table = pd.DataFrame()\n\n    row_index = 0\n    for cls in classifiers:\n\n        MLA_name = cls.__class__.__name__\n        model_table.loc[row_index, 'Model Name'] = MLA_name\n        \n        cv_results = cross_validate(\n            cls,\n            X,\n            y,\n            cv=cv,\n            scoring=('accuracy','f1','roc_auc'),\n            return_train_score=True,\n            n_jobs=-1\n        )\n        model_table.loc[row_index, 'Train Roc/AUC Mean'] = cv_results[\n            'train_roc_auc'].mean()\n        model_table.loc[row_index, 'Test Roc/AUC Mean'] = cv_results[\n            'test_roc_auc'].mean()\n        model_table.loc[row_index, 'Test Roc/AUC Std'] = cv_results['test_roc_auc'].std()\n\n        model_table.loc[row_index, 'Time'] = cv_results['fit_time'].mean()\n\n        row_index += 1        \n\n    model_table.sort_values(by=['Test Roc/AUC Mean'],\n                            ascending=False,\n                            inplace=True)\n\n    return model_table","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:53:30.964003Z","iopub.execute_input":"2022-06-03T16:53:30.964731Z","iopub.status.idle":"2022-06-03T16:53:30.977636Z","shell.execute_reply.started":"2022-06-03T16:53:30.964683Z","shell.execute_reply":"2022-06-03T16:53:30.976420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Results\n","metadata":{}},{"cell_type":"code","source":"#Hiển thị kết quả mô hình mặc định:\n\nraw_models = model_check(X, y, classifiers, kf)\ndisplay(raw_models)","metadata":{"execution":{"iopub.status.busy":"2022-06-03T16:53:34.764152Z","iopub.execute_input":"2022-06-03T16:53:34.764853Z","iopub.status.idle":"2022-06-03T17:03:44.951894Z","shell.execute_reply.started":"2022-06-03T16:53:34.764810Z","shell.execute_reply":"2022-06-03T17:03:44.949919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction\n\n#### Sử dụng môi trường dự đoán cạnh tranh cụ thể, dự đoán các mẫu thử nghiệm và gửi chúng bằng cách sử dụng gói 'riiideducation'.","metadata":{}},{"cell_type":"code","source":"# Nhập gói riid:\n\nimport riiideducation\n\nenv = riiideducation.make_env()\n\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2022-06-03T17:04:03.219366Z","iopub.execute_input":"2022-06-03T17:04:03.220032Z","iopub.status.idle":"2022-06-03T17:04:03.275524Z","shell.execute_reply.started":"2022-06-03T17:04:03.219977Z","shell.execute_reply":"2022-06-03T17:04:03.274364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit mô hình\n\nlight.fit(X, y)\n\n# Đặt lại các chỉ mục để hợp nhất trước:\n\ncnt_ans=cnt_ans.reset_index()\nusr_ans=usr_ans.reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-06-03T17:04:07.506594Z","iopub.execute_input":"2022-06-03T17:04:07.507022Z","iopub.status.idle":"2022-06-03T17:06:05.223655Z","shell.execute_reply.started":"2022-06-03T17:04:07.506977Z","shell.execute_reply":"2022-06-03T17:06:05.222274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tải và dự đoán các mẫu thử nghiệm\n\nfor (test_df, sample_prediction_df) in iter_test:\n    test_df = test_df.merge(usr_ans, how = 'left', on = 'user_id')\n    test_df = test_df.merge(cnt_ans, how = 'left', on = 'content_id')\n    test_df = pd.merge_ordered(test_df,questions_df[['question_id','part']], how='left', left_on='content_id', right_on='question_id', fill_method='ffill')\n    test_df['prior_question_had_explanation'] = test_df['prior_question_had_explanation'].fillna(value = False).astype(bool)\n    test_df['prior_question_elapsed_time'].fillna(0,  inplace=True)\n    test_df['avg_correct_answer'].fillna(0.5, inplace=True)\n    test_df['avg_correct_answer_c'].fillna(0.5, inplace=True)\n    test_df.fillna(value = -1, inplace = True)\n    test_df[\"prior_question_had_explanation_enc\"] = lb_make.fit_transform(test_df[\"prior_question_had_explanation\"])\n    \n    \n    y_pred = light.predict_proba(test_df[['avg_correct_answer','num_of_correct','total_answers', 'avg_correct_answer_c', 'prior_question_elapsed_time','prior_question_had_explanation_enc','part']])[:,1]\n    test_df['answered_correctly'] = y_pred\n    env.predict(test_df.loc[test_df['content_type_id'] == 0, ['row_id', 'answered_correctly']])\n","metadata":{"execution":{"iopub.status.busy":"2022-06-03T17:06:10.807983Z","iopub.execute_input":"2022-06-03T17:06:10.808404Z","iopub.status.idle":"2022-06-03T17:06:11.880738Z","shell.execute_reply.started":"2022-06-03T17:06:10.808370Z","shell.execute_reply":"2022-06-03T17:06:11.879769Z"},"trusted":true},"execution_count":null,"outputs":[]}]}