{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Nội dung\n\n[1 Exploring Train](#1-Exploring-Train)\n\n[2 Exploring Questions](#2-Exploring-Questions)\n\n[3 Exploring Lectures](#3-Exploring-Lectures)\n\n[4 Train pipeline](#4-Train-pipeline)","metadata":{"papermill":{"duration":0.074059,"end_time":"2020-12-17T09:47:54.969296","exception":false,"start_time":"2020-12-17T09:47:54.895237","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport os\nfrom matplotlib.ticker import FuncFormatter\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/riiid-test-answer-prediction'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_kg_hide-input":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","papermill":{"duration":1.184028,"end_time":"2020-12-17T09:47:56.375118","exception":false,"start_time":"2020-12-17T09:47:55.19109","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-01-14T07:45:34.245336Z","iopub.execute_input":"2022-01-14T07:45:34.245751Z","iopub.status.idle":"2022-01-14T07:45:34.999783Z","shell.execute_reply.started":"2022-01-14T07:45:34.245703Z","shell.execute_reply":"2022-01-14T07:45:34.998665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntrain = pd.read_pickle(\"../input/riid-multi-format-train-set/riiid_train.pkl.gzip\")\n\nprint(\"Train size:\", train.shape)","metadata":{"_kg_hide-input":true,"papermill":{"duration":45.316915,"end_time":"2020-12-17T09:48:41.918495","exception":false,"start_time":"2020-12-17T09:47:56.60158","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-01-14T07:45:35.002005Z","iopub.execute_input":"2022-01-14T07:45:35.002646Z","iopub.status.idle":"2022-01-14T07:46:21.590267Z","shell.execute_reply.started":"2022-01-14T07:45:35.002596Z","shell.execute_reply":"2022-01-14T07:46:21.589493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Kiểm tra mỗi column chiếm bao nhiêu bộ nhớ","metadata":{"papermill":{"duration":0.078327,"end_time":"2020-12-17T09:48:42.076745","exception":false,"start_time":"2020-12-17T09:48:41.998418","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train.memory_usage(deep=True)","metadata":{"papermill":{"duration":20.158522,"end_time":"2020-12-17T09:49:02.319994","exception":false,"start_time":"2020-12-17T09:48:42.161472","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-01-14T07:46:21.591558Z","iopub.execute_input":"2022-01-14T07:46:21.591911Z","iopub.status.idle":"2022-01-14T07:46:39.952106Z","shell.execute_reply.started":"2022-01-14T07:46:21.591881Z","shell.execute_reply":"2022-01-14T07:46:39.950898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Cột prior_question_had_explanation để dạng object tốn nhiều bộ nhớ > chuyển về bool ","metadata":{"papermill":{"duration":0.074831,"end_time":"2020-12-17T09:49:02.640215","exception":false,"start_time":"2020-12-17T09:49:02.565384","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train['prior_question_had_explanation'] = train['prior_question_had_explanation'].astype('boolean')\n\ntrain.memory_usage(deep=True)","metadata":{"papermill":{"duration":32.19594,"end_time":"2020-12-17T09:49:34.911773","exception":false,"start_time":"2020-12-17T09:49:02.715833","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-01-14T07:46:39.953696Z","iopub.execute_input":"2022-01-14T07:46:39.954379Z","iopub.status.idle":"2022-01-14T07:47:07.146282Z","shell.execute_reply.started":"2022-01-14T07:46:39.954334Z","shell.execute_reply":"2022-01-14T07:47:07.145414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-01-14T07:47:07.15208Z","iopub.execute_input":"2022-01-14T07:47:07.154276Z","iopub.status.idle":"2022-01-14T07:47:07.172763Z","shell.execute_reply.started":"2022-01-14T07:47:07.154234Z","shell.execute_reply":"2022-01-14T07:47:07.171758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Tổng bộ nhớ đã giảm đi khoảng 600Mb khi đổi kiểu dữ liệu của prior_question_had_explanation về bool","metadata":{"papermill":{"duration":0.07468,"end_time":"2020-12-17T09:49:35.064678","exception":false,"start_time":"2020-12-17T09:49:34.989998","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time\n\nquestions = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/questions.csv')\nlectures = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/lectures.csv')\nexample_test = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/example_test.csv')\nexample_sample_submission = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/example_sample_submission.csv')","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.132324,"end_time":"2020-12-17T09:49:35.272284","exception":false,"start_time":"2020-12-17T09:49:35.13996","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-01-14T07:47:07.17898Z","iopub.execute_input":"2022-01-14T07:47:07.181269Z","iopub.status.idle":"2022-01-14T07:47:07.257464Z","shell.execute_reply.started":"2022-01-14T07:47:07.181226Z","shell.execute_reply":"2022-01-14T07:47:07.256687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1 Exploring Train\n\nThe columns in the train file are described as:\n* row_id: (int64) ID của row.\n* timestamp: (int64) thời gian tính bằng mili giây giữa lần tương tác của người dùng này và sự kiện đầu tiên hoàn thành từ người dùng đó.\n* user_id: (int32) id người dùng.\n* content_id: (int16) mã ID cho tương tác của người dùng\n* content_type_id: (int8) 0 nếu sự kiện là một câu hỏi được đặt ra cho người dùng, 1 nếu sự kiện là người dùng đang xem một bài giảng.\n* task_container_id: (int16) Mã id cho loạt câu hỏi hoặc bài giảng. Ví dụ: một người dùng có thể thấy ba câu hỏi liên tiếp trước khi xem giải thích cho bất kỳ câu hỏi nào trong số đó. Ba câu đó sẽ chia sẻ một task_container_id.\n* user_answer: (int8) câu trả lời của người dùng cho câu hỏi, nếu có. Đọc -1 là null cho các bài giảng.\n* answer_correctly: (int8) nếu người dùng trả lời đúng. Đọc -1 là null cho các bài giảng.\n* prior_question_elapsed_time: (float32) Thời gian trung bình tính bằng mili giây người dùng trả lời từng câu hỏi trong gói câu hỏi trước đó, bỏ qua bất kỳ bài giảng nào ở giữa. Không có giá trị cho gói câu hỏi hoặc bài giảng đầu tiên của người dùng. Lưu ý rằng thời gian là thời gian trung bình mà người dùng dành để giải quyết từng câu hỏi trong nhóm trước đó.\n* prior_question_had_explanation: (bool) Người dùng có thấy lời giải thích và (các) câu trả lời chính xác hay không sau khi trả lời gói câu hỏi trước đó, bỏ qua bất kỳ bài giảng nào ở giữa. Giá trị được chia sẻ trên một gói câu hỏi và không có giá trị đối với gói câu hỏi hoặc bài giảng đầu tiên của người dùng. Thông thường, một số câu hỏi đầu tiên mà người dùng nhìn thấy là một phần của kiểm tra chẩn đoán tích hợp trong đó họ không nhận được bất kỳ phản hồi nào.\n\n","metadata":{"papermill":{"duration":0.076475,"end_time":"2020-12-17T09:49:35.426088","exception":false,"start_time":"2020-12-17T09:49:35.349613","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(f'Số user: {train.user_id.nunique()}')","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.914477,"end_time":"2020-12-17T09:49:36.603953","exception":false,"start_time":"2020-12-17T09:49:35.689476","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-01-14T07:47:07.261463Z","iopub.execute_input":"2022-01-14T07:47:07.263582Z","iopub.status.idle":"2022-01-14T07:47:07.981341Z","shell.execute_reply.started":"2022-01-14T07:47:07.263541Z","shell.execute_reply":"2022-01-14T07:47:07.980491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Content_type_id = False nếu user trả lời câu hỏi và = True khi user xem lecture.","metadata":{"papermill":{"duration":0.075835,"end_time":"2020-12-17T09:49:36.759379","exception":false,"start_time":"2020-12-17T09:49:36.683544","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train.content_type_id.value_counts()","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.856267,"end_time":"2020-12-17T09:49:37.69222","exception":false,"start_time":"2020-12-17T09:49:36.835953","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-01-14T07:47:07.982645Z","iopub.execute_input":"2022-01-14T07:47:07.983014Z","iopub.status.idle":"2022-01-14T07:47:08.636535Z","shell.execute_reply.started":"2022-01-14T07:47:07.982983Z","shell.execute_reply":"2022-01-14T07:47:08.635837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Content_id là mã đại diện cho tương tác của người dùng, mã này đại diện cho câu hỏi khi mà content_type = False(ám chỉ câu hỏi) và ngược lại là cho lecture","metadata":{"papermill":{"duration":0.081492,"end_time":"2020-12-17T09:49:37.904535","exception":false,"start_time":"2020-12-17T09:49:37.823043","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(f'Có tất cả {train.content_id.nunique()} content_id unique trong train set trong đó content_id là câu hỏi có {train[train.content_type_id == False].content_id.nunique()} row.')","metadata":{"_kg_hide-input":true,"papermill":{"duration":9.014258,"end_time":"2020-12-17T09:49:46.996737","exception":false,"start_time":"2020-12-17T09:49:37.982479","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-01-14T07:47:08.637757Z","iopub.execute_input":"2022-01-14T07:47:08.638104Z","iopub.status.idle":"2022-01-14T07:47:16.721452Z","shell.execute_reply.started":"2022-01-14T07:47:08.638076Z","shell.execute_reply":"2022-01-14T07:47:16.720703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cids = train.content_id.value_counts()[:20]\n\nfig = plt.figure(figsize=(12,6))\nax = cids.plot.bar()\nplt.title(\"Top 20 content_id(câu hỏi) xuất hiện nhiều nhất\")\nplt.xticks(rotation=90)\nplt.xlabel(\"content_id\")\nplt.ylabel(\"count\")\nax.get_yaxis().set_major_formatter(FuncFormatter(lambda x, p: format(int(x), ',')))\nplt.show()","metadata":{"_kg_hide-input":true,"papermill":{"duration":4.30588,"end_time":"2020-12-17T09:49:51.381593","exception":false,"start_time":"2020-12-17T09:49:47.075713","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-01-14T07:47:16.72358Z","iopub.execute_input":"2022-01-14T07:47:16.723955Z","iopub.status.idle":"2022-01-14T07:47:20.30654Z","shell.execute_reply.started":"2022-01-14T07:47:16.723915Z","shell.execute_reply":"2022-01-14T07:47:20.305714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"task_container_id: (int16) id cho loạt câu hỏi hoặc lecture. Ví dụ: một người dùng có thể thấy ba câu hỏi liên tiếp trước khi xem giải thích cho bất kỳ câu hỏi nào trong số đó. Ba câu hỏi đó sẽ chia sẻ một task_container_id .","metadata":{"papermill":{"duration":0.077525,"end_time":"2020-12-17T09:49:51.537513","exception":false,"start_time":"2020-12-17T09:49:51.459988","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(f'Có {train.task_container_id.nunique()} loạt câu hỏi hoặc lecture unique')","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.960111,"end_time":"2020-12-17T09:49:52.575894","exception":false,"start_time":"2020-12-17T09:49:51.615783","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-01-14T07:47:20.307968Z","iopub.execute_input":"2022-01-14T07:47:20.308335Z","iopub.status.idle":"2022-01-14T07:47:21.24676Z","shell.execute_reply.started":"2022-01-14T07:47:20.308296Z","shell.execute_reply":"2022-01-14T07:47:21.245811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"user_answer. Các câu hỏi là trắc nghiệm (đáp án 0-3). Như phần mô tả dữ liệu ban đầu thì -1 không có câu trả lời (khi tương tác là lecture không phải câu hỏi).","metadata":{"papermill":{"duration":0.079066,"end_time":"2020-12-17T09:49:52.73657","exception":false,"start_time":"2020-12-17T09:49:52.657504","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train.user_answer.value_counts()","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.880937,"end_time":"2020-12-17T09:49:53.747823","exception":false,"start_time":"2020-12-17T09:49:52.866886","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-01-14T07:47:21.25078Z","iopub.execute_input":"2022-01-14T07:47:21.252886Z","iopub.status.idle":"2022-01-14T07:47:22.004527Z","shell.execute_reply.started":"2022-01-14T07:47:21.252842Z","shell.execute_reply":"2022-01-14T07:47:22.003832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"timestamp: (int64)thời gian tính bằng mili giây giữa lần tương tác của người dùng đến khi hoàn thành sự kiện đầu tiên từ người dùng đó. Có thể thấy, hầu hết các tương tác là từ những người dùng chưa hoạt động lâu trên hệ thống","metadata":{"papermill":{"duration":0.081355,"end_time":"2020-12-17T09:49:53.909651","exception":false,"start_time":"2020-12-17T09:49:53.828296","status":"completed"},"tags":[]}},{"cell_type":"code","source":"#1 year = 31536000000 ms\nts = train['timestamp']/(31536000000/12)\nfig = plt.figure(figsize=(12,6))\nts.plot.hist(bins=100)\nplt.title(\"Histogram of timestamp(by month)\")\nplt.xticks(rotation=0)\nplt.xlabel(\"Số tháng giữa lần tương tác của người dùng và sự kiện hoàn thành đầu tiên từ người dùng đó \")\nplt.show()","metadata":{"_kg_hide-input":true,"papermill":{"duration":14.35882,"end_time":"2020-12-17T09:50:08.348441","exception":false,"start_time":"2020-12-17T09:49:53.989621","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-01-14T07:47:22.005755Z","iopub.execute_input":"2022-01-14T07:47:22.006102Z","iopub.status.idle":"2022-01-14T07:47:35.488697Z","shell.execute_reply.started":"2022-01-14T07:47:22.006073Z","shell.execute_reply":"2022-01-14T07:47:35.48772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Có {train[train.timestamp == 0].user_id.nunique()}/{train.user_id.nunique()} user có timestamp bắt đầu từ 0.')","metadata":{"_kg_hide-input":true,"papermill":{"duration":1.168462,"end_time":"2020-12-17T09:50:09.761185","exception":false,"start_time":"2020-12-17T09:50:08.592723","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-01-14T07:47:35.490248Z","iopub.execute_input":"2022-01-14T07:47:35.490662Z","iopub.status.idle":"2022-01-14T07:47:36.42222Z","shell.execute_reply.started":"2022-01-14T07:47:35.490617Z","shell.execute_reply":"2022-01-14T07:47:36.42128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Có thể thấy dữ liệu này thể hiện đầy đủ lịch sử tương tác của user với hệ thống, không có trường hợp chen giữa quá trình","metadata":{}},{"cell_type":"markdown","source":"# answered_correctly\nanswered_correctly là mục tiêu để dự đoán, nếu không tính các trường hợp tương tác là xem lecture thì có khoảng 1/3 số câu trả lời là sai","metadata":{"papermill":{"duration":0.082594,"end_time":"2020-12-17T09:50:09.925612","exception":false,"start_time":"2020-12-17T09:50:09.843018","status":"completed"},"tags":[]}},{"cell_type":"code","source":"correct = train[train.answered_correctly != -1].answered_correctly.value_counts(ascending=True)\n\nfig = plt.figure(figsize=(12,4))\ncorrect.plot.barh()\nfor i, v in zip(correct.index, correct.values):\n    plt.text(v, i, '{:,}'.format(v), color='white', fontsize=14, ha='right', va='center')\nplt.title(\"Questions answered correctly\")\nplt.xticks(rotation=0)\nplt.xlabel(\"count\")\nplt.ylabel(\"label\")\nplt.show()","metadata":{"_kg_hide-input":true,"papermill":{"duration":6.761291,"end_time":"2020-12-17T09:50:16.772571","exception":false,"start_time":"2020-12-17T09:50:10.01128","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-01-14T07:47:36.423712Z","iopub.execute_input":"2022-01-14T07:47:36.424095Z","iopub.status.idle":"2022-01-14T07:47:43.587446Z","shell.execute_reply.started":"2022-01-14T07:47:36.424054Z","shell.execute_reply":"2022-01-14T07:47:43.586711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def correct(field):\n    correct = train[train.answered_correctly != -1].groupby([field, 'answered_correctly'], as_index=False).size()\n    correct = correct.pivot(index= field, columns='answered_correctly', values='size')\n    correct['Percent_correct'] = round(correct.iloc[:,1]/(correct.iloc[:,0] + correct.iloc[:,1]),2)\n    correct = correct.sort_values(by = \"Percent_correct\", ascending = False)\n    correct = correct.iloc[:,2]\n    return(correct)","metadata":{"execution":{"iopub.status.busy":"2022-01-14T07:47:43.590467Z","iopub.execute_input":"2022-01-14T07:47:43.590805Z","iopub.status.idle":"2022-01-14T07:47:43.600631Z","shell.execute_reply.started":"2022-01-14T07:47:43.590771Z","shell.execute_reply":"2022-01-14T07:47:43.599889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Những người dùng đã đăng ký tương đối gần đây hoạt động kém hơn một chút so với những người dùng hoạt động lâu hơn. Chia dataset thành 5 bin theo  col timestamp","metadata":{"papermill":{"duration":0.085674,"end_time":"2020-12-17T09:50:16.943726","exception":false,"start_time":"2020-12-17T09:50:16.858052","status":"completed"},"tags":[]}},{"cell_type":"code","source":"bin_labels_5 = ['Bin_1', 'Bin_2', 'Bin_3', 'Bin_4', 'Bin_5']\ntrain['ts_bin'] = pd.qcut(train['timestamp'], q=5, labels=bin_labels_5)\n\nbins_correct = correct(\"ts_bin\")\nbins_correct = bins_correct.sort_index()\n\nfig = plt.figure(figsize=(12,6))\nplt.bar(bins_correct.index, bins_correct.values)\nfor i, v in zip(bins_correct.index, bins_correct.values):\n    plt.text(i, v, v, color='white', fontsize=14, va='top', ha='center')\nplt.title(\"Percent answered_correctly for 5 bins of timestamp\")\nplt.xticks(rotation=0)\nplt.show()","metadata":{"_kg_hide-input":true,"papermill":{"duration":23.77443,"end_time":"2020-12-17T09:50:40.805035","exception":false,"start_time":"2020-12-17T09:50:17.030605","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-01-14T07:47:43.602183Z","iopub.execute_input":"2022-01-14T07:47:43.602633Z","iopub.status.idle":"2022-01-14T07:48:06.380251Z","shell.execute_reply.started":"2022-01-14T07:47:43.602591Z","shell.execute_reply":"2022-01-14T07:48:06.379388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Phân phối của tỷ lệ trả lời đúng câu hỏi theo task_container_id","metadata":{"papermill":{"duration":0.0873,"end_time":"2020-12-17T09:50:40.977397","exception":false,"start_time":"2020-12-17T09:50:40.890097","status":"completed"},"tags":[]}},{"cell_type":"code","source":"task_id_correct = correct(\"task_container_id\")\nfig = plt.figure(figsize=(12,6))\ntask_id_correct.plot.hist(bins=40)\nplt.title(\"Histogram of percent_correct grouped by task_container_id\")\nplt.xticks(rotation=0)\nplt.show()","metadata":{"_kg_hide-input":true,"papermill":{"duration":14.06462,"end_time":"2020-12-17T09:50:55.124815","exception":false,"start_time":"2020-12-17T09:50:41.060195","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-01-14T07:48:06.381651Z","iopub.execute_input":"2022-01-14T07:48:06.382068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user_percent = train[train.answered_correctly != -1].groupby('user_id')['answered_correctly'].agg(Mean='mean', Answers='count')\nsns.boxplot( x=user_percent.Answers);\nplt.show()","metadata":{"_kg_hide-input":true,"papermill":{"duration":20.201349,"end_time":"2020-12-17T09:51:15.580272","exception":false,"start_time":"2020-12-17T09:50:55.378923","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Dựa vào boxplot có thể thấy điểm bắt đầu xuất hiện outlier là từ khoảng 1000 nên chọn chặn trên tại điểm này để biểu diễn với sample=1000\n- Xét biểu đồ tương quan giữa tỷ lệ trả lời đúng và số lượng câu trả lời của user bên dưới, có thể thấy xu hướng tỷ lệ trả lời đúng tăng lên theo số lượng câu trả lời","metadata":{}},{"cell_type":"code","source":"user_percent = user_percent.query('Answers <= 1000').sample(n=1000, random_state=1)\n\nfig = plt.figure(figsize=(12,6))\nx = user_percent.Answers\ny = user_percent.Mean\nplt.scatter(x, y, marker='o')\nplt.title(\"Percent answered correctly vs number of questions answered User\")\nplt.xticks(rotation=0)\nplt.xlabel(\"Number of questions answered\")\nplt.ylabel(\"Percent answered correctly\")\nz = np.polyfit(x, y, 1)\np = np.poly1d(z)\nplt.plot(x,p(x),\"r--\")\n\nplt.show()\n","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.36616,"end_time":"2020-12-17T09:51:16.034808","exception":false,"start_time":"2020-12-17T09:51:15.668648","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"content_percent =  train[train.answered_correctly != -1].groupby('content_id')['answered_correctly'].agg(Mean='mean', Answers='count')\nsns.boxplot( x=content_percent.Answers);\nplt.show()","metadata":{"_kg_hide-input":true,"papermill":{"duration":23.762823,"end_time":"2020-12-17T09:51:40.117255","exception":false,"start_time":"2020-12-17T09:51:16.354432","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Trên tổng số {len(content_percent)} câu hỏi có {len(content_percent[content_percent.Answers > 25000])} được trả lời nhiều hơn 25,000 lần')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Dựa vào boxplot có thể thấy điểm bắt đầu xuất hiện outlier là từ khoảng 25000 nên chọn chặn trên tại điểm này để biểu diễn với sample=1000\n-  Xét biểu đồ tương quan giữa tỷ lệ trả lời đúng và *số lần được trả lời của 1 content_id(câu hỏi)* bên dưới có thể thấy xu hướng tỷ lệ trả lời đúng giảm dần theo số lượng câu trả lời","metadata":{"papermill":{"duration":0.108258,"end_time":"2020-12-17T09:51:16.260382","exception":false,"start_time":"2020-12-17T09:51:16.152124","status":"completed"},"tags":[]}},{"cell_type":"code","source":"content_percent = content_percent.query('Answers <= 25000').sample(n=1000, random_state=42)\n\nfig = plt.figure(figsize=(12,6))\nx = content_percent.Answers\ny = content_percent.Mean\nplt.scatter(x, y, marker='o')\nplt.title(\"Percent answered correctly versus number of questions answered Content_id\")\nplt.xticks(rotation=0)\nplt.xlabel(\"Number of questions answered\")\nplt.ylabel(\"Percent answered correctly\")\nz = np.polyfit(x, y, 1)\np = np.poly1d(z)\nplt.plot(x,p(x),\"r--\")\n\nplt.show()\n","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.289123,"end_time":"2020-12-17T09:51:40.495062","exception":false,"start_time":"2020-12-17T09:51:40.205939","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Được xem các gợi ý trước khi trả lời(prior_question_had_explanation) có tác dụng tới kết quả trả lời câu hỏi hay không? Có thể thấy phần trăm trả lời đúng cao hơn khoảng 17% khi có gợi ý.\n- Ngoài ra, cũng thấy rằng phần trăm được trả lời đúng cho các giá trị còn Nan gần True hơn False. ","metadata":{"papermill":{"duration":0.087556,"end_time":"2020-12-17T09:51:40.670274","exception":false,"start_time":"2020-12-17T09:51:40.582718","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train.prior_question_had_explanation.value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pq = train[train.answered_correctly != -1].groupby(['prior_question_had_explanation'], dropna=False).agg({'answered_correctly': ['mean', 'count']})\n# pq.index = pq.index.astype(str)\nprint(pq)\npq = pq.iloc[:,0]\nfig = plt.figure(figsize=(12,4))\npq.plot.barh()\nplt.title(\"Answered_correctly versus Prior Question had explanation\")\nplt.xlabel(\"Percent answered correctly\")\nplt.ylabel(\"Prior question had explanation\")\nplt.xticks(rotation=0)\nplt.show()","metadata":{"_kg_hide-input":true,"papermill":{"duration":21.748301,"end_time":"2020-12-17T09:52:02.506822","exception":false,"start_time":"2020-12-17T09:51:40.758521","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- prior_question_elapsed_time: Có thể thấy thời gian trung bình để đưa ra câu trả lời của user, có thể thấy không có sự khác biết giữa nhãn 0 và 1( đều ở khaongr 25s)","metadata":{"papermill":{"duration":0.089909,"end_time":"2020-12-17T09:52:02.685987","exception":false,"start_time":"2020-12-17T09:52:02.596078","status":"completed"},"tags":[]}},{"cell_type":"code","source":"pq = train[train.answered_correctly != -1]\npq = pq[['prior_question_elapsed_time', 'answered_correctly']]\npq = pq.groupby(['answered_correctly']).agg({'answered_correctly': ['count'], 'prior_question_elapsed_time': ['mean']})\npq","metadata":{"_kg_hide-input":true,"papermill":{"duration":20.671025,"end_time":"2020-12-17T09:52:23.445971","exception":false,"start_time":"2020-12-17T09:52:02.774946","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Nhưng dựa vào biểu đồ dưới đây với sample = 1000 ta thấy được có 1 xu hướng đi xuống của tỷ lệ trả lời đúng > có thể dùng mean của prior_question_elapsed_time là 1 feature","metadata":{"papermill":{"duration":0.088729,"end_time":"2020-12-17T09:52:23.626325","exception":false,"start_time":"2020-12-17T09:52:23.537596","status":"completed"},"tags":[]}},{"cell_type":"code","source":"mean_pq = train.prior_question_elapsed_time.astype(\"float64\").mean()\n\ncondition = ((train.answered_correctly != -1) & (train.prior_question_elapsed_time.notna()))\npq = train[condition][['prior_question_elapsed_time', 'answered_correctly']].sample(n=1000, random_state=42)\npq = pq.set_index('prior_question_elapsed_time').iloc[:,0]\n\nfig = plt.figure(figsize=(12,6))\nx = pq.index\ny = pq.values\nplt.scatter(x, y, marker='o')\nplt.title(\"Answered_correctly versus prior_question_elapsed_time\")\nplt.xticks(rotation=0)\nplt.xlabel(\"Prior_question_elapsed_time\")\nplt.ylabel(\"Answered_correctly\")\nplt.vlines(mean_pq, ymin=-0.1, ymax=1.1)\nplt.text(x= 27000, y=0.4, s='mean')\nplt.text(x=80000, y=0.6, s='trend')\nz = np.polyfit(x, y, 1)\np = np.poly1d(z)\nplt.plot(x,p(x),\"r--\")\nplt.show()","metadata":{"_kg_hide-input":true,"papermill":{"duration":12.410783,"end_time":"2020-12-17T09:52:36.127151","exception":false,"start_time":"2020-12-17T09:52:23.716368","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2 Exploring Questions\n\n* question_id: khóa ngoại cho cột train/test content_id, khi content_id  là câu hỏi (0).\n* bundle_id: mã mà các câu hỏi được phân phát cùng nhau.\n* correct_answer: câu trả lời cho câu hỏi. Có thể so sánh với cột train user_answer để kiểm tra xem user có đúng hay không.\n* part: phần liên quan của bài thi TOEIC.\n* tags: một hoặc nhiều mã thẻ chi tiết cho câu hỏi. Ý nghĩa của các thẻ sẽ không được cung cấp, nhưng những mã này đủ để nhóm các câu hỏi lại với nhau. \n","metadata":{"papermill":{"duration":0.091554,"end_time":"2020-12-17T09:52:36.313083","exception":false,"start_time":"2020-12-17T09:52:36.221529","status":"completed"},"tags":[]}},{"cell_type":"code","source":"questions.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"questions.head()","metadata":{"papermill":{"duration":0.109456,"end_time":"2020-12-17T09:52:36.514818","exception":false,"start_time":"2020-12-17T09:52:36.405362","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"questions.shape","metadata":{"papermill":{"duration":0.10389,"end_time":"2020-12-17T09:52:36.711367","exception":false,"start_time":"2020-12-17T09:52:36.607477","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"questions[questions.tags.isna()]","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.114798,"end_time":"2020-12-17T09:52:37.105354","exception":false,"start_time":"2020-12-17T09:52:36.990556","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"questions['tags'] = questions['tags'].astype(str)\n\ntags = [x.split() for x in questions[questions.tags != \"nan\"].tags.values]\ntags = [item for elem in tags for item in elem]\ntags = set(tags)\ntags = list(tags)\nprint(f'Có {len(tags)} unique tag')","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.253063,"end_time":"2020-12-17T09:52:39.056844","exception":false,"start_time":"2020-12-17T09:52:38.803781","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Số lượng câu trả lời Đúng và Sai của mỗi câu hỏi","metadata":{"papermill":{"duration":0.094224,"end_time":"2020-12-17T09:52:39.247308","exception":false,"start_time":"2020-12-17T09:52:39.153084","status":"completed"},"tags":[]}},{"cell_type":"code","source":"tags_list = [x.split() for x in questions.tags.values]\nquestions['tags'] = tags_list\nquestions.head()\n\ncorrect = train[train.answered_correctly != -1].groupby([\"content_id\", 'answered_correctly'], as_index=False).size()\ncorrect = correct.pivot(index= \"content_id\", columns='answered_correctly', values='size')\ncorrect.columns = ['Wrong', 'Right']\ncorrect = correct.fillna(0)\ncorrect[['Wrong', 'Right']] = correct[['Wrong', 'Right']].astype(int)\nquestions = questions.merge(correct, left_on = \"question_id\", right_on = \"content_id\", how = \"left\")\nquestions.head()","metadata":{"_kg_hide-input":true,"papermill":{"duration":14.205475,"end_time":"2020-12-17T09:52:53.546413","exception":false,"start_time":"2020-12-17T09:52:39.340938","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntags_df = pd.DataFrame()\nfor x in range(len(tags)):\n    df = questions[questions.tags.apply(lambda l: tags[x] in l)]\n    df1 = df.agg({'Wrong': ['sum'], 'Right': ['sum']})\n    df1['Total_questions'] = df1.Wrong + df1.Right\n    df1['Question_ids_with_tag'] = len(df)\n    df1['tag'] = tags[x]\n    df1 = df1.set_index('tag')\n    tags_df = tags_df.append(df1)\n\ntags_df[['Wrong', 'Right', 'Total_questions']] = tags_df[['Wrong', 'Right', 'Total_questions']].astype(int)\ntags_df['Percent_correct'] = tags_df.Right/tags_df.Total_questions\ntags_df = tags_df.sort_values(by = \"Percent_correct\")\n\ntags_df.head()","metadata":{"papermill":{"duration":2.462101,"end_time":"2020-12-17T09:52:56.688564","exception":false,"start_time":"2020-12-17T09:52:54.226463","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Lấy ngưỡng tỷ lệ trả lời đúng là 0.6 có thể thấy sự phân biệt rõ ràng về tỷ lệ trả lời đúng giữ 10 câu khó và 10 câu dễ nhất trong tập câu hỏi\\","metadata":{"papermill":{"duration":0.099623,"end_time":"2020-12-17T09:52:56.883279","exception":false,"start_time":"2020-12-17T09:52:56.783656","status":"completed"},"tags":[]}},{"cell_type":"code","source":"select_rows = list(range(0,10)) + list(range(178, len(tags_df)))\ntags_select = tags_df.iloc[select_rows,4]\n\nfig = plt.figure(figsize=(12,6))\nx = tags_select.index\ny = tags_select.values\nclrs = ['red' if y < 0.6 else 'green' for y in tags_select.values]\ntags_select.plot.bar(x, y, color=clrs)\nplt.title(\"Ten hardest and ten easiest tags\")\nplt.xlabel(\"Tag\")\nplt.ylabel(\"Percent answers correct of questions with the tag\")\nplt.xticks(rotation=90)\nplt.show()","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.342496,"end_time":"2020-12-17T09:52:57.322357","exception":false,"start_time":"2020-12-17T09:52:56.979861","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Nhưng số lượng câu khó có tỷ lệ trả lời đúng thấp thì chỉ có khoảng 250k câu trả lời, ít hơn nhiều các câu khác.","metadata":{"papermill":{"duration":0.101594,"end_time":"2020-12-17T09:52:57.521122","exception":false,"start_time":"2020-12-17T09:52:57.419528","status":"completed"},"tags":[]}},{"cell_type":"code","source":"tags_select = tags_df.sort_values(by = \"Total_questions\", ascending = False).iloc[:30,:]\ntags_select = tags_select[\"Total_questions\"]\n\nfig = plt.figure(figsize=(12,6))\nax = tags_select.plot.bar()\nplt.title(\"Thirty tags with most questions answered\")\nplt.xticks(rotation=90)\nplt.ticklabel_format(style='plain', axis='y')\nax.get_yaxis().set_major_formatter(FuncFormatter(lambda x, p: format(int(x), ','))) \nplt.show()","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.396702,"end_time":"2020-12-17T09:52:58.015015","exception":false,"start_time":"2020-12-17T09:52:57.618313","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Column \"part\", theo tìm hiểu và dựa vào mô tả dữ liệu thì phần part này liên quan đến bài test TOEIC. Có 200 câu hỏi phải trả lời trong hai giờ ở phần Nghe (khoảng 45 phút, 100 câu hỏi) và Đọc (75 phút, 100 câu hỏi).\n\n * Phần nghe bao gồm Phần 1-4 (Phần Nghe (khoảng 45 phút, 100 câu hỏi)).\n\n * Phần đọc bao gồm Phần 5-7 (Phần Đọc (75 phút, 100 câu hỏi)). ","metadata":{"papermill":{"duration":0.099893,"end_time":"2020-12-17T09:52:58.213189","exception":false,"start_time":"2020-12-17T09:52:58.113296","status":"completed"},"tags":[]}},{"cell_type":"code","source":"fig = plt.figure(figsize=(12,8))\nax1 = fig.add_subplot(211)\nax1 = questions.groupby(\"part\").count()['question_id'].plot.bar()\nplt.title(\"Counts of part\")\nplt.xlabel(\"Part\")\nplt.xticks(rotation=0)\n\npart = questions.groupby('part').agg({'Wrong': ['sum'], 'Right': ['sum']})\npart['Percent_correct'] = part.Right/(part.Right + part.Wrong)\npart = part.iloc[:,2]\n\nax2 = fig.add_subplot(212)\nplt.bar(part.index, part.values)\nfor i, v in zip(part.index, part.values):\n    plt.text(i, v, round(v,2), color='white', fontweight='bold', fontsize=14, va='top', ha='center')\n\nplt.title(\"Percent_correct by part\")\nplt.xlabel(\"Part\")\nplt.xticks(rotation=0)\nplt.tight_layout(pad=2)\nplt.show()","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.483854,"end_time":"2020-12-17T09:52:58.991167","exception":false,"start_time":"2020-12-17T09:52:58.507313","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Dựa vào biểu đồ trên có thể thấy là part 5 có số lượng câu hỏi nhiều nhất và cũng là phần khó nhất","metadata":{"papermill":{"duration":0.097774,"end_time":"2020-12-17T09:52:58.409349","exception":false,"start_time":"2020-12-17T09:52:58.311575","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# 3 Exploring Lectures\n* Lect_id: khóa ngoại cho cột train/test content_id, khi content_id là lecture(1).\n* part: category code for the lecture.\n* tag: mã thẻ cho bài giảng. Ý nghĩa của các thẻ sẽ không được cung cấp, nhưng những mã tag này có thể dùng để nhóm các lecture lại với nhau.\n* type_of: mô tả ngắn gọn về mục đích của lecture\n","metadata":{"papermill":{"duration":0.106104,"end_time":"2020-12-17T09:52:59.197361","exception":false,"start_time":"2020-12-17T09:52:59.091257","status":"completed"},"tags":[]}},{"cell_type":"code","source":"lectures.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lectures.head()","metadata":{"papermill":{"duration":0.115846,"end_time":"2020-12-17T09:52:59.412895","exception":false,"start_time":"2020-12-17T09:52:59.297049","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Có  {lectures.shape[0]} lecture_id.')","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.109159,"end_time":"2020-12-17T09:52:59.622439","exception":false,"start_time":"2020-12-17T09:52:59.51328","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lect_type_of = lectures.type_of.value_counts()\n\nfig = plt.figure(figsize=(12,6))\nplt.bar(lect_type_of.index, lect_type_of.values)\nfor i, v in zip(lect_type_of.index, lect_type_of.values):\n    plt.text(i, v, v, color='black', fontweight='bold', fontsize=14, va='bottom', ha='center')\nplt.title(\"Types of lectures\")\nplt.xlabel(\"type_of\")\nplt.ylabel(\"Count lecture_id\")\nplt.xticks(rotation=0)\nplt.show()","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.264007,"end_time":"2020-12-17T09:53:00.19132","exception":false,"start_time":"2020-12-17T09:52:59.927313","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Các user xem qua lecture có tỷ lệ trả lời đúng cao hơn là không xem, có thể thấy điều này qua biểu đồ dưới đây","metadata":{"papermill":{"duration":0.100957,"end_time":"2020-12-17T09:53:00.394107","exception":false,"start_time":"2020-12-17T09:53:00.29315","status":"completed"},"tags":[]}},{"cell_type":"code","source":"user_lect = train.groupby([\"user_id\", \"answered_correctly\"]).size().unstack()\nuser_lect.columns = ['Lecture', 'Wrong', 'Right']\nuser_lect['Lecture'] = user_lect['Lecture'].fillna(0)\nuser_lect = user_lect.astype('Int64')\nuser_lect['Watches_lecture'] = np.where(user_lect.Lecture > 0, True, False)\n\nwatches_l = user_lect.groupby(\"Watches_lecture\").agg({'Wrong': ['sum'], 'Right': ['sum']})\nprint(user_lect.Watches_lecture.value_counts())\n\nwatches_l['Percent_correct'] = watches_l.Right/(watches_l.Right + watches_l.Wrong)\n\nwatches_l = watches_l.iloc[:,2]\n\nfig = plt.figure(figsize=(12,4))\nwatches_l.plot.barh()\nfor i, v in zip(watches_l.index, watches_l.values):\n    plt.text(v, i, round(v,2), color='white', fontweight='bold', fontsize=14, ha='right', va='center')\n\nplt.title(\"User watches lectures: Percent_correct\")\nplt.xlabel(\"Percent correct\")\nplt.ylabel(\"User watched at least one lecture\")\nplt.xticks(rotation=0)\nplt.show()","metadata":{"_kg_hide-input":true,"papermill":{"duration":6.598951,"end_time":"2020-12-17T09:53:07.09487","exception":false,"start_time":"2020-12-17T09:53:00.495919","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_lect = train.groupby([\"task_container_id\", \"answered_correctly\"]).size().unstack()\nbatch_lect.columns = ['Lecture', 'Wrong', 'Right']\nbatch_lect['Lecture'] = batch_lect['Lecture'].fillna(0)\nbatch_lect = batch_lect.astype('Int64')\nbatch_lect['Percent_correct'] = batch_lect.Right/(batch_lect.Wrong + batch_lect.Right)\nbatch_lect['Percent_lecture'] = batch_lect.Lecture/(batch_lect.Lecture + batch_lect.Wrong + batch_lect.Right)\nbatch_lect = batch_lect.sort_values(by = \"Percent_lecture\", ascending = False)\n\nprint(f'Số lượng bài giảng cao nhất được xem trong một task_container_id: {batch_lect.Lecture.max()}.')","metadata":{"_kg_hide-input":true,"papermill":{"duration":7.565143,"end_time":"2020-12-17T09:53:14.974522","exception":false,"start_time":"2020-12-17T09:53:07.409379","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_lect.head()","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.120948,"end_time":"2020-12-17T09:53:15.412176","exception":false,"start_time":"2020-12-17T09:53:15.291228","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Dựa vào biểu đồ tương quan ở dưới, có thể thấy không có sự tương quan hay quan hệ nào giữ tỷ lệ trả lời đúng với tỷ lệ lecture trong các task_container ","metadata":{"papermill":{"duration":0.10428,"end_time":"2020-12-17T09:53:15.620272","exception":false,"start_time":"2020-12-17T09:53:15.515992","status":"completed"},"tags":[]}},{"cell_type":"code","source":"batch = batch_lect.iloc[:, 3:]\n\nfig = plt.figure(figsize=(12,6))\nx = batch.Percent_lecture\ny = batch.Percent_correct\nplt.scatter(x, y, marker='o')\nplt.title(\"Percent lectures in a task_container versus percent answered correctly\")\nplt.xticks(rotation=0)\nplt.xlabel(\"Percent lectures\")\nplt.ylabel(\"Percent answered correctly\")\n\nplt.show()\n","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.324996,"end_time":"2020-12-17T09:53:16.04921","exception":false,"start_time":"2020-12-17T09:53:15.724214","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_lect['Has_lecture'] = np.where(batch_lect.Lecture == 0, False, True)\nprint(f'{batch_lect[batch_lect.Has_lecture == True].shape[0]} task_container_ids có lecture và {batch_lect[batch_lect.Has_lecture == False].shape[0]} task_container_ids không có lecture.')","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.123677,"end_time":"2020-12-17T09:53:16.500762","exception":false,"start_time":"2020-12-17T09:53:16.377085","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_lect = batch_lect[['Wrong', 'Right', 'Has_lecture']]\nbatch_lect = batch_lect.groupby(\"Has_lecture\").sum()\nbatch_lect['Percent_correct'] = batch_lect.Right/(batch_lect.Wrong + batch_lect.Right)\nbatch_lect = batch_lect[['Percent_correct']]\nbatch_lect","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.128129,"end_time":"2020-12-17T09:53:16.735376","exception":false,"start_time":"2020-12-17T09:53:16.607247","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Khi có lecture trong 1 task_container_id không ảnh hưởng tới kết quả trả lời, các task_container_id không có lecture có câu trả lời đúng nhiều hơn khoảng 8% so với các task_container_id  có bài giảng. ","metadata":{"papermill":{"duration":0.106777,"end_time":"2020-12-17T09:53:16.269922","exception":false,"start_time":"2020-12-17T09:53:16.163145","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# 4 Train pipeline","metadata":{}},{"cell_type":"code","source":"%reset -f","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport riiideducation\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport matplotlib.style as style\nstyle.use('fivethirtyeight')\nimport seaborn as sns\nimport os\nimport lightgbm as lgb\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.preprocessing import LabelEncoder\nimport gc\nimport sys\npd.set_option('display.max_rows', None)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ncols_to_load = ['row_id', 'user_id', 'answered_correctly', 'content_id', 'prior_question_had_explanation', 'prior_question_elapsed_time']\ntrain = pd.read_pickle(\"../input/riid-multi-format-train-set/riiid_train.pkl.gzip\")[cols_to_load]\ntrain['prior_question_had_explanation'] = train['prior_question_had_explanation'].astype('boolean')\n\nprint(\"Train size:\", train.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nquestions = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/questions.csv')\nlectures = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/lectures.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n#adding user features\nuser_df = train[train.answered_correctly != -1].groupby('user_id').agg({'answered_correctly': ['count', 'mean']}).reset_index()\nuser_df.columns = ['user_id', 'user_questions', 'user_mean']\n\nuser_lect = train.groupby([\"user_id\", \"answered_correctly\"]).size().unstack()\nuser_lect.columns = ['Lecture', 'Wrong', 'Right']\nuser_lect = user_lect[['Lecture']].fillna(0).astype('int8')\n#user_lect = user_lect.astype('int8')\nuser_lect['watches_lecture'] = np.where(user_lect.Lecture > 0, 1, 0)\nuser_lect = user_lect.reset_index()\nuser_lect = user_lect[['user_id', 'watches_lecture']]\n\nuser_df = user_df.merge(user_lect, on = \"user_id\", how = \"left\")\ndel user_lect\nuser_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n#adding content features\ncontent_df = train[train.answered_correctly != -1].groupby('content_id').agg({'answered_correctly': ['count', 'mean']}).reset_index()\ncontent_df.columns = ['content_id', 'content_questions', 'content_mean']\ncontent_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Split train/val set sử dụng data của tác giả tito(https://www.kaggle.com/its7171/cv-strategy)","metadata":{}},{"cell_type":"code","source":"%%time\ncv2_train = pd.read_pickle(\"../input/riiid-cross-validation-files/cv2_train.pickle\")['row_id']\ncv2_valid = pd.read_pickle(\"../input/riiid-cross-validation-files/cv2_valid.pickle\")['row_id']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- convert type của prior_question_elapsed_time sang float64 rồi mới lấy mean để có kết quả chính xác, issue được đề cập tại https://www.kaggle.com/c/riiid-test-answer-prediction/discussion/195032","metadata":{}},{"cell_type":"code","source":"train = train[train.answered_correctly != -1]\n\nmean_prior = train.prior_question_elapsed_time.astype(\"float64\").mean()\n\nvalidation = train[train.row_id.isin(cv2_valid)]\ntrain = train[train.row_id.isin(cv2_train)]\n\nvalidation = validation.drop(columns = \"row_id\")\ntrain = train.drop(columns = \"row_id\")\n\ndel cv2_train, cv2_valid\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_enc = LabelEncoder()\n\ntrain = train.merge(user_df, on = \"user_id\", how = \"left\")\ntrain = train.merge(content_df, on = \"content_id\", how = \"left\")\ntrain['content_questions'].fillna(0, inplace = True)\ntrain['content_mean'].fillna(0.5, inplace = True)\ntrain['watches_lecture'].fillna(0, inplace = True)\ntrain['user_questions'].fillna(0, inplace = True)\ntrain['user_mean'].fillna(0.5, inplace = True)\ntrain['prior_question_elapsed_time'].fillna(mean_prior, inplace = True)\ntrain['prior_question_had_explanation'].fillna(False, inplace = True)\nlabel_enc.fit(train['prior_question_had_explanation'])\ntrain['prior_question_had_explanation'] = label_enc.transform(train['prior_question_had_explanation'])\ntrain[['content_questions', 'user_questions']] = train[['content_questions', 'user_questions']].astype(int)\ntrain.sample(5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation = validation.merge(user_df, on = \"user_id\", how = \"left\")\nvalidation = validation.merge(content_df, on = \"content_id\", how = \"left\")\nvalidation['content_questions'].fillna(0, inplace = True)\nvalidation['content_mean'].fillna(0.5, inplace = True)\nvalidation['watches_lecture'].fillna(0, inplace = True)\nvalidation['user_questions'].fillna(0, inplace = True)\nvalidation['user_mean'].fillna(0.5, inplace = True)\nvalidation['prior_question_elapsed_time'].fillna(mean_prior, inplace = True)\nvalidation['prior_question_had_explanation'].fillna(False, inplace = True)\nvalidation['prior_question_had_explanation'] = label_enc.transform(validation['prior_question_had_explanation'])\nvalidation[['content_questions', 'user_questions']] = validation[['content_questions', 'user_questions']].astype(int)\nvalidation.sample(5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# features = ['user_questions', 'user_mean', 'content_questions', 'content_mean', 'watches_lecture',\n#             'prior_question_elapsed_time', 'prior_question_had_explanation']\n\nfeatures = ['user_questions', 'user_mean', 'content_questions', 'content_mean', 'prior_question_elapsed_time']\n\ntrain = train.sample(n=10000000, random_state = 1)\n\ny_train = train['answered_correctly']\ntrain = train[features]\n\ny_val = validation['answered_correctly']\nvalidation = validation[features]\nparams = {'objective': 'binary',\n          'metric': 'auc',\n          'seed': 2020,\n          'learning_rate': 0.1,\n          \"boosting_type\": \"gbdt\" \n         }\nlgb_train = lgb.Dataset(train, y_train, categorical_feature = None)\nlgb_eval = lgb.Dataset(validation, y_val, categorical_feature = None)\ndel train, y_train, validation, y_val\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel = lgb.train(\n    params, lgb_train,\n    valid_sets=[lgb_train, lgb_eval],\n    verbose_eval=50,\n    num_boost_round=10000,\n    early_stopping_rounds=8\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb.plot_importance(model)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"env = riiideducation.make_env()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"iter_test = env.iter_test()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for (test_df, sample_prediction_df) in iter_test:\n    test_df = test_df.merge(user_df, on = \"user_id\", how = \"left\")\n    test_df = test_df.merge(content_df, on = \"content_id\", how = \"left\")\n    test_df['content_questions'].fillna(0, inplace = True)\n    test_df['content_mean'].fillna(0.5, inplace = True)\n    test_df['watches_lecture'].fillna(0, inplace = True)\n    test_df['user_questions'].fillna(0, inplace = True)\n    test_df['user_mean'].fillna(0.5, inplace = True)\n    test_df['prior_question_elapsed_time'].fillna(mean_prior, inplace = True)\n    test_df['prior_question_had_explanation'].fillna(False, inplace = True)\n    test_df['prior_question_had_explanation'] = label_enc.transform(test_df['prior_question_had_explanation'])\n    test_df[['content_questions', 'user_questions']] = test_df[['content_questions', 'user_questions']].astype(int)\n    test_df['answered_correctly'] =  model.predict(test_df)\n    env.predict(test_df.loc[test_df['content_type_id'] == 0, ['row_id', 'answered_correctly']])","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}