{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"ROOT = \"/kaggle/input/riiid-test-answer-prediction\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls {ROOT}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"questions = pd.read_csv(f\"{ROOT}/questions.csv\")\nlectures = pd.read_csv(f\"{ROOT}/lectures.csv\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Question Tags One hot encode"},{"metadata":{"trusted":true},"cell_type":"code","source":"tags_lst = []\nfor tags in questions[~questions[\"tags\"].isnull()][\"tags\"]:\n    tags_lst += tags.split()\n    \ntags_lst = list(map(int, set(tags_lst)))\ntags_lst.sort()\n\nlen(tags_lst), max(tags_lst)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lst = []\nfor question_id, tags in questions[[\"question_id\", \"tags\"]].values:\n    ohe = np.zeros(188)\n    if str(tags) != \"nan\":\n        for tag in tags.split():\n            ohe += np.eye(188)[int(tag)]\n    \n    lst.append([question_id]+ohe.tolist())\ntags_df = pd.DataFrame(lst, columns=[\"question_id\"]+[f\"q_tag_{i}\" for i in range(188)]).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tags_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tags_df.to_csv(\"question_tags_ohe.csv\", index=None)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Lectures Tag One hot encode"},{"metadata":{"trusted":true},"cell_type":"code","source":"type_of_dic = {v: i for i, v in enumerate(lectures[\"type_of\"].unique())}\ntype_of_dic","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lst = []\nfor lecture_id, tag, type_of in lectures[[\"lecture_id\", \"tag\", \"type_of\"]].values:\n    row = [lecture_id, type_of_dic[type_of]] + np.eye(188)[tag].tolist()\n    lst.append(row)\nl_tag_df = pd.DataFrame(lst, columns=[\"lecture_id\", \"type_of\"] + [f\"l_tag_{i}\" for i in range(188)]).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"l_tag_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"l_tag_df.to_csv(\"lecture_tags_ohe.csv\", index=None)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}