{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport gc","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"dtypes = {\n    \"row_id\": \"int64\",\n    \"timestamp\": \"int64\",\n    \"user_id\": \"int32\",\n    \"content_id\": \"int16\",\n    \"content_type_id\": \"int16\",\n    \"task_container_id\": \"int16\",\n    \"user_answer\": \"int8\",\n    \"answered_correctly\": \"int8\",\n    \"prior_question_elapsed_time\": \"float32\", \n    \"prior_question_had_explanation\": \"boolean\"\n}\n\ndata = pd.read_csv(\"../input/riiid-test-answer-prediction/train.csv\", dtype=dtypes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## prior_question_elapsed_time"},{"metadata":{"trusted":true},"cell_type":"code","source":"temp_ =data.loc[(data.prior_question_elapsed_time.isnull()) & (data.timestamp != 0)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp_.content_type_id.value_counts().plot(kind = 'pie', autopct = '%0.3f')\nplt.title(\"Content Type when timestamp not equal to zero and prior_question_elapsed_time is null\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.distplot(temp_.timestamp)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# temp_.loc[(temp_.content_type_id == 1)]\ndata.loc[(data.timestamp == 0) & (data.prior_question_elapsed_time == 0)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.loc[((data.user_id == 369317294) & (data.prior_question_elapsed_time.isnull())) | ((data.user_id == 369317294) & (data.timestamp == 0))]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp_ =data.loc[data.prior_question_elapsed_time == 0]\nplt.figure(figsize = (8, 20))\nplt.subplot(121)\nplt.subtitles('when prior_question_elapsed_time')\nplt.title(\"Is 0\")\ntemp_.answered_correctly.value_counts().plot(kind = 'pie', autopct = '%0.3f')\nplt.subplot(122)\nplt.title('Is null')\ndata.loc[(data.prior_question_elapsed_time.isnull()) & (data.content_type_id == 0)].answered_correctly.value_counts().plot(kind = 'pie', autopct = '%0.3f')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gc.collect()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"The Distribution of Taget feature (when prior_question_elapsed_time is equal to zero ) is equal to the Distribution of Target Feature (when prior_question_elapsed_time is Nan) so in my perspection it won't affect when we replace Nan value with 0 when content type id = 0 and -1 when content type id = 1"},{"metadata":{},"cell_type":"markdown","source":"## Null Values"},{"metadata":{"trusted":true},"cell_type":"code","source":"prior_q_e_time = data.loc[data.prior_question_elapsed_time.isnull()]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"prior_q_e_time.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data['prior_question_elapsed_time'] = data['prior_question_elapsed_time'].fillna(0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data['prior_question_had_explanation'] = data['prior_question_had_explanation'].fillna(0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data['prior_question_had_explanation'] = data['prior_question_had_explanation'].replace({True : 1, False : 0})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"question_data = pd.read_csv(r'../input/riiid-test-answer-prediction/questions.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"question_data.loc[question_data.tags.isnull()]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":" from collections import Counter\n    \ntags = question_data['tags'].loc[question_data['part'] == 6]\ntags = tags.loc[tags.isnull() == False]\ntags = tags.str.split(' ')\ntags = [j for i in tags.values for j in i]\ntotal_count = len(tags)\ntags = Counter(tags)\nx = list(tags.keys())\ny = np.array(list(tags.values()))\ny = y / total_count","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(15, 15))\nplt.barh(x, y)\nplt.vlines(x=0.5, ymin = '179', ymax = len(tags), color = 'red',  linestyles = 'dashed')\nplt.title('Count of Invidual tags in part - 6')\nplt.xlabel('Count')\nplt.ylabel('Tag')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.loc[data['content_id'] == 10033]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp_ = data.iloc[62750278 - 5 :62750278 + 5, :]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp_.merge(question_data, left_on = 'content_id', right_on = 'question_id')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.loc[(data.user_id == 1333688829) & ((data.task_container_id == 1126) | (data.task_container_id == 1128))].merge(question_data, left_on = 'content_id', right_on = 'question_id')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp_ = data.loc[(data.user_id == 1333688829) & (data.content_type_id == 0)].merge(question_data, left_on = 'content_id', right_on = 'question_id')\ntags = temp_['tags'].loc[temp_['part'] == 6]\ntags = tags.loc[tags.isnull() == False]\ntags = tags.str.split(' ')\ntags = [j for i in tags.values for j in i]\ntotal_count = len(tags)\ntags = Counter(tags)\nx1 = list(tags.keys())\ny1 = np.array(list(tags.values()))\ny1 = y1 / total_count","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"answer = temp_['answered_correctly'].loc[temp_.part == 6]\nplt.pie(answer.value_counts(), labels = [1, 0], autopct = '%0.2f')\nplt.legend()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lecture_data = pd.read_csv('../input/riiid-test-answer-prediction/lectures.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lecture_data.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp_ = question_data.merge(lecture_data, left_on = 'question_id', right_on = 'lecture_id', how = 'outer')\ntemp_.loc[temp_.question_id == 10033]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp_.loc[temp_.lecture_id.isnull() == False].head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"There is only one nan value in tags columns of the question dataset. To find the best value of in place of nan tag done some comparition with train andlecture dataset. Such question was asked only once in the whole train dataset and it was asked inbetween part 5 question but the nan tag question is belong to part - 6.\n\nQuestions that asked before and after nan tag question is of tag \"8\" from that we can suspect that it may belongs to tag 8 but there is no evidence that it is.\n\nI check wheather the question and lecture dataset that has same id also have same tags, but it is not same.\n\nIn other hand, when i check the distribution of the tag part - 6 question that asked to the user_id \"1333688829\" is almost same as the distribution of part-6 questions in question \ndataset."},{"metadata":{},"cell_type":"markdown","source":"Main Problem wheater we want to remove the nan tag question from both train and question dataset or we want to mark it with -1.\n\nIn my perspective i decided to mark the tag as -1."},{"metadata":{"trusted":true},"cell_type":"code","source":"question_data.tags = question_data.tags.fillna(-1)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Anlysing the First Interaction of each User"},{"metadata":{"trusted":true},"cell_type":"code","source":"first_interaction = data.groupby('user_id').first()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize = (12, 20))\nax  = plt.subplot(121)\nfirst_interaction.answered_correctly.value_counts().plot(kind = 'pie', ax = ax, autopct='%1.3f%%', startangle=270, fontsize=17)\nax = plt.subplot(122)\nfirst_interaction.content_type_id.value_counts().plot(kind = 'pie', ax = ax, autopct='%1.3f%%', startangle=270, fontsize=17)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"34% unique content_id's are used"},{"metadata":{},"cell_type":"markdown","source":"Work on progress...."}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}