{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\n    '/kaggle/input/riiid-test-answer-prediction/train.csv', \n    low_memory=False, \n    nrows=10**7, \n    dtype={\n        'row_id': 'int64', 'timestamp': 'int64', 'user_id': 'int32', 'content_id': 'int16', 'content_type_id': 'int8',\n        'task_container_id': 'int16', 'user_answer': 'int8', 'answered_correctly': 'int8', 'prior_question_elapsed_time': 'float32', \n        'prior_question_had_explanation': 'boolean'\n    }\n)\ntrain_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# EXPLORATION OF THE questions & lectures datasets\nI will start by an explaratory analysis for the question Csv file. we will explore the columns and how we can get extract informations from them.\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"questions = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/questions.csv')\nquestions","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"questions['correct_answer'].hist() ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"questions['part'].hist() ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df1 = questions.groupby(\"tags\",as_index=False)[\"question_id\"].count() \ndf1 = df1.sort_values(by='question_id', ascending=False)\ndf1.head(15)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df2 = questions.groupby(\"part\",as_index=False)[\"question_id\"].count() \ndf2 = df2.sort_values(by='question_id', ascending=False)\ndf2.head(15)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df3 = questions.groupby(\"correct_answer\",as_index=False)[\"tags\"].count() \ndf3 = df3.sort_values(by='tags', ascending=False)\ndf3.head(15)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df4 = questions.groupby(\"correct_answer\",as_index=False)[\"question_id\"].count() \ndf4 = df4.sort_values(by='question_id', ascending=False)\ndf4.head(15)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/lectures.csv')\nlectures","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df5 = lectures.groupby(\"part\",as_index=False)[\"lecture_id\"].count() \ndf5 = df5.sort_values(by='lecture_id', ascending=False)\ndf5.head(15)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df6 = lectures.groupby(\"tag\",as_index=False)[\"lecture_id\"].count() \ndf6 = df6.sort_values(by=\"lecture_id\", ascending=False)\ndf6.head(15)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# TRAIN EDA"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.describe(include=\"all\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for col in train_df:\n    print(col,len(train_df[col].unique()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.corr().style.background_gradient(cmap='coolwarm')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* there is a negative correaltion btween answering correctly and the content type id variable. corr = -0.4  \n* there is a positive correlation between the timestamp and the task container id. corr = 0.4 \n"},{"metadata":{},"cell_type":"markdown","source":"join the two dataframes : questions and lectures "},{"metadata":{"trusted":true},"cell_type":"code","source":"df_row = pd.concat([questions, lectures])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_row","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_row['correct_answer'].value_counts() ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_row.corr().style.background_gradient(cmap='coolwarm')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# ****WORK IN PROGRESS**"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}