{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"papermill":{"duration":0.023295,"end_time":"2022-06-03T21:13:10.412151","exception":false,"start_time":"2022-06-03T21:13:10.388856","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-22T09:53:10.369692Z","iopub.execute_input":"2023-03-22T09:53:10.370612Z","iopub.status.idle":"2023-03-22T09:53:10.402499Z","shell.execute_reply.started":"2023-03-22T09:53:10.370512Z","shell.execute_reply":"2023-03-22T09:53:10.400544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv('../input/student-performance-and-game-play/train_labels.csv')\ntrain_labels['question'] = [(label.split('_')[1][1:]) for label in train_labels['session_id']]\ntrain_labels","metadata":{"execution":{"iopub.status.busy":"2023-03-22T09:53:10.404456Z","iopub.execute_input":"2023-03-22T09:53:10.404835Z","iopub.status.idle":"2023-03-22T09:53:11.132747Z","shell.execute_reply.started":"2023-03-22T09:53:10.404804Z","shell.execute_reply":"2023-03-22T09:53:11.131903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q_mean = train_labels.groupby('question')['correct'].mean()\nq_mean.name = 'q_mean'\nq_mean = q_mean.reset_index()\nq_mean","metadata":{"execution":{"iopub.status.busy":"2023-03-22T09:53:11.137070Z","iopub.execute_input":"2023-03-22T09:53:11.139254Z","iopub.status.idle":"2023-03-22T09:53:11.211004Z","shell.execute_reply.started":"2023-03-22T09:53:11.139203Z","shell.execute_reply":"2023-03-22T09:53:11.209858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"papermill":{"duration":0.036198,"end_time":"2022-06-03T21:13:10.454318","exception":false,"start_time":"2022-06-03T21:13:10.41812","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-22T09:53:11.215421Z","iopub.execute_input":"2023-03-22T09:53:11.217585Z","iopub.status.idle":"2023-03-22T09:53:11.240028Z","shell.execute_reply.started":"2023-03-22T09:53:11.217541Z","shell.execute_reply":"2023-03-22T09:53:11.239011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter = 0\n# The API will deliver two dataframes in this specific order,\n# for every session+level grouping (one group per session for each checkpoint)\nfor (test, sample_submission) in iter_test:\n    if counter == 0:\n        print(sample_submission.head())\n        #print(test.columns)\n    #print(test.shape)\n    \n        \n    ## users make predictions here using the test data\n    sample_submission['question'] = [int(label.split('_')[1][1:]) for label in sample_submission['session_id']]\n    df = sample_submission\n    df.loc[df.question == 1, 'correct'] = 1 \n    df.loc[df.question == 2, 'correct'] = 1 \n    df.loc[df.question == 3, 'correct'] = 1 \n    df.loc[df.question == 4, 'correct'] = 1 \n    df.loc[df.question == 5, 'correct'] = 0 \n    df.loc[df.question == 6, 'correct'] = 1 \n    df.loc[df.question == 7, 'correct'] = 1 \n    df.loc[df.question == 8, 'correct'] = 0 \n    df.loc[df.question == 9, 'correct'] = 1 \n    df.loc[df.question == 10, 'correct'] = 0 \n    df.loc[df.question == 11, 'correct'] = 1 \n    df.loc[df.question == 12, 'correct'] = 1 \n    df.loc[df.question == 13, 'correct'] = 0 \n    df.loc[df.question == 14, 'correct'] = 1 \n    df.loc[df.question == 15, 'correct'] = 0\n    df.loc[df.question == 16, 'correct'] = 1\n    df.loc[df.question == 17, 'correct'] = 1\n    df.loc[df.question == 18, 'correct'] = 1\n\n\n    ## env.predict appends the session+level sample_submission to the overall\n    ## submission\n    env.predict(sample_submission[['session_id', 'correct']])\n    counter += 1","metadata":{"papermill":{"duration":0.337707,"end_time":"2022-06-03T21:13:10.798069","exception":false,"start_time":"2022-06-03T21:13:10.460362","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-22T09:53:11.242779Z","iopub.execute_input":"2023-03-22T09:53:11.243122Z","iopub.status.idle":"2023-03-22T09:53:11.311463Z","shell.execute_reply.started":"2023-03-22T09:53:11.243093Z","shell.execute_reply":"2023-03-22T09:53:11.309926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## the end result is a submission file containing all test session predictions\n! head submission.csv","metadata":{"papermill":{"duration":0.767504,"end_time":"2022-06-03T21:13:11.572788","exception":false,"start_time":"2022-06-03T21:13:10.805284","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-03-22T09:53:11.315753Z","iopub.execute_input":"2023-03-22T09:53:11.319807Z","iopub.status.idle":"2023-03-22T09:53:12.442921Z","shell.execute_reply.started":"2023-03-22T09:53:11.319745Z","shell.execute_reply":"2023-03-22T09:53:12.441281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}