{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# this code is modification of the original code from \n# https://www.kaggle.com/code/cpmpml/random-submission/\n# when i read it i got confused of how those 18 values are assigned. Then i realize it is \n# from the minimum mean value of 18 questions. \n# therefore I modified the last section to make it clear","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"papermill":{"duration":0.023295,"end_time":"2022-06-03T21:13:10.412151","exception":false,"start_time":"2022-06-03T21:13:10.388856","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T21:58:45.146253Z","iopub.execute_input":"2023-06-18T21:58:45.146585Z","iopub.status.idle":"2023-06-18T21:58:45.150276Z","shell.execute_reply.started":"2023-06-18T21:58:45.146558Z","shell.execute_reply":"2023-06-18T21:58:45.149603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv('../input/student-performance-and-game-play/train_labels.csv')\ntrain_labels['question'] = [(label.split('_')[1][1:]) for label in train_labels['session_id']]\nq_mean = train_labels.groupby('question').correct.mean()\nq_mean.name = 'q_mean'\nq_mean = q_mean.reset_index()","metadata":{"execution":{"iopub.status.busy":"2023-06-18T21:58:46.903200Z","iopub.execute_input":"2023-06-18T21:58:46.903553Z","iopub.status.idle":"2023-06-18T21:58:47.514957Z","shell.execute_reply.started":"2023-06-18T21:58:46.903526Z","shell.execute_reply":"2023-06-18T21:58:47.513745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"papermill":{"duration":0.036198,"end_time":"2022-06-03T21:13:10.454318","exception":false,"start_time":"2022-06-03T21:13:10.41812","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T21:58:48.859832Z","iopub.execute_input":"2023-06-18T21:58:48.860205Z","iopub.status.idle":"2023-06-18T21:58:48.886785Z","shell.execute_reply.started":"2023-06-18T21:58:48.860177Z","shell.execute_reply":"2023-06-18T21:58:48.885334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter = 0\n# The API will deliver two dataframes in this specific order,\n# for every session+level grouping (one group per session for each checkpoint)\nfor (test, sample_submission) in iter_test:\n    if counter == 0:\n        print(sample_submission.head())\n        #print(test.columns)\n    #print(test.shape)\n    \n    ## users make predictions here using the test data\n    sample_submission['question'] = [int(\n        label.split('_')[1][1:]) for label in sample_submission['session_id']]\n    \n    \n   \n    # Calculate the mean values for each question\n    mean_values = sample_submission.groupby('question').correct.mean() \n    \n    \n    # Determine the threshold for the lowest mean value, use minimum 5 values\n    minimum_number = 5\n    threshold = mean_values.nsmallest(minimum_number).mean()\n    \n    # Create a dictionary to map question numbers to correct values\n    question_mapping = {question: 1 if mean >= threshold else 0 for question, mean in mean_values.items()}\n    \n    df = sample_submission\n    \n    # Assign correct values based on the question_mapping dictionary\n    df['correct'] = df['question'].map(question_mapping)\n    \n\n    ## env.predict appends the session+level sample_submission to the overall\n    ## submission\n    env.predict(sample_submission[['session_id', 'correct']])\n    counter += 1","metadata":{"papermill":{"duration":0.337707,"end_time":"2022-06-03T21:13:10.798069","exception":false,"start_time":"2022-06-03T21:13:10.460362","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T22:08:53.419191Z","iopub.execute_input":"2023-06-18T22:08:53.419548Z","iopub.status.idle":"2023-06-18T22:08:53.494600Z","shell.execute_reply.started":"2023-06-18T22:08:53.419518Z","shell.execute_reply":"2023-06-18T22:08:53.493437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## the end result is a submission file containing all test session predictions\n! head submission.csv","metadata":{"papermill":{"duration":0.767504,"end_time":"2022-06-03T21:13:11.572788","exception":false,"start_time":"2022-06-03T21:13:10.805284","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T22:09:03.237927Z","iopub.execute_input":"2023-06-18T22:09:03.238242Z","iopub.status.idle":"2023-06-18T22:09:03.509608Z","shell.execute_reply.started":"2023-06-18T22:09:03.238216Z","shell.execute_reply":"2023-06-18T22:09:03.508497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}