{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport datetime as datetime\nimport warnings\nwarnings.filterwarnings('ignore')\n\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error  \nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.preprocessing import StandardScaler","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:56:37.944518Z","iopub.execute_input":"2023-06-02T15:56:37.945026Z","iopub.status.idle":"2023-06-02T15:56:39.452493Z","shell.execute_reply.started":"2023-06-02T15:56:37.944998Z","shell.execute_reply":"2023-06-02T15:56:39.451490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\ndataset_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\nprint(\"Full train dataset shape is {}\".format(dataset_df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:56:39.454092Z","iopub.execute_input":"2023-06-02T15:56:39.454379Z","iopub.status.idle":"2023-06-02T15:58:43.946121Z","shell.execute_reply.started":"2023-06-02T15:56:39.454354Z","shell.execute_reply":"2023-06-02T15:58:43.945236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=dataset_df.copy()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:43.947602Z","iopub.execute_input":"2023-06-02T15:58:43.948106Z","iopub.status.idle":"2023-06-02T15:58:44.589689Z","shell.execute_reply.started":"2023-06-02T15:58:43.948077Z","shell.execute_reply":"2023-06-02T15:58:44.588801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_set = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:44.591854Z","iopub.execute_input":"2023-06-02T15:58:44.592329Z","iopub.status.idle":"2023-06-02T15:58:44.629504Z","shell.execute_reply.started":"2023-06-02T15:58:44.592302Z","shell.execute_reply":"2023-06-02T15:58:44.628687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train=train_set.copy()\ntest=test_set.copy()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:44.630802Z","iopub.execute_input":"2023-06-02T15:58:44.631305Z","iopub.status.idle":"2023-06-02T15:58:44.634928Z","shell.execute_reply.started":"2023-06-02T15:58:44.631276Z","shell.execute_reply":"2023-06-02T15:58:44.634133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_label_set=pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:44.636141Z","iopub.execute_input":"2023-06-02T15:58:44.636616Z","iopub.status.idle":"2023-06-02T15:58:45.023134Z","shell.execute_reply.started":"2023-06-02T15:58:44.636589Z","shell.execute_reply":"2023-06-02T15:58:45.022199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels=train_label_set.copy()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:45.024297Z","iopub.execute_input":"2023-06-02T15:58:45.026921Z","iopub.status.idle":"2023-06-02T15:58:45.036292Z","shell.execute_reply.started":"2023-06-02T15:58:45.026886Z","shell.execute_reply":"2023-06-02T15:58:45.035223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:45.037558Z","iopub.execute_input":"2023-06-02T15:58:45.037909Z","iopub.status.idle":"2023-06-02T15:58:45.090603Z","shell.execute_reply.started":"2023-06-02T15:58:45.037880Z","shell.execute_reply":"2023-06-02T15:58:45.089803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.level_group.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:45.091644Z","iopub.execute_input":"2023-06-02T15:58:45.092394Z","iopub.status.idle":"2023-06-02T15:58:45.304196Z","shell.execute_reply.started":"2023-06-02T15:58:45.092363Z","shell.execute_reply":"2023-06-02T15:58:45.303408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.level.value_counts().sort_index()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:45.308350Z","iopub.execute_input":"2023-06-02T15:58:45.308663Z","iopub.status.idle":"2023-06-02T15:58:45.524511Z","shell.execute_reply.started":"2023-06-02T15:58:45.308638Z","shell.execute_reply":"2023-06-02T15:58:45.523583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[['name']].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:45.525960Z","iopub.execute_input":"2023-06-02T15:58:45.526280Z","iopub.status.idle":"2023-06-02T15:58:45.853852Z","shell.execute_reply.started":"2023-06-02T15:58:45.526254Z","shell.execute_reply":"2023-06-02T15:58:45.852902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.event_name.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:45.855011Z","iopub.execute_input":"2023-06-02T15:58:45.855318Z","iopub.status.idle":"2023-06-02T15:58:46.036561Z","shell.execute_reply.started":"2023-06-02T15:58:45.855292Z","shell.execute_reply":"2023-06-02T15:58:46.035607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:46.037734Z","iopub.execute_input":"2023-06-02T15:58:46.038103Z","iopub.status.idle":"2023-06-02T15:58:46.065810Z","shell.execute_reply.started":"2023-06-02T15:58:46.038078Z","shell.execute_reply":"2023-06-02T15:58:46.064834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### train.index.max()\n- 26296945","metadata":{}},{"cell_type":"code","source":"train.loc[26296945]","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:46.067205Z","iopub.execute_input":"2023-06-02T15:58:46.067510Z","iopub.status.idle":"2023-06-02T15:58:46.077107Z","shell.execute_reply.started":"2023-06-02T15:58:46.067485Z","shell.execute_reply":"2023-06-02T15:58:46.076209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#train-test concatination\ndf= pd.concat((train,test),ignore_index=True,axis=0)","metadata":{}},{"cell_type":"code","source":"train.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:46.078492Z","iopub.execute_input":"2023-06-02T15:58:46.078777Z","iopub.status.idle":"2023-06-02T15:58:46.111679Z","shell.execute_reply.started":"2023-06-02T15:58:46.078752Z","shell.execute_reply":"2023-06-02T15:58:46.110644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.room_fqid.apply(lambda x: x[  x.find('.')+1   :    ((x[x.find('.')+1:].find('.')))+ x.find('.')+1   ]).value_counts()\ntest.room_fqid.apply(lambda x: x[  x.find('.')+1   :    ((x[x.find('.')+1:].find('.')))+ x.find('.')+1   ]).value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:46.113131Z","iopub.execute_input":"2023-06-02T15:58:46.113486Z","iopub.status.idle":"2023-06-02T15:58:51.250907Z","shell.execute_reply.started":"2023-06-02T15:58:46.113457Z","shell.execute_reply":"2023-06-02T15:58:51.249885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.room_fqid = train.room_fqid.apply(lambda x: x[  x.find('.')+1   :    ((x[x.find('.')+1:].find('.')))+ x.find('.')+1   ])\ntest.room_fqid = test.room_fqid.apply(lambda x: x[  x.find('.')+1   :    ((x[x.find('.')+1:].find('.')))+ x.find('.')+1   ])\n","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:51.252512Z","iopub.execute_input":"2023-06-02T15:58:51.252839Z","iopub.status.idle":"2023-06-02T15:58:52.889358Z","shell.execute_reply.started":"2023-06-02T15:58:51.252812Z","shell.execute_reply":"2023-06-02T15:58:52.888432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['click_event']= train.event_name.apply(lambda x: 1 if 'click' in x   else 0)\ntest['click_event']= test.event_name.apply(lambda x: 1 if 'click' in x   else 0)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:52.890668Z","iopub.execute_input":"2023-06-02T15:58:52.891366Z","iopub.status.idle":"2023-06-02T15:58:53.317462Z","shell.execute_reply.started":"2023-06-02T15:58:52.891335Z","shell.execute_reply":"2023-06-02T15:58:53.316462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:53.318717Z","iopub.execute_input":"2023-06-02T15:58:53.319015Z","iopub.status.idle":"2023-06-02T15:58:53.325528Z","shell.execute_reply.started":"2023-06-02T15:58:53.318990Z","shell.execute_reply":"2023-06-02T15:58:53.324583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:53.326793Z","iopub.execute_input":"2023-06-02T15:58:53.327096Z","iopub.status.idle":"2023-06-02T15:58:53.344379Z","shell.execute_reply.started":"2023-06-02T15:58:53.327071Z","shell.execute_reply":"2023-06-02T15:58:53.343329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels['q']=train_labels.session_id.apply(lambda x: x[x.find('_')+1:])","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:53.345787Z","iopub.execute_input":"2023-06-02T15:58:53.346247Z","iopub.status.idle":"2023-06-02T15:58:53.605123Z","shell.execute_reply.started":"2023-06-02T15:58:53.346218Z","shell.execute_reply":"2023-06-02T15:58:53.604185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.groupby(['q'])","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:53.606486Z","iopub.execute_input":"2023-06-02T15:58:53.606787Z","iopub.status.idle":"2023-06-02T15:58:53.613591Z","shell.execute_reply.started":"2023-06-02T15:58:53.606761Z","shell.execute_reply":"2023-06-02T15:58:53.612603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\nq_set= pd.get_dummies(train_labels['q'])","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:53.614895Z","iopub.execute_input":"2023-06-02T15:58:53.615204Z","iopub.status.idle":"2023-06-02T15:58:53.707308Z","shell.execute_reply.started":"2023-06-02T15:58:53.615179Z","shell.execute_reply":"2023-06-02T15:58:53.706285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scapper={1:'Yes',0:'No'}","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:53.708728Z","iopub.execute_input":"2023-06-02T15:58:53.709032Z","iopub.status.idle":"2023-06-02T15:58:53.713059Z","shell.execute_reply.started":"2023-06-02T15:58:53.709006Z","shell.execute_reply":"2023-06-02T15:58:53.712168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q_set.replace(scapper,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:53.714423Z","iopub.execute_input":"2023-06-02T15:58:53.714735Z","iopub.status.idle":"2023-06-02T15:58:55.843368Z","shell.execute_reply.started":"2023-06-02T15:58:53.714711Z","shell.execute_reply":"2023-06-02T15:58:55.842464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.session_id.apply(lambda x: x[:x.find('_'):]).drop_duplicates()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:55.844591Z","iopub.execute_input":"2023-06-02T15:58:55.845230Z","iopub.status.idle":"2023-06-02T15:58:56.190822Z","shell.execute_reply.started":"2023-06-02T15:58:55.845200Z","shell.execute_reply":"2023-06-02T15:58:56.189754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q_set.columns.sort_values()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:56.192212Z","iopub.execute_input":"2023-06-02T15:58:56.192543Z","iopub.status.idle":"2023-06-02T15:58:56.198835Z","shell.execute_reply.started":"2023-06-02T15:58:56.192516Z","shell.execute_reply":"2023-06-02T15:58:56.198045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = ['q1', 'q10', 'q11', 'q12', 'q13', 'q14', 'q15', 'q16', 'q17', 'q18', 'q2', 'q3', 'q4', 'q5', 'q6', 'q7', 'q8', 'q9']\n\nsorted_data = sorted(data, key=lambda x: int(x[1:]))\n\nprint(sorted_data)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:56.207012Z","iopub.execute_input":"2023-06-02T15:58:56.207759Z","iopub.status.idle":"2023-06-02T15:58:56.214853Z","shell.execute_reply.started":"2023-06-02T15:58:56.207720Z","shell.execute_reply":"2023-06-02T15:58:56.213772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target=pd.DataFrame(columns=['session_id','q1', 'q2', 'q3', 'q4', 'q5', 'q6', 'q7', 'q8', 'q9', 'q10', 'q11', 'q12', 'q13', 'q14', 'q15', 'q16', 'q17', 'q18'])","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:56.216283Z","iopub.execute_input":"2023-06-02T15:58:56.216588Z","iopub.status.idle":"2023-06-02T15:58:56.230553Z","shell.execute_reply.started":"2023-06-02T15:58:56.216563Z","shell.execute_reply":"2023-06-02T15:58:56.229560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target.session_id = train_labels.session_id.apply(lambda x: x[:x.find('_'):]).drop_duplicates()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:56.231856Z","iopub.execute_input":"2023-06-02T15:58:56.232410Z","iopub.status.idle":"2023-06-02T15:58:56.584723Z","shell.execute_reply.started":"2023-06-02T15:58:56.232373Z","shell.execute_reply":"2023-06-02T15:58:56.583900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:56.585964Z","iopub.execute_input":"2023-06-02T15:58:56.586447Z","iopub.status.idle":"2023-06-02T15:58:56.607707Z","shell.execute_reply.started":"2023-06-02T15:58:56.586412Z","shell.execute_reply":"2023-06-02T15:58:56.606846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels[train_labels.session_id==str(f'20090312431273200_q{1}')]","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:56.608987Z","iopub.execute_input":"2023-06-02T15:58:56.610012Z","iopub.status.idle":"2023-06-02T15:58:56.723659Z","shell.execute_reply.started":"2023-06-02T15:58:56.609983Z","shell.execute_reply":"2023-06-02T15:58:56.722797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels[train_labels.q=='q1'].correct","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:56.724856Z","iopub.execute_input":"2023-06-02T15:58:56.725806Z","iopub.status.idle":"2023-06-02T15:58:56.808196Z","shell.execute_reply.started":"2023-06-02T15:58:56.725777Z","shell.execute_reply":"2023-06-02T15:58:56.806968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:56.809925Z","iopub.execute_input":"2023-06-02T15:58:56.810938Z","iopub.status.idle":"2023-06-02T15:58:56.837329Z","shell.execute_reply.started":"2023-06-02T15:58:56.810906Z","shell.execute_reply":"2023-06-02T15:58:56.836296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target.q1=train_labels[train_labels.q=='q1'].correct","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:56.838665Z","iopub.execute_input":"2023-06-02T15:58:56.839010Z","iopub.status.idle":"2023-06-02T15:58:56.923476Z","shell.execute_reply.started":"2023-06-02T15:58:56.838981Z","shell.execute_reply":"2023-06-02T15:58:56.922375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(1,19):\n    target[f'q{i}']=train_labels[train_labels.q==f'q{i}'].correct.values","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:56.924729Z","iopub.execute_input":"2023-06-02T15:58:56.925051Z","iopub.status.idle":"2023-06-02T15:58:58.282967Z","shell.execute_reply.started":"2023-06-02T15:58:56.925014Z","shell.execute_reply":"2023-06-02T15:58:58.281815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:58.284296Z","iopub.execute_input":"2023-06-02T15:58:58.284604Z","iopub.status.idle":"2023-06-02T15:58:58.290827Z","shell.execute_reply.started":"2023-06-02T15:58:58.284578Z","shell.execute_reply":"2023-06-02T15:58:58.289885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:58.292248Z","iopub.execute_input":"2023-06-02T15:58:58.292602Z","iopub.status.idle":"2023-06-02T15:58:58.306142Z","shell.execute_reply.started":"2023-06-02T15:58:58.292573Z","shell.execute_reply":"2023-06-02T15:58:58.304858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The following are the descriptions of the fields:\n\n- `session_id`: The ID of the session the event took place in\n- `index`: The index of the event for the session\n- `elapsed_time`: How much time has passed (in milliseconds) between the start of the session and when the event was recorded\n- `event_name`: The name of the event type\n- `name`: The event name (e.g., identifies whether a `notebook_click` is opening or closing the notebook)\n- `level`: What level of the game the event occurred in (0 to 22)\n- `page`: The page number of the event (only for notebook-related events)\n- `room_coor_x`: The coordinates of the click in reference to the in-game room (only for click events)\n- `room_coor_y`: The coordinates of the click in reference to the in-game room (only for click events)\n- `screen_coor_x`: The coordinates of the click in reference to the player’s screen (only for click events)\n- `screen_coor_y`: The coordinates of the click in reference to the player’s screen (only for click events)\n- `hover_duration`: How long (in milliseconds) the hover happened for (only for hover events)\n- `text`: The text the player sees during this event\n- `fqid`: The fully qualified ID of the event\n- `room_fqid`: The fully qualified ID of the room the event took place in\n- `text_fqid`: The fully qualified ID of the text\n- `fullscreen`: Whether the player is in fullscreen mode\n- `hq`: Whether the game is in high-quality\n- `music`: Whether the game music is on or off\n- `level_group`: Which group of levels - and group of questions - this row belongs to (0-4, 5-12, 13-22)\n","metadata":{}},{"cell_type":"code","source":"train[train.level_group=='13-22' ][['elapsed_time','session_id']].groupby(['session_id'])[['elapsed_time']].max()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:58:58.307396Z","iopub.execute_input":"2023-06-02T15:58:58.307738Z","iopub.status.idle":"2023-06-02T15:59:01.075864Z","shell.execute_reply.started":"2023-06-02T15:58:58.307709Z","shell.execute_reply":"2023-06-02T15:59:01.074811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:01.077192Z","iopub.execute_input":"2023-06-02T15:59:01.077492Z","iopub.status.idle":"2023-06-02T15:59:01.085069Z","shell.execute_reply.started":"2023-06-02T15:59:01.077467Z","shell.execute_reply":"2023-06-02T15:59:01.084120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"train = train[train['level'] < 19]\ntrain['level_group'].replace({'13-22':'13-18'},inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T17:26:50.915290Z","iopub.execute_input":"2023-05-29T17:26:50.915625Z"}}},{"cell_type":"code","source":"train.tail()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:01.086425Z","iopub.execute_input":"2023-06-02T15:59:01.086768Z","iopub.status.idle":"2023-06-02T15:59:01.120127Z","shell.execute_reply.started":"2023-06-02T15:59:01.086741Z","shell.execute_reply":"2023-06-02T15:59:01.119137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#level 0-post game start\ntrain[train.level==0][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:01.121551Z","iopub.execute_input":"2023-06-02T15:59:01.121851Z","iopub.status.idle":"2023-06-02T15:59:01.265811Z","shell.execute_reply.started":"2023-06-02T15:59:01.121826Z","shell.execute_reply":"2023-06-02T15:59:01.264775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target['Post_start_duration']=train[train.level==0][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max().values","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:01.267258Z","iopub.execute_input":"2023-06-02T15:59:01.267565Z","iopub.status.idle":"2023-06-02T15:59:01.405733Z","shell.execute_reply.started":"2023-06-02T15:59:01.267540Z","shell.execute_reply":"2023-06-02T15:59:01.404532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(train[train.level_group=='0-4'][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max()-target['Post_start_duration'].values).values","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:01.407127Z","iopub.execute_input":"2023-06-02T15:59:01.407464Z","iopub.status.idle":"2023-06-02T15:59:02.009475Z","shell.execute_reply.started":"2023-06-02T15:59:01.407438Z","shell.execute_reply":"2023-06-02T15:59:02.008448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for inter in ['0-4','5-12','13-18']   :\n    if inter!='0-4':\n        target[f'{inter}_elapsed_time']= train[train.level_group==inter][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max()\n    else:\n        target[f'1-4_elapsed_time']= (train[train.level_group==inter][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max() - target['Post_start_duration'].values).values","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:02.010786Z","iopub.execute_input":"2023-06-02T15:59:02.011094Z","iopub.status.idle":"2023-06-02T15:59:03.841137Z","shell.execute_reply.started":"2023-06-02T15:59:02.011068Z","shell.execute_reply":"2023-06-02T15:59:03.839926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- When making predection make sur to make 3 multiclass predections in order to add corresponding columns 0-4, 5-12, 13-22","metadata":{}},{"cell_type":"code","source":"target.columns","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:03.842352Z","iopub.execute_input":"2023-06-02T15:59:03.842839Z","iopub.status.idle":"2023-06-02T15:59:03.849722Z","shell.execute_reply.started":"2023-06-02T15:59:03.842810Z","shell.execute_reply":"2023-06-02T15:59:03.848948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"target=target[['session_id', 'q1', 'q2', 'q3', 'q4', 'q5', 'q6', 'q7', 'q8', 'q9',\n       'q10', 'q11', 'q12', 'q13', 'q14', 'q15', 'q16', 'q17', 'q18','Post_start_duration',\n       '1-4_elapsed_time','5-12_elapsed_time',  \n       '13-18_elapsed_time']]","metadata":{}},{"cell_type":"code","source":"target.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:03.850988Z","iopub.execute_input":"2023-06-02T15:59:03.851881Z","iopub.status.idle":"2023-06-02T15:59:03.877346Z","shell.execute_reply.started":"2023-06-02T15:59:03.851852Z","shell.execute_reply":"2023-06-02T15:59:03.876582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"((train[train.level==i][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max().values) - (train[train.level==i][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.min().values))","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:03.878563Z","iopub.execute_input":"2023-06-02T15:59:03.879640Z","iopub.status.idle":"2023-06-02T15:59:04.924003Z","shell.execute_reply.started":"2023-06-02T15:59:03.879609Z","shell.execute_reply":"2023-06-02T15:59:04.923010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(1,19)   :\n    try:\n        target[f'q{i}_elapsed_time']= ((train[train.level==i][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max().values) - (train[train.level==i][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.min().values))\n    except:\n        print(f\"Error in {i}\")\n        try:\n            target[f'q{i}_elapsed_time']= ((train[train.level==i+1][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.min().values) - (train[train.level==i-1][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max().values))\n        except:\n            print(f\"Error again in {i}\")\n            ","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:04.925268Z","iopub.execute_input":"2023-06-02T15:59:04.925638Z","iopub.status.idle":"2023-06-02T15:59:12.473037Z","shell.execute_reply.started":"2023-06-02T15:59:04.925611Z","shell.execute_reply":"2023-06-02T15:59:12.472099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"for i in [7,15]   :\n    try:\n        target[f'q{i}_elapsed_time']= ((train[train.level==i+1][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.min().values) - (train[train.level==i-1][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max().values))\n    except:\n        print(f\"Error in {i}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-30T12:36:09.806986Z","iopub.execute_input":"2023-05-30T12:36:09.807304Z","iopub.status.idle":"2023-05-30T12:36:10.616134Z","shell.execute_reply.started":"2023-05-30T12:36:09.807278Z","shell.execute_reply":"2023-05-30T12:36:10.614997Z"}}},{"cell_type":"code","source":"target.columns","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:12.474153Z","iopub.execute_input":"2023-06-02T15:59:12.474735Z","iopub.status.idle":"2023-06-02T15:59:12.480747Z","shell.execute_reply.started":"2023-06-02T15:59:12.474706Z","shell.execute_reply":"2023-06-02T15:59:12.479894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"import xgboost as xgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nfrom xgboost import XGBClassifier\nfor i in range(1,19):\n    # Prepare your feature matrix (X) and target matrix (y\n    if i in range(1,4):\n        X = target[['1-4_elapsed_time',f'q{i}_elapsed_time','Post_start_duration']]\n    elif i in range(4,14):\n        X = target[['5-12_elapsed_time',f'q{i}_elapsed_time','Post_start_duration']]\n    else:\n        X = target[['13-18_elapsed_time',f'q{i}_elapsed_time','Post_start_duration']]\n    y = target[[f'q{i}']]\n\n    # Split the data into training and testing sets\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n    # Define the XGBoost classifier for multi-output\n    num_classes=4\n    model = XGBClassifier(random_state=0,max_depth=5)\n\n    # Train the model\n    model.fit(X_train, y_train)\n\n    # Make predictions on the test set\n    y_pred = model.predict(X_test)\n\n    # Evaluate the accuracy of the model for each output\n    from sklearn.metrics import classification_report\n    print(f'Score of Q{i}: \\n {classification_report(y_pred,y_test)}')\n","metadata":{"execution":{"iopub.status.busy":"2023-06-01T17:29:15.205369Z","iopub.execute_input":"2023-06-01T17:29:15.206063Z","iopub.status.idle":"2023-06-01T17:29:30.479187Z","shell.execute_reply.started":"2023-06-01T17:29:15.206031Z","shell.execute_reply":"2023-06-01T17:29:30.478426Z"}}},{"cell_type":"markdown","source":"- In this project we are only focusing on how the session passed from level 0-18;i added 0 cause behavious before starting can be somehow with an effect","metadata":{}},{"cell_type":"code","source":"train.level.max()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:12.481798Z","iopub.execute_input":"2023-06-02T15:59:12.482688Z","iopub.status.idle":"2023-06-02T15:59:12.496791Z","shell.execute_reply.started":"2023-06-02T15:59:12.482659Z","shell.execute_reply":"2023-06-02T15:59:12.495872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:12.498061Z","iopub.execute_input":"2023-06-02T15:59:12.498909Z","iopub.status.idle":"2023-06-02T15:59:12.532074Z","shell.execute_reply.started":"2023-06-02T15:59:12.498878Z","shell.execute_reply":"2023-06-02T15:59:12.531340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"X_test=pd.DataFrame(columns=X.columns.tolist())\nX_test","metadata":{"execution":{"iopub.status.busy":"2023-06-02T00:32:44.736734Z","iopub.execute_input":"2023-06-02T00:32:44.737051Z","iopub.status.idle":"2023-06-02T00:32:45.131425Z","shell.execute_reply.started":"2023-06-02T00:32:44.737025Z","shell.execute_reply":"2023-06-02T00:32:45.130343Z"}}},{"cell_type":"code","source":"target_test=pd.DataFrame()#columns=X.columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:12.533073Z","iopub.execute_input":"2023-06-02T15:59:12.533690Z","iopub.status.idle":"2023-06-02T15:59:12.538498Z","shell.execute_reply.started":"2023-06-02T15:59:12.533661Z","shell.execute_reply":"2023-06-02T15:59:12.537569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = test[test['level'] < 19]\ntest['level_group'].replace({'13-22':'13-18'},inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:12.539533Z","iopub.execute_input":"2023-06-02T15:59:12.539809Z","iopub.status.idle":"2023-06-02T15:59:12.552897Z","shell.execute_reply.started":"2023-06-02T15:59:12.539786Z","shell.execute_reply":"2023-06-02T15:59:12.552078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_test['Post_start_duration']=test[test.level==0][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max().values","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:12.553905Z","iopub.execute_input":"2023-06-02T15:59:12.554700Z","iopub.status.idle":"2023-06-02T15:59:12.565087Z","shell.execute_reply.started":"2023-06-02T15:59:12.554671Z","shell.execute_reply":"2023-06-02T15:59:12.564301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for inter in ['0-4','5-12','13-18']   :\n    if inter!='0-4':\n        target_test[f'{inter}_elapsed_time']= test[test.level_group==inter][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max().values\n    else:\n        target_test[f'1-4_elapsed_time']= (test[test.level_group==inter][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max() - target_test['Post_start_duration'].values).values\n","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:12.566075Z","iopub.execute_input":"2023-06-02T15:59:12.566722Z","iopub.status.idle":"2023-06-02T15:59:12.585733Z","shell.execute_reply.started":"2023-06-02T15:59:12.566694Z","shell.execute_reply":"2023-06-02T15:59:12.584792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_test","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:12.587047Z","iopub.execute_input":"2023-06-02T15:59:12.587350Z","iopub.status.idle":"2023-06-02T15:59:12.602827Z","shell.execute_reply.started":"2023-06-02T15:59:12.587325Z","shell.execute_reply":"2023-06-02T15:59:12.602123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_test['session_id']=test.session_id.unique()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:12.604038Z","iopub.execute_input":"2023-06-02T15:59:12.604583Z","iopub.status.idle":"2023-06-02T15:59:12.614600Z","shell.execute_reply.started":"2023-06-02T15:59:12.604556Z","shell.execute_reply":"2023-06-02T15:59:12.613672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(1,19)   :\n    try:\n        target_test[f'q{i}_elapsed_time']= ((test[test.level==i][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max().values) - (test[test.level==i][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.min().values))\n    except:\n        print(f\"Error in {i}\")\n        try:\n            target[f'q{i}_elapsed_time']= ((train[train.level==i+1][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.min().values) - (train[train.level==i-1][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max().values))\n        except:\n            print(f\"Error again in {i}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:12.615955Z","iopub.execute_input":"2023-06-02T15:59:12.616248Z","iopub.status.idle":"2023-06-02T15:59:12.696529Z","shell.execute_reply.started":"2023-06-02T15:59:12.616223Z","shell.execute_reply":"2023-06-02T15:59:12.695608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub= pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:12.697851Z","iopub.execute_input":"2023-06-02T15:59:12.698138Z","iopub.status.idle":"2023-06-02T15:59:12.709233Z","shell.execute_reply.started":"2023-06-02T15:59:12.698113Z","shell.execute_reply":"2023-06-02T15:59:12.708320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"import xgboost as xgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nfrom xgboost import XGBClassifier\nfor i in range(1,19):\n    # Prepare your feature matrix (X) and target matrix (y\n    if i in range(1,4):\n        X = target[['1-4_elapsed_time',f'q{i}_elapsed_time','Post_start_duration']]\n    elif i in range(4,14):\n        X = target[['5-12_elapsed_time',f'q{i}_elapsed_time','Post_start_duration']]\n    else:\n        X = target[['13-18_elapsed_time',f'q{i}_elapsed_time','Post_start_duration']]\n    y = target[[f'q{i}']]\n\n    # Split the data into training and testing sets\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n    # Define the XGBoost classifier for multi-output\n    num_classes=4\n    model = XGBClassifier(random_state=0,max_depth=5)\n\n    # Train the model\n    model.fit(X_train, y_train)\n\n    # Make predictions on the test set\n    y_pred = model.predict(X_test)\n\n    # Evaluate the accuracy of the model for each output\n    from sklearn.metrics import classification_report\n    print(f'Score of Q{i}: \\n {classification_report(y_pred,y_test)}')\n    \n    #Results:\n    if i in range(1,4):\n        X_fin = target_test[['1-4_elapsed_time',f'q{i}_elapsed_time','Post_start_duration']]\n    elif i in range(4,14):\n        X_fin = target_test[['5-12_elapsed_time',f'q{i}_elapsed_time','Post_start_duration']]\n    else:\n        X_fin = target_test[['13-18_elapsed_time',f'q{i}_elapsed_time','Post_start_duration']]\n        \n    y_fin=model.predict(X_fin)\n    sub.loc[sub['session_id'].str.endswith(f'q{i}'), 'correct']=y_fin\n    \n\n\n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-01T17:29:30.733319Z","iopub.execute_input":"2023-06-01T17:29:30.733830Z","iopub.status.idle":"2023-06-01T17:29:45.458382Z","shell.execute_reply.started":"2023-06-01T17:29:30.733803Z","shell.execute_reply":"2023-06-01T17:29:45.457488Z"}}},{"cell_type":"code","source":"sub.loc[sub['session_id'].str.endswith(f'{i}'), 'correct'] \n","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:12.710417Z","iopub.execute_input":"2023-06-02T15:59:12.711237Z","iopub.status.idle":"2023-06-02T15:59:12.721040Z","shell.execute_reply.started":"2023-06-02T15:59:12.711209Z","shell.execute_reply":"2023-06-02T15:59:12.720186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:12.722236Z","iopub.execute_input":"2023-06-02T15:59:12.722970Z","iopub.status.idle":"2023-06-02T15:59:12.738223Z","shell.execute_reply.started":"2023-06-02T15:59:12.722941Z","shell.execute_reply":"2023-06-02T15:59:12.737199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_test","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:12.739313Z","iopub.execute_input":"2023-06-02T15:59:12.739718Z","iopub.status.idle":"2023-06-02T15:59:12.758850Z","shell.execute_reply.started":"2023-06-02T15:59:12.739690Z","shell.execute_reply":"2023-06-02T15:59:12.757963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jo_wilder_310\nenv = jo_wilder_310.make_env()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:12.760127Z","iopub.execute_input":"2023-06-02T15:59:12.760453Z","iopub.status.idle":"2023-06-02T15:59:12.785084Z","shell.execute_reply.started":"2023-06-02T15:59:12.760427Z","shell.execute_reply":"2023-06-02T15:59:12.784249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"iter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:12.786394Z","iopub.execute_input":"2023-06-02T15:59:12.786872Z","iopub.status.idle":"2023-06-02T15:59:12.790254Z","shell.execute_reply.started":"2023-06-02T15:59:12.786844Z","shell.execute_reply":"2023-06-02T15:59:12.789527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's try understand how this works\n**Note that the sample_submission.csv provided for your usage also includes a grouping variable, session_level**\n- since we have 3 groups: 0-4,5-12,13-22, i believe there will be thre iteration,and at each itetiration i should class by level to make each question have its prediction","metadata":{}},{"cell_type":"markdown","source":"from xgboost import XGBClassifier\nfor (test, sample_submission) in iter_test:\n    test.fillna(0,inplace=True)\n    grp = test.level_group.values[0]\n    target_test=pd.DataFrame()#columns=X.columns.tolist()\n    #target_test['Post_start_duration']=test[test.level==0][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max().values\n    inter=grp   \n    rang={'0-4':range(1,4),'5-12':range(4,14),'13-22':range(14,19)}\n    if inter=='5-12':\n        target_test[f'{inter}_elapsed_time']= test[test.level_group==inter][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max().values\n    elif inter=='13-22':\n        target_test['13-18_elapsed_time']= test[test.level_group==inter][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max().values\n      \n    else:\n        target_test[f'1-4_elapsed_time']= (test[test.level_group==inter][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max()).values\n    target_test['session_id']=test.session_id.unique()\n    for i in rang[inter]   :\n        try:\n            target_test[f'q{i}_elapsed_time']= ((test[test.level==i][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.max().values) - (test[test.level==i][['elapsed_time','session_id']].groupby(['session_id']).elapsed_time.min().values))\n        except:\n            print(f\"Error again in {i}\")\n    if inter=='5-12':\n        target_test[f'q4_elapsed_time']= target_test[f'q5_elapsed_time']-2000\n        target_test[f'q13_elapsed_time']= target_test[f'q12_elapsed_time']+2000\n\n    \n    #'0-4','5-12','13-22'\n    if grp=='0-4':\n        for lev in range(1,4):\n            #training on f'q{lev}'\n            X = target[['1-4_elapsed_time',f'q{lev}_elapsed_time']]        \n            y = target[[f'q{lev}']]\n                                    # Split the data into training and testing sets\n\n            X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n            # Define the XGBoost classifier for multi-output\n            num_classes=4\n            model = XGBClassifier(random_state=0,max_depth=5)\n\n            # Train the model\n            model.fit(X_train, y_train)\n            #-----------------------------------------------------------------\n            # predicting f'q{lev}'\n            X_fin =target_test.loc[target_test.session_id==test.session_id[0],['1-4_elapsed_time',f'q{lev}_elapsed_time']]\n            mask = (sample_submission.session_id.str.contains(f'q{lev}'))&(sample_submission.session_id.str.startswith(f'{test.session_id[0]}'))\n            predictions = model.predict(X_fin)\n            sample_submission.loc[mask,'correct'] = predictions.astype(int)\n    elif grp=='5-12':\n        for lev in range(4,14):\n            #training on f'q{lev}'\n            X = target[['5-12_elapsed_time',f'q{lev}_elapsed_time']]\n            y = target[[f'q{lev}']]\n                        # Split the data into training and testing sets\n            X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n            # Define the XGBoost classifier for multi-output\n            num_classes=4\n            model = XGBClassifier(random_state=0,max_depth=5)\n\n            # Train the model\n            model.fit(X_train, y_train)\n            #-----------------------------------------------------------------\n            # predicting f'q{lev}'\n            X_fin =target_test.loc[target_test.session_id==test.session_id[0],['5-12_elapsed_time',f'q{lev}_elapsed_time']]\n            mask = (sample_submission.session_id.str.contains(f'q{lev}'))&(sample_submission.session_id.str.startswith(f'{test.session_id[0]}'))\n            predictions = model.predict(X_fin)\n            sample_submission.loc[mask,'correct'] = predictions.astype(int)\n    else:\n        for lev in range(14,19):\n            #training on f'q{lev}'\n            X = target[['13-18_elapsed_time',f'q{lev}_elapsed_time']]\n            y = target[[f'q{lev}']]\n                                    # Split the data into training and testing sets\n\n            X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n            # Define the XGBoost classifier for multi-output\n            num_classes=4\n            model = XGBClassifier(random_state=0,max_depth=5)\n\n            # Train the model\n            model.fit(X_train, y_train)\n            #-----------------------------------------------------------------\n            # predicting f'q{lev}'\n            X_fin =target_test.loc[target_test.session_id==test.session_id[0],['13-18_elapsed_time',f'q{lev}_elapsed_time']]\n            mask = (sample_submission.session_id.str.contains(f'q{lev}'))&(sample_submission.session_id.str.startswith(f'{test.session_id[0]}'))\n            predictions = model.predict(X_fin)\n            sample_submission.loc[mask,'correct'] = predictions.astype(int)\n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:34:15.266321Z","iopub.execute_input":"2023-06-02T15:34:15.266703Z","iopub.status.idle":"2023-06-02T15:34:50.178485Z","shell.execute_reply.started":"2023-06-02T15:34:15.266683Z","shell.execute_reply":"2023-06-02T15:34:50.177250Z"}}},{"cell_type":"code","source":"from xgboost import XGBClassifier\nimport traceback\n\ntry:\n    for (test, sample_submission) in iter_test:\n        test.fillna(0, inplace=True)\n        print(test)\n        grp = test.level_group.values[0]\n        target_test = pd.DataFrame()\n        \n        inter = grp\n        rang = {'0-4': range(1, 4), '5-12': range(4, 14), '13-22': range(14, 19)}\n        \n        if inter == '5-12':\n            target_test[f'{inter}_elapsed_time'] = test[test.level_group == inter][['elapsed_time', 'session_id']].groupby(['session_id']).elapsed_time.max().values\n        elif inter == '13-22':\n            target_test['13-18_elapsed_time'] = test[test.level_group == inter][['elapsed_time', 'session_id']].groupby(['session_id']).elapsed_time.max().values\n        else:\n            target_test[f'1-4_elapsed_time'] = (test[test.level_group == inter][['elapsed_time', 'session_id']].groupby(['session_id']).elapsed_time.max()).values\n        \n        target_test['session_id'] = test.session_id.unique()\n        \n        for i in rang[inter]:\n            try:\n                target_test[f'q{i}_elapsed_time'] = ((test[test.level == i][['elapsed_time', 'session_id']].groupby(['session_id']).elapsed_time.max().values) - (test[test.level == i][['elapsed_time', 'session_id']].groupby(['session_id']).elapsed_time.min().values))\n            except Exception as e:\n                print(f\"Error occurred in {i}: {traceback.format_exc()}\")\n        \n        if inter == '5-12':\n            target_test[f'q4_elapsed_time'] = target_test[f'q5_elapsed_time'] - 2000\n            target_test[f'q13_elapsed_time'] = target_test[f'q12_elapsed_time'] + 2000\n    \n        if grp == '0-4':\n            for lev in range(1, 4):\n                try:\n                    X = target[['1-4_elapsed_time', f'q{lev}_elapsed_time']]\n                    y = target[[f'q{lev}']]\n                    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n                    num_classes = 4\n                    model = XGBClassifier(random_state=0, max_depth=5)\n                    model.fit(X_train, y_train)\n                    \n                    X_fin = target_test.loc[target_test.session_id == test.session_id[0], ['1-4_elapsed_time', f'q{lev}_elapsed_time']]\n                    mask = (sample_submission.session_id.str.contains(f'q{lev}')) & (sample_submission.session_id.str.startswith(f'{test.session_id[0]}'))\n                    predictions = model.predict(X_fin)\n                    sample_submission.loc[mask, 'correct'] = predictions.astype(int)\n                except Exception as e:\n                    print(f\"Error occurred in grp='0-4', lev={lev}: {traceback.format_exc()}\")\n        elif grp == '5-12':\n            for lev in range(4, 14):\n                try:\n                    X = target[['5-12_elapsed_time', f'q{lev}_elapsed_time']]\n                    y = target[[f'q{lev}']]\n                    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n                    num_classes = 4\n                    model = XGBClassifier(random_state=0, max_depth=5)\n                    model.fit(X_train, y_train)\n                    \n                    X_fin = target_test.loc[target_test.session_id == test.session_id[0], ['5-12_elapsed_time', f'q{lev}_elapsed_time']]\n                    mask = (sample_submission.session_id.str.contains(f'q{lev}')) & (sample_submission.session_id.str.startswith(f'{test.session_id[0]}'))\n                    predictions = model.predict(X_fin)\n                    sample_submission.loc[mask, 'correct'] = predictions.astype(int)\n                except Exception as e:\n                    print(f\"Error occurred in grp='5-12', lev={lev}: {traceback.format_exc()}\")\n        else:\n            for lev in range(14, 19):\n                try:\n                    X = target[['13-18_elapsed_time', f'q{lev}_elapsed_time']]\n                    y = target[[f'q{lev}']]\n                    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n                    num_classes = 4\n                    model = XGBClassifier(random_state=0, max_depth=5)\n                    model.fit(X_train, y_train)\n                    \n                    X_fin = target_test.loc[target_test.session_id == test.session_id[0], ['13-18_elapsed_time', f'q{lev}_elapsed_time']]\n                    mask = (sample_submission.session_id.str.contains(f'q{lev}')) & (sample_submission.session_id.str.startswith(f'{test.session_id[0]}'))\n                    predictions = model.predict(X_fin)\n                    sample_submission.loc[mask, 'correct'] = predictions.astype(int)\n                except Exception as e:\n                    print(f\"Error occurred in grp='13-22', lev={lev}: {traceback.format_exc()}\")\n        \n        env.predict(sample_submission)\n\nexcept Exception as e:\n    print(f\"Unexpected error occurred: {traceback.format_exc()}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:12.791472Z","iopub.execute_input":"2023-06-02T15:59:12.791910Z","iopub.status.idle":"2023-06-02T15:59:57.213764Z","shell.execute_reply.started":"2023-06-02T15:59:12.791885Z","shell.execute_reply":"2023-06-02T15:59:57.212781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_test","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:57.215216Z","iopub.execute_input":"2023-06-02T15:59:57.215820Z","iopub.status.idle":"2023-06-02T15:59:57.227763Z","shell.execute_reply.started":"2023-06-02T15:59:57.215790Z","shell.execute_reply":"2023-06-02T15:59:57.226948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:57.231315Z","iopub.execute_input":"2023-06-02T15:59:57.231642Z","iopub.status.idle":"2023-06-02T15:59:58.511505Z","shell.execute_reply.started":"2023-06-02T15:59:57.231597Z","shell.execute_reply":"2023-06-02T15:59:58.510431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_test","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:58.512848Z","iopub.execute_input":"2023-06-02T15:59:58.513520Z","iopub.status.idle":"2023-06-02T15:59:58.526196Z","shell.execute_reply.started":"2023-06-02T15:59:58.513487Z","shell.execute_reply":"2023-06-02T15:59:58.525311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:58.527541Z","iopub.execute_input":"2023-06-02T15:59:58.527821Z","iopub.status.idle":"2023-06-02T15:59:58.567004Z","shell.execute_reply.started":"2023-06-02T15:59:58.527797Z","shell.execute_reply":"2023-06-02T15:59:58.566228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target","metadata":{"execution":{"iopub.status.busy":"2023-06-02T15:59:58.568058Z","iopub.execute_input":"2023-06-02T15:59:58.568836Z","iopub.status.idle":"2023-06-02T15:59:58.599393Z","shell.execute_reply.started":"2023-06-02T15:59:58.568807Z","shell.execute_reply":"2023-06-02T15:59:58.598592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}