{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport plotly as py\nimport plotly.graph_objs as go\nimport plotly.express as px\nfrom plotly.subplots import make_subplots","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:00:34.172290Z","iopub.execute_input":"2023-03-06T05:00:34.173371Z","iopub.status.idle":"2023-03-06T05:00:37.431403Z","shell.execute_reply.started":"2023-03-06T05:00:34.173313Z","shell.execute_reply":"2023-03-06T05:00:37.430245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold, GroupKFold\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import f1_score","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:00:37.436431Z","iopub.execute_input":"2023-03-06T05:00:37.436984Z","iopub.status.idle":"2023-03-06T05:00:37.739856Z","shell.execute_reply.started":"2023-03-06T05:00:37.436949Z","shell.execute_reply":"2023-03-06T05:00:37.738554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reduce Memory Usage\n# reference : https://www.kaggle.com/code/arjanso/reducing-dataframe-memory-size-by-65 @ARJANGROEN\n\ndef reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:00:37.742809Z","iopub.execute_input":"2023-03-06T05:00:37.743503Z","iopub.status.idle":"2023-03-06T05:00:37.758719Z","shell.execute_reply.started":"2023-03-06T05:00:37.743456Z","shell.execute_reply":"2023-03-06T05:00:37.757228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:00:37.765598Z","iopub.execute_input":"2023-03-06T05:00:37.766373Z","iopub.status.idle":"2023-03-06T05:01:47.372800Z","shell.execute_reply.started":"2023-03-06T05:00:37.766323Z","shell.execute_reply":"2023-03-06T05:01:47.371764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = reduce_memory_usage(train)\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:01:47.374307Z","iopub.execute_input":"2023-03-06T05:01:47.374904Z","iopub.status.idle":"2023-03-06T05:02:02.501202Z","shell.execute_reply.started":"2023-03-06T05:01:47.374867Z","shell.execute_reply":"2023-03-06T05:02:02.499954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:02:02.502635Z","iopub.execute_input":"2023-03-06T05:02:02.503223Z","iopub.status.idle":"2023-03-06T05:02:02.510010Z","shell.execute_reply.started":"2023-03-06T05:02:02.503186Z","shell.execute_reply":"2023-03-06T05:02:02.509066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:02:02.511203Z","iopub.execute_input":"2023-03-06T05:02:02.511894Z","iopub.status.idle":"2023-03-06T05:02:02.697232Z","shell.execute_reply.started":"2023-03-06T05:02:02.511848Z","shell.execute_reply":"2023-03-06T05:02:02.695907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\")\ntarget.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:02:02.698996Z","iopub.execute_input":"2023-03-06T05:02:02.699499Z","iopub.status.idle":"2023-03-06T05:02:02.957619Z","shell.execute_reply.started":"2023-03-06T05:02:02.699453Z","shell.execute_reply":"2023-03-06T05:02:02.956241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target['session'] = target.session_id.apply(lambda x: int(x.split('_')[0]) )\ntarget['q'] = target.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nprint( target.shape )\ntarget.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:02:02.959079Z","iopub.execute_input":"2023-03-06T05:02:02.959767Z","iopub.status.idle":"2023-03-06T05:02:03.354425Z","shell.execute_reply.started":"2023-03-06T05:02:02.959724Z","shell.execute_reply":"2023-03-06T05:02:03.353055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:02:03.355849Z","iopub.execute_input":"2023-03-06T05:02:03.357095Z","iopub.status.idle":"2023-03-06T05:02:03.539782Z","shell.execute_reply.started":"2023-03-06T05:02:03.357045Z","shell.execute_reply":"2023-03-06T05:02:03.538446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def summary(df):\n    print(f'data shape: {df.shape}')\n    summ = pd.DataFrame(df.dtypes, columns=['data type'])\n    summ['#missing'] = df.isnull().sum().values * 100\n    summ['%missing'] = df.isnull().sum().values / len(df)\n    summ['#unique'] = df.nunique().values\n    desc = pd.DataFrame(df.describe(include='all').transpose())\n    summ['min'] = desc['min'].values\n    summ['max'] = desc['max'].values\n    summ['first value'] = df.loc[0].values\n    summ['second value'] = df.loc[1].values\n    summ['third value'] = df.loc[2].values\n    \n    return summ","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:02:03.541976Z","iopub.execute_input":"2023-03-06T05:02:03.543249Z","iopub.status.idle":"2023-03-06T05:02:03.552985Z","shell.execute_reply.started":"2023-03-06T05:02:03.543192Z","shell.execute_reply":"2023-03-06T05:02:03.551621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary_table = summary(train)\nsummary_table","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:02:03.555052Z","iopub.execute_input":"2023-03-06T05:02:03.555871Z","iopub.status.idle":"2023-03-06T05:02:21.902312Z","shell.execute_reply.started":"2023-03-06T05:02:03.555825Z","shell.execute_reply":"2023-03-06T05:02:21.901019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Feature Engineering","metadata":{}},{"cell_type":"code","source":"CATS = ['event_name', 'fqid', 'room_fqid', 'text']\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y','screen_coor_x', 'screen_coor_y', 'hover_duration']\n","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:02:21.906715Z","iopub.execute_input":"2023-03-06T05:02:21.907385Z","iopub.status.idle":"2023-03-06T05:02:21.914062Z","shell.execute_reply.started":"2023-03-06T05:02:21.907339Z","shell.execute_reply":"2023-03-06T05:02:21.912844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create dummies\njust_dummies = pd.get_dummies(train['event_name'])\n\ntrain = pd.concat([train, just_dummies], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:02:21.916019Z","iopub.execute_input":"2023-03-06T05:02:21.916853Z","iopub.status.idle":"2023-03-06T05:02:22.506003Z","shell.execute_reply.started":"2023-03-06T05:02:21.916806Z","shell.execute_reply":"2023-03-06T05:02:22.504660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I comment this code because of the RAM issue\n# train.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:02:22.507978Z","iopub.execute_input":"2023-03-06T05:02:22.508796Z","iopub.status.idle":"2023-03-06T05:02:22.513057Z","shell.execute_reply.started":"2023-03-06T05:02:22.508753Z","shell.execute_reply":"2023-03-06T05:02:22.511999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['event_name'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:02:22.514960Z","iopub.execute_input":"2023-03-06T05:02:22.515806Z","iopub.status.idle":"2023-03-06T05:02:22.608023Z","shell.execute_reply.started":"2023-03-06T05:02:22.515750Z","shell.execute_reply":"2023-03-06T05:02:22.606556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data\nEVENTS = ['navigate_click','person_click','cutscene_click','object_click',\n          'map_hover','notification_click','map_click','observation_click',\n          'checkpoint']","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:02:22.609931Z","iopub.execute_input":"2023-03-06T05:02:22.610347Z","iopub.status.idle":"2023-03-06T05:02:22.616360Z","shell.execute_reply.started":"2023-03-06T05:02:22.610306Z","shell.execute_reply":"2023-03-06T05:02:22.615251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(train):\n    \n    dfs = []\n    for c in CATS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    for c in EVENTS: \n        train[c] = (train.event_name == c).astype('int8')\n    for c in EVENTS + ['elapsed_time']:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('sum')\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n    train = train.drop(EVENTS,axis=1)\n        \n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:02:22.618182Z","iopub.execute_input":"2023-03-06T05:02:22.618973Z","iopub.status.idle":"2023-03-06T05:02:22.632396Z","shell.execute_reply.started":"2023-03-06T05:02:22.618930Z","shell.execute_reply":"2023-03-06T05:02:22.630876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf = feature_engineer(train)\nprint( df.shape )\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:02:22.634496Z","iopub.execute_input":"2023-03-06T05:02:22.634947Z","iopub.status.idle":"2023-03-06T05:02:45.400105Z","shell.execute_reply.started":"2023-03-06T05:02:22.634905Z","shell.execute_reply":"2023-03-06T05:02:45.398358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:02:45.402247Z","iopub.execute_input":"2023-03-06T05:02:45.402837Z","iopub.status.idle":"2023-03-06T05:02:45.604468Z","shell.execute_reply.started":"2023-03-06T05:02:45.402779Z","shell.execute_reply":"2023-03-06T05:02:45.603126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check data type\ndf.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:02:45.607338Z","iopub.execute_input":"2023-03-06T05:02:45.607812Z","iopub.status.idle":"2023-03-06T05:02:45.630586Z","shell.execute_reply.started":"2023-03-06T05:02:45.607767Z","shell.execute_reply":"2023-03-06T05:02:45.628990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\nALL_USERS = df.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:02:45.632794Z","iopub.execute_input":"2023-03-06T05:02:45.634391Z","iopub.status.idle":"2023-03-06T05:02:45.643989Z","shell.execute_reply.started":"2023-03-06T05:02:45.634334Z","shell.execute_reply":"2023-03-06T05:02:45.642701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train XGBoost Model","metadata":{}},{"cell_type":"code","source":"gkf = GroupKFold(n_splits=5)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\nmodels = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#'*25)\n    print('### Fold',i+1)\n    print('#'*25)\n    \n    xgb_params = {\n    'objective' : 'binary:logistic',\n    'eval_metric':'logloss',\n    'learning_rate': 0.05,\n    'max_depth': 4,\n    'n_estimators': 1000,\n    'early_stopping_rounds': 50,\n    'tree_method':'hist',\n    'subsample':0.8,\n    'colsample_bytree': 0.4,\n    'use_label_encoder' : False}\n    \n    # ITERATE THRU QUESTIONS 1 THRU 18\n    for t in range(1,19):\n        \n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t<=3: grp = '0-4'\n        elif t<=13: grp = '5-12'\n        elif t<=22: grp = '13-22'\n            \n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = target.loc[target.q==t].set_index('session').loc[train_users]\n        \n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = target.loc[target.q==t].set_index('session').loc[valid_users]\n        \n        # TRAIN MODEL        \n        clf =  XGBClassifier(**xgb_params)\n        clf.fit(train_x[FEATURES].astype('float32'), train_y['correct'],\n                eval_set=[ (valid_x[FEATURES].astype('float32'), valid_y['correct']) ],\n                verbose=0)\n        print(f'{t}({clf.best_ntree_limit}), ',end='')\n        \n        # SAVE MODEL, PREDICT VALID OOF\n        models[f'{grp}_{t}'] = clf\n        oof.loc[valid_users, t-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]\n        \n    print()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:10:42.897711Z","iopub.execute_input":"2023-03-06T05:10:42.898955Z","iopub.status.idle":"2023-03-06T05:12:38.124446Z","shell.execute_reply.started":"2023-03-06T05:10:42.898889Z","shell.execute_reply":"2023-03-06T05:12:38.122984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = target.loc[target.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:12:38.127221Z","iopub.execute_input":"2023-03-06T05:12:38.128237Z","iopub.status.idle":"2023-03-06T05:12:38.231838Z","shell.execute_reply.started":"2023-03-06T05:12:38.128174Z","shell.execute_reply":"2023-03-06T05:12:38.230545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score = 0; best_threshold = 0\n\nfor threshold in np.arange(0.4,0.81,0.01):\n    print(f'{threshold:.02f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score:\n        best_score = m\n        best_threshold = threshold","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:12:38.233540Z","iopub.execute_input":"2023-03-06T05:12:38.234304Z","iopub.status.idle":"2023-03-06T05:12:41.638137Z","shell.execute_reply.started":"2023-03-06T05:12:38.234257Z","shell.execute_reply":"2023-03-06T05:12:41.636052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_threshold:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:12:41.640867Z","iopub.execute_input":"2023-03-06T05:12:41.641825Z","iopub.status.idle":"2023-03-06T05:12:41.995910Z","shell.execute_reply.started":"2023-03-06T05:12:41.641759Z","shell.execute_reply":"2023-03-06T05:12:41.994646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_threshold).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_threshold).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:20:03.972336Z","iopub.execute_input":"2023-03-06T05:20:03.972968Z","iopub.status.idle":"2023-03-06T05:20:04.152811Z","shell.execute_reply.started":"2023-03-06T05:20:03.972918Z","shell.execute_reply":"2023-03-06T05:20:04.151293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Infer Test Data","metadata":{}},{"cell_type":"code","source":"# IMPORT KAGGLE API\nimport jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:20:16.006913Z","iopub.execute_input":"2023-03-06T05:20:16.007426Z","iopub.status.idle":"2023-03-06T05:20:16.046785Z","shell.execute_reply.started":"2023-03-06T05:20:16.007381Z","shell.execute_reply":"2023-03-06T05:20:16.045572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CLEAR MEMORY\nimport gc\ndel train, target, df, oof, true\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:47:11.075534Z","iopub.execute_input":"2023-03-06T05:47:11.078841Z","iopub.status.idle":"2023-03-06T05:47:11.376978Z","shell.execute_reply.started":"2023-03-06T05:47:11.078754Z","shell.execute_reply":"2023-03-06T05:47:11.375166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (sample_submission, test) in iter_test:\n    \n    # FEATURE ENGINEER TEST DATA\n    df = feature_engineer(test)\n    \n    # INFER TEST DATA\n    grp = test.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        clf = models[f'{grp}_{t}']\n        p = clf.predict_proba(df[FEATURES].astype('float32'))[:,1]\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        sample_submission.loc[mask,'correct'] = int(p.item()>best_threshold)\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:47:53.616444Z","iopub.execute_input":"2023-03-06T05:47:53.617005Z","iopub.status.idle":"2023-03-06T05:47:54.686365Z","shell.execute_reply.started":"2023-03-06T05:47:53.616959Z","shell.execute_reply":"2023-03-06T05:47:54.684972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"EDA submission.csv","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('submission.csv')\nprint( df.shape )\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:48:04.517625Z","iopub.execute_input":"2023-03-06T05:48:04.518849Z","iopub.status.idle":"2023-03-06T05:48:04.537146Z","shell.execute_reply.started":"2023-03-06T05:48:04.518789Z","shell.execute_reply":"2023-03-06T05:48:04.535638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.correct.mean())","metadata":{"execution":{"iopub.status.busy":"2023-03-06T05:48:09.249068Z","iopub.execute_input":"2023-03-06T05:48:09.250132Z","iopub.status.idle":"2023-03-06T05:48:09.259442Z","shell.execute_reply.started":"2023-03-06T05:48:09.250073Z","shell.execute_reply":"2023-03-06T05:48:09.257742Z"},"trusted":true},"execution_count":null,"outputs":[]}]}