{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45533,"databundleVersionId":5748852,"sourceType":"competition"}],"dockerImageVersionId":30474,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"I have borrowed some ideas from chris deotte's https://www.kaggle.com/code/cdeotte/xgboost-baseline-0-680 noteook and modified with extra features and NN baseline instead of XGBOOST , we can experiment further by ensembling neural networks prediction with catboost and xgboost to get better results\n\nTo avoid memory error. We accomplish this by reading train data in chunks and feature engineering in chunks. Note that another way to avoid memory error is to use two notebooks. Train models in one notebook that has 32GB RAM (and save models), and then submit the required 8GB RAM notebook (with loaded models) as a second notebook. (Discussion here).","metadata":{}},{"cell_type":"code","source":"import pandas as pd, numpy as np, gc\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom xgboost import XGBClassifier\nfrom catboost import CatBoostClassifier\nfrom sklearn.metrics import f1_score\nimport numpy as np\nfrom keras.models import Sequential\nfrom keras.layers import Dense,Dropout\nimport warnings \nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-26T02:30:35.507521Z","iopub.execute_input":"2023-05-26T02:30:35.507957Z","iopub.status.idle":"2023-05-26T02:30:42.970997Z","shell.execute_reply.started":"2023-05-26T02:30:35.507906Z","shell.execute_reply":"2023-05-26T02:30:42.969988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The train data is now 4.7GB! To avoid memory error, we will read the train data in as 10 pieces and feature engineer each piece before reading the next piece. This works because feature engineering shrinks the size of each piece.","metadata":{}},{"cell_type":"code","source":"# READ USER ID ONLY\ntmp = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\",usecols=[0])\ntmp = tmp.groupby('session_id').session_id.agg('count')","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:30:42.972886Z","iopub.execute_input":"2023-05-26T02:30:42.973611Z","iopub.status.idle":"2023-05-26T02:31:49.139235Z","shell.execute_reply.started":"2023-05-26T02:30:42.973576Z","shell.execute_reply":"2023-05-26T02:31:49.138230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# COMPUTE READS AND SKIPS\nPIECES = 10\nCHUNK = int( np.ceil(len(tmp)/PIECES) )\nreads = []\nskips = [0]\nfor k in range(PIECES):\n    a = k*CHUNK\n    b = (k+1)*CHUNK\n    if b>len(tmp): b=len(tmp)\n    r = tmp.iloc[a:b].sum()\n    reads.append(r)\n    skips.append(skips[-1]+r)\n    \nprint(f'To avoid memory error, we will read train in {PIECES} pieces of sizes:')\nprint(reads)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:31:49.140681Z","iopub.execute_input":"2023-05-26T02:31:49.141033Z","iopub.status.idle":"2023-05-26T02:31:49.150456Z","shell.execute_reply.started":"2023-05-26T02:31:49.141002Z","shell.execute_reply":"2023-05-26T02:31:49.149183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', nrows=reads[1])\nprint('Train size of first piece:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:31:49.153320Z","iopub.execute_input":"2023-05-26T02:31:49.153911Z","iopub.status.idle":"2023-05-26T02:31:55.392609Z","shell.execute_reply.started":"2023-05-26T02:31:49.153877Z","shell.execute_reply":"2023-05-26T02:31:55.391595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]) )\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nprint( targets.shape )\ntargets.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:31:55.394226Z","iopub.execute_input":"2023-05-26T02:31:55.394959Z","iopub.status.idle":"2023-05-26T02:31:56.620322Z","shell.execute_reply.started":"2023-05-26T02:31:55.394905Z","shell.execute_reply":"2023-05-26T02:31:56.619311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check if nulls are present for each column\ntrain.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:31:56.621838Z","iopub.execute_input":"2023-05-26T02:31:56.623188Z","iopub.status.idle":"2023-05-26T02:32:01.090943Z","shell.execute_reply.started":"2023-05-26T02:31:56.623151Z","shell.execute_reply":"2023-05-26T02:32:01.089950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# FEATURE ENGINEERING","metadata":{}},{"cell_type":"code","source":"num_features = list(train.select_dtypes(exclude=\"object\").columns)\ncat_features = list(train.select_dtypes(include=\"object\").columns)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:32:01.092173Z","iopub.execute_input":"2023-05-26T02:32:01.092517Z","iopub.status.idle":"2023-05-26T02:32:01.297382Z","shell.execute_reply.started":"2023-05-26T02:32:01.092485Z","shell.execute_reply":"2023-05-26T02:32:01.296420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#excluding first two ( session_id and index ) from numerical features list\nnum_features = num_features[2:]","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:32:01.298693Z","iopub.execute_input":"2023-05-26T02:32:01.299496Z","iopub.status.idle":"2023-05-26T02:32:01.304100Z","shell.execute_reply.started":"2023-05-26T02:32:01.299461Z","shell.execute_reply":"2023-05-26T02:32:01.303008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_features","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:32:01.305803Z","iopub.execute_input":"2023-05-26T02:32:01.306208Z","iopub.status.idle":"2023-05-26T02:32:01.316844Z","shell.execute_reply.started":"2023-05-26T02:32:01.306173Z","shell.execute_reply":"2023-05-26T02:32:01.315794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in cat_features:\n    print(i)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:32:01.322006Z","iopub.execute_input":"2023-05-26T02:32:01.322256Z","iopub.status.idle":"2023-05-26T02:32:01.327602Z","shell.execute_reply.started":"2023-05-26T02:32:01.322235Z","shell.execute_reply":"2023-05-26T02:32:01.326508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EVENTS = list(train['event_name'].unique())","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:32:01.329225Z","iopub.execute_input":"2023-05-26T02:32:01.329582Z","iopub.status.idle":"2023-05-26T02:32:01.530096Z","shell.execute_reply.started":"2023-05-26T02:32:01.329524Z","shell.execute_reply":"2023-05-26T02:32:01.529120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(EVENTS)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:32:01.531577Z","iopub.execute_input":"2023-05-26T02:32:01.532231Z","iopub.status.idle":"2023-05-26T02:32:01.538240Z","shell.execute_reply.started":"2023-05-26T02:32:01.532198Z","shell.execute_reply":"2023-05-26T02:32:01.537356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in cat_features:\n    print(train[i].value_counts())\n    print(\"===============\")","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:32:01.539824Z","iopub.execute_input":"2023-05-26T02:32:01.540510Z","iopub.status.idle":"2023-05-26T02:32:03.588245Z","shell.execute_reply.started":"2023-05-26T02:32:01.540479Z","shell.execute_reply":"2023-05-26T02:32:03.587248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:32:03.590746Z","iopub.execute_input":"2023-05-26T02:32:03.591302Z","iopub.status.idle":"2023-05-26T02:32:03.612797Z","shell.execute_reply.started":"2023-05-26T02:32:03.591275Z","shell.execute_reply":"2023-05-26T02:32:03.611884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:32:03.614177Z","iopub.execute_input":"2023-05-26T02:32:03.615071Z","iopub.status.idle":"2023-05-26T02:32:04.673286Z","shell.execute_reply.started":"2023-05-26T02:32:03.615035Z","shell.execute_reply":"2023-05-26T02:32:04.672212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.groupby(['session_id','level_group'])['name'].nunique()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:32:04.674909Z","iopub.execute_input":"2023-05-26T02:32:04.675288Z","iopub.status.idle":"2023-05-26T02:32:05.249797Z","shell.execute_reply.started":"2023-05-26T02:32:04.675252Z","shell.execute_reply":"2023-05-26T02:32:05.248654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineering(train):\n    from scipy.stats import mode\n\n    dfs = []\n    for c in cat_features:\n        tmp = train.groupby(['session_id','level_group'])[c].nunique()\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n        \n        \n    for c in cat_features:\n        tmp = train.groupby(['session_id','level_group'])[c].count()\n        tmp.name = tmp.name + '_count'\n        dfs.append(tmp)    \n        \n    for c in num_features:\n        tmp = train.groupby(['session_id','level_group'])[c].mean()\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in num_features:\n        tmp = train.groupby(['session_id','level_group'])[c].std()\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    \n    for c in num_features:\n        tmp = train.groupby(['session_id','level_group'])[c].sum()\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n        \n    \n    \n    for c in num_features:\n        tmp = train.groupby(['session_id','level_group'])[c].count()\n        tmp.name = tmp.name + '_count'\n        dfs.append(tmp)\n        \n        \n    for c in num_features:\n        tmp = train.groupby(['session_id','level_group'])[c].quantile()\n        tmp.name = tmp.name + '_quantile'\n        dfs.append(tmp)\n        \n        \n    for c in num_features:\n        tmp = train.groupby(['session_id','level_group'])[c].max()\n        tmp.name = tmp.name + '_max'\n        dfs.append(tmp)\n        \n    for c in num_features:\n        tmp = train.groupby(['session_id','level_group'])[c].min()\n        tmp.name = tmp.name + '_min'\n        dfs.append(tmp)\n    \n    for c in num_features:\n        tmp = train.groupby(['session_id','level_group'])[c].median()\n        tmp.name = tmp.name + '_median'\n        dfs.append(tmp)\n\n    for c in EVENTS: \n        train[c] = (train['event_name'] == c).astype('int8')\n        \n    for c in EVENTS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('sum')\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n        \n    train = train.drop(EVENTS,axis=1)\n        \n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:32:05.251456Z","iopub.execute_input":"2023-05-26T02:32:05.251839Z","iopub.status.idle":"2023-05-26T02:32:05.269088Z","shell.execute_reply.started":"2023-05-26T02:32:05.251804Z","shell.execute_reply":"2023-05-26T02:32:05.268246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new = feature_engineering(train)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:32:05.270576Z","iopub.execute_input":"2023-05-26T02:32:05.270936Z","iopub.status.idle":"2023-05-26T02:32:47.573118Z","shell.execute_reply.started":"2023-05-26T02:32:05.270890Z","shell.execute_reply":"2023-05-26T02:32:47.571902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:32:47.574641Z","iopub.execute_input":"2023-05-26T02:32:47.574998Z","iopub.status.idle":"2023-05-26T02:32:47.595125Z","shell.execute_reply.started":"2023-05-26T02:32:47.574965Z","shell.execute_reply":"2023-05-26T02:32:47.593891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# PROCESS TRAIN DATA IN PIECES\nall_pieces = []\nprint(f'Processing train as {PIECES} pieces to avoid memory error... ')\nfor k in range(PIECES):\n    print(k,', ',end='')\n    SKIPS = 0\n    if k>0: SKIPS = range(1,skips[k]+1)\n    train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv',\n                        nrows=reads[k], skiprows=SKIPS)\n    df = feature_engineering(train)\n    all_pieces.append(df)\n    \n# CONCATENATE ALL PIECES\nprint('\\n')\ndel train; gc.collect()\ndf = pd.concat(all_pieces, axis=0)\nprint('Shape of all train data after feature engineering:', df.shape )\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:32:47.596666Z","iopub.execute_input":"2023-05-26T02:32:47.597625Z","iopub.status.idle":"2023-05-26T02:43:08.289587Z","shell.execute_reply.started":"2023-05-26T02:32:47.597589Z","shell.execute_reply":"2023-05-26T02:43:08.288592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df1  = df[df['level_group']==\"0-4\"]\n# df2  = df[df['level_group']==\"5-12\"]\n# df3  = df[df['level_group']==\"13-22\"]\n# LEVELS = [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, \n#           13, 14, 15, 16, 17, 18, 19, 20, 21, 22]","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:43:08.290986Z","iopub.execute_input":"2023-05-26T02:43:08.291759Z","iopub.status.idle":"2023-05-26T02:43:08.296900Z","shell.execute_reply.started":"2023-05-26T02:43:08.291721Z","shell.execute_reply":"2023-05-26T02:43:08.295785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df.to_csv(\"train_preprocessed.csv\",index=False)\n# df.to_csv(\"test_preprocesed.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:43:08.298567Z","iopub.execute_input":"2023-05-26T02:43:08.298947Z","iopub.status.idle":"2023-05-26T02:43:08.310887Z","shell.execute_reply.started":"2023-05-26T02:43:08.298898Z","shell.execute_reply":"2023-05-26T02:43:08.309972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\nprint('We will train with', len(FEATURES) ,'features')\nALL_USERS = df.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:43:08.312390Z","iopub.execute_input":"2023-05-26T02:43:08.312725Z","iopub.status.idle":"2023-05-26T02:43:08.324978Z","shell.execute_reply.started":"2023-05-26T02:43:08.312689Z","shell.execute_reply":"2023-05-26T02:43:08.324059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ARTIFICIAL NEURAL NETWORK","metadata":{}},{"cell_type":"markdown","source":"I hav created a custom NN with 3 skip connections, in this way we can increase the layers of NN model and can avoid vanishing gradient problems as well\n\nWe train one model for each of 18 questions. Furthermore, we use data from level_groups = '0-4' to train model for questions 1-3, and level groups '5-12' to train questions 4 thru 13 and level groups '13-22' to train questions 14 thru 18. Because this is the data we get (to predict corresponding questions) from Kaggle's inference API during test inference. We can improve our model by saving a user's previous data from earlier level_groups and using that to predict future level_groups","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.preprocessing import LabelEncoder\nfrom tensorflow.keras.models import Sequential,Model\nfrom tensorflow.keras.layers import Dense,Input,Add\nfrom tensorflow.keras.optimizers import Adam\nimport keras\nimport tensorflow as tf\ngkf = GroupKFold(n_splits=5)\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS), 18)), index=ALL_USERS)\nmodels = {}\n\n# COMPUTE CV SCORE WITH 5 GROUP K FOLD\nfor i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n    print('#' * 25)\n    print('### Fold', i + 1)\n    print('#' * 25)\n\n    \n    def build_model():\n        input_layer = Input(shape=(len(FEATURES),))\n        hidden_layer_1 = Dense(64, activation='relu')(input_layer)\n        hidden_layer_2 = Dense(32, activation='relu')(hidden_layer_1)\n        hidden_layer_3 = Dense(64, activation='relu')(hidden_layer_2)\n        hidden_layer_4 = Dense(32, activation='relu')(hidden_layer_3)\n        hidden_layer_5 = Dense(64, activation='relu')(hidden_layer_4)\n        hidden_layer_6 = Dense(32, activation='relu')(Add()([hidden_layer_1, hidden_layer_5]))\n    #     skip_1 = Add()([hidden_layer_1, hidden_layer_5])\n    \n        hidden_layer_7 = Dense(64, activation='relu')(hidden_layer_6)\n        hidden_layer_8 = Dense(32, activation='relu')(Add()([hidden_layer_3, hidden_layer_7]))\n        hidden_layer_9 = Dense(64, activation='relu')(hidden_layer_8)\n        hidden_layer_10 = Dense(8, activation='relu')(Add()([hidden_layer_5, hidden_layer_9]))\n\n        skip_4 = Add()([hidden_layer_2, hidden_layer_8])\n        output_layer = Dense(1, activation='sigmoid')(skip_4)\n        model = Model(inputs=input_layer, outputs=output_layer)\n        return model\n# Build the model\n    model = build_model()\n#     tf.keras.utils.plot_model(model, show_shapes=True)\n    from tensorflow.keras.callbacks import EarlyStopping,ReduceLROnPlateau\n    early = EarlyStopping(monitor=\"val_loss\", mode= \"min\", patience=7)\n#     learning_rate_reduction = ReduceLROnPlateau(monitor=\"val_loss\", patience =5, verbose=1,factor=0.02,min_learning_rate=0.0001)\n    callbacks_list = [early]\n    model.compile(optimizer=Adam(learning_rate=0.05), loss='binary_crossentropy',metrics=['accuracy'])\n\n    # ITERATE THROUGH QUESTIONS 1 THRU 18\n    for t in range(1, 19):\n\n        # USE THIS TRAIN DATA WITH THESE QUESTIONS\n        if t <= 3:\n            grp = '0-4'\n        elif t <= 13:\n            grp = '5-12'\n        elif t <= 22:\n            grp = '13-22'\n\n        # TRAIN DATA\n        train_x = df.iloc[train_index]\n        train_x = train_x.loc[train_x.level_group == grp]\n        train_users = train_x.index.values\n        train_y = targets.loc[targets.q == t].set_index('session').loc[train_users]\n\n        # VALID DATA\n        valid_x = df.iloc[test_index]\n        valid_x = valid_x.loc[valid_x.level_group == grp]\n        valid_users = valid_x.index.values\n        valid_y = targets.loc[targets.q == t].set_index('session').loc[valid_users]\n\n        # CONVERT LABELS TO NUMERIC\n        label_encoder = LabelEncoder()\n        train_y_encoded = label_encoder.fit_transform(train_y['correct'])\n        valid_y_encoded = label_encoder.transform(valid_y['correct'])\n\n        # TRAIN MODEL\n        \n        \n        model.fit(train_x[FEATURES].astype('float32'), train_y_encoded,\n                  validation_data=(valid_x[FEATURES].astype('float32'), valid_y_encoded),\n                  epochs=5,callbacks= callbacks_list)\n\n        # SAVE MODEL, PREDICT VALID OOF\n        models[f'{grp}_{t}'] = model\n        oof.loc[valid_users, t - 1] = model.predict(valid_x[FEATURES].astype('float32'))[:, 0]\n\n    print()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-26T02:43:08.326621Z","iopub.execute_input":"2023-05-26T02:43:08.327349Z","iopub.status.idle":"2023-05-26T03:10:34.287033Z","shell.execute_reply.started":"2023-05-26T02:43:08.327317Z","shell.execute_reply":"2023-05-26T03:10:34.285992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\ntrue = oof.copy()\nfor k in range(18):\n    # GET TRUE LABELS\n    tmp = targets.loc[targets.q == k+1].set_index('session').loc[ALL_USERS]\n    true[k] = tmp.correct.values","metadata":{"execution":{"iopub.status.busy":"2023-05-26T03:10:34.288639Z","iopub.execute_input":"2023-05-26T03:10:34.288989Z","iopub.status.idle":"2023-05-26T03:10:34.388246Z","shell.execute_reply.started":"2023-05-26T03:10:34.288954Z","shell.execute_reply":"2023-05-26T03:10:34.387044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predictions by neural network model on each session id for each question\noof","metadata":{"execution":{"iopub.status.busy":"2023-05-26T03:10:34.389863Z","iopub.execute_input":"2023-05-26T03:10:34.390294Z","iopub.status.idle":"2023-05-26T03:10:34.425062Z","shell.execute_reply.started":"2023-05-26T03:10:34.390256Z","shell.execute_reply":"2023-05-26T03:10:34.423656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof.values.reshape((-1))","metadata":{"execution":{"iopub.status.busy":"2023-05-26T03:10:34.427652Z","iopub.execute_input":"2023-05-26T03:10:34.428136Z","iopub.status.idle":"2023-05-26T03:10:34.437444Z","shell.execute_reply.started":"2023-05-26T03:10:34.428093Z","shell.execute_reply":"2023-05-26T03:10:34.436225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"true","metadata":{"execution":{"iopub.status.busy":"2023-05-26T03:10:34.444719Z","iopub.execute_input":"2023-05-26T03:10:34.445073Z","iopub.status.idle":"2023-05-26T03:10:34.465520Z","shell.execute_reply.started":"2023-05-26T03:10:34.445040Z","shell.execute_reply":"2023-05-26T03:10:34.464892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores= []\nthresholds = []\nbest_score =0\nbest_thresh=0\n\nfor thresh in np.arange(0.3,0.9,0.01):\n    print(f'{thresh}, ',end='')\n    preds = (oof.values.reshape((-1))>thresh).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')  \n    scores.append(m)\n    thresholds.append(thresh)\n    if m>best_score:\n        best_score = m\n        best_thresh = thresh","metadata":{"execution":{"iopub.status.busy":"2023-05-26T03:10:34.466559Z","iopub.execute_input":"2023-05-26T03:10:34.467416Z","iopub.status.idle":"2023-05-26T03:10:41.904863Z","shell.execute_reply.started":"2023-05-26T03:10:34.467384Z","shell.execute_reply":"2023-05-26T03:10:41.903964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_thresh], [best_score], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score:.3f} at Best Threshold = {best_thresh:.3}',size=18)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T03:10:41.906338Z","iopub.execute_input":"2023-05-26T03:10:41.906692Z","iopub.status.idle":"2023-05-26T03:10:42.225325Z","shell.execute_reply.started":"2023-05-26T03:10:41.906658Z","shell.execute_reply":"2023-05-26T03:10:42.224435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('When using optimal threshold...')\nfor k in range(18):\n        \n    # COMPUTE F1 SCORE PER QUESTION\n    m = f1_score(true[k].values, (oof[k].values>best_thresh).astype('int'), average='macro')\n    print(f'Q{k}: F1 =',m)\n    \n# COMPUTE F1 SCORE OVERALL\nm = f1_score(true.values.reshape((-1)), (oof.values.reshape((-1))>best_thresh).astype('int'), average='macro')\nprint('==> Overall F1 =',m)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T03:10:42.226598Z","iopub.execute_input":"2023-05-26T03:10:42.227757Z","iopub.status.idle":"2023-05-26T03:10:42.470282Z","shell.execute_reply.started":"2023-05-26T03:10:42.227722Z","shell.execute_reply":"2023-05-26T03:10:42.469290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# IMPORT KAGGLE API\nimport jo_wilder_310\nenv = jo_wilder_310.make_env()\niter_test = env.iter_test()\n\n# CLEAR MEMORY\nimport gc\ndel targets, df, oof, true\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T03:10:42.471579Z","iopub.execute_input":"2023-05-26T03:10:42.472136Z","iopub.status.idle":"2023-05-26T03:10:42.852905Z","shell.execute_reply.started":"2023-05-26T03:10:42.472101Z","shell.execute_reply":"2023-05-26T03:10:42.851976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n    \n    # FEATURE ENGINEER TEST DATA\n    df = feature_engineering(test)\n    \n    # INFER TEST DATA\n    grp = test.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        clf = models[f'{grp}_{t}']\n        p = clf.predict(df[FEATURES].astype('float32'))[0]\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        sample_submission.loc[mask,'correct'] = int( p > best_thresh )\n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T03:10:42.855730Z","iopub.execute_input":"2023-05-26T03:10:42.856107Z","iopub.status.idle":"2023-05-26T03:10:47.917599Z","shell.execute_reply.started":"2023-05-26T03:10:42.856074Z","shell.execute_reply":"2023-05-26T03:10:47.916708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('submission.csv')\nprint( df.shape )\ndf","metadata":{"execution":{"iopub.status.busy":"2023-05-26T03:37:17.245914Z","iopub.execute_input":"2023-05-26T03:37:17.246590Z","iopub.status.idle":"2023-05-26T03:37:17.265820Z","shell.execute_reply.started":"2023-05-26T03:37:17.246556Z","shell.execute_reply":"2023-05-26T03:37:17.263144Z"},"trusted":true},"execution_count":null,"outputs":[]}]}