{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nimport os\n\nimport pandas as pd\nimport numpy as np\nimport warnings\nimport pickle\nimport polars as pl\nfrom collections import defaultdict\nfrom itertools import combinations\nimport pyarrow as pa\n\nimport lightgbm as lgb\nfrom lightgbm import LGBMClassifier\nfrom lightgbm import Booster\nfrom lightgbm import early_stopping\nfrom lightgbm import log_evaluation\n\nfrom sklearn.model_selection import GroupKFold, KFold, train_test_split\nfrom sklearn.metrics import roc_auc_score, f1_score\n\n\nimport matplotlib.pyplot as plt\nfrom colorama import Fore, Back, Style","metadata":{"execution":{"iopub.status.busy":"2023-05-27T08:17:36.064188Z","iopub.execute_input":"2023-05-27T08:17:36.064620Z","iopub.status.idle":"2023-05-27T08:17:38.228021Z","shell.execute_reply.started":"2023-05-27T08:17:36.064571Z","shell.execute_reply":"2023-05-27T08:17:38.226935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dtypes = {\"session_id\": pl.Int64,\"elapsed_time\": pl.Int64,\"event_name\": pl.Categorical,\n                \"name\": pl.Categorical,\"level\": pl.Int8,\"page\": pl.Float32,\n                \"room_coor_x\": pl.Float32,\"room_coor_y\": pl.Float32,\"screen_coor_x\": pl.Float32,\n                \"screen_coor_y\": pl.Float32,\"hover_duration\": pl.Float32,\"text\": pl.Utf8,\n                \"fqid\": pl.Utf8,\"room_fqid\": pl.Categorical,\"text_fqid\": pl.Utf8,\n                \"fullscreen\": pl.Int8,\"hq\": pl.Int8,\"music\": pl.Int8,\"level_group\": pl.Categorical\n               }\n\ntest_dtypes = {\"session_id\": pl.Int64,\"elapsed_time\": pl.Int64,\"event_name\": pl.Categorical,\n                \"name\": pl.Categorical,\"level\": pl.Int8,\"page\": pl.Float32,\n                \"room_coor_x\": pl.Float32,\"room_coor_y\": pl.Float32,\"screen_coor_x\": pl.Float32,\n                \"screen_coor_y\": pl.Float32,\"hover_duration\": pl.Float32,\"text\": pl.Utf8,\n                \"fqid\": pl.Utf8,\"room_fqid\": pl.Categorical,\"text_fqid\": pl.Utf8,\n                \"fullscreen\": pl.Int8,\"hq\": pl.Int8,\"music\": pl.Int8,\"level_group\": pl.Categorical,\n               \"session_level\":pl.Int8}","metadata":{"execution":{"iopub.status.busy":"2023-05-27T07:25:27.518980Z","iopub.execute_input":"2023-05-27T07:25:27.519368Z","iopub.status.idle":"2023-05-27T07:25:27.529332Z","shell.execute_reply.started":"2023-05-27T07:25:27.519339Z","shell.execute_reply":"2023-05-27T07:25:27.528140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATS = ['event_name', 'name', 'fqid', 'room_fqid', 'text_fqid']\nNUMS = ['page', 'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y',\n        'hover_duration', 'elapsed_time_diff']\nDIALOGS = ['that', 'this', 'it', 'you','find','found','Found','notebook','Wells','wells','help','need', 'Oh','Ooh','Jo', 'flag', 'can','and','is','the','to']\n\nname_feature = ['basic', 'undefined', 'close', 'open', 'prev', 'next']\nevent_name_feature = ['cutscene_click', 'person_click', 'navigate_click',\n       'observation_click', 'notification_click', 'object_click',\n       'object_hover', 'map_hover', 'map_click', 'checkpoint',\n       'notebook_click']\n\n# from https://www.kaggle.com/code/leehomhuang/catboost-baseline-with-lots-features-inference :\nfqid_lists = ['worker', 'archivist', 'gramps', 'wells', 'toentry', 'confrontation', 'crane_ranger', 'groupconvo', 'flag_girl', 'tomap', 'tostacks', 'tobasement', 'archivist_glasses', 'boss', 'journals', 'seescratches', 'groupconvo_flag', 'cs', 'teddy', 'expert', 'businesscards', 'ch3start', 'tunic.historicalsociety', 'tofrontdesk', 'savedteddy', 'plaque', 'glasses', 'tunic.drycleaner', 'reader_flag', 'tunic.library', 'tracks', 'tunic.capitol_2', 'trigger_scarf', 'reader', 'directory', 'tunic.capitol_1', 'journals.pic_0.next', 'unlockdoor', 'tunic', 'what_happened', 'tunic.kohlcenter', 'tunic.humanecology', 'colorbook', 'logbook', 'businesscards.card_0.next', 'journals.hub.topics', 'logbook.page.bingo', 'journals.pic_1.next', 'journals_flag', 'reader.paper0.next', 'tracks.hub.deer', 'reader_flag.paper0.next', 'trigger_coffee', 'wellsbadge', 'journals.pic_2.next', 'tomicrofiche', 'journals_flag.pic_0.bingo', 'plaque.face.date', 'notebook', 'tocloset_dirty', 'businesscards.card_bingo.bingo', 'businesscards.card_1.next', 'tunic.wildlife', 'tunic.hub.slip', 'tocage', 'journals.pic_2.bingo', 'tocollectionflag', 'tocollection', 'chap4_finale_c', 'chap2_finale_c', 'lockeddoor', 'journals_flag.hub.topics', 'tunic.capitol_0', 'reader_flag.paper2.bingo', 'photo', 'tunic.flaghouse', 'reader.paper1.next', 'directory.closeup.archivist', 'intro', 'businesscards.card_bingo.next', 'reader.paper2.bingo', 'retirement_letter', 'remove_cup', 'journals_flag.pic_0.next', 'magnify', 'coffee', 'key', 'togrampa', 'reader_flag.paper1.next', 'janitor', 'tohallway', 'chap1_finale', 'report', 'outtolunch', 'journals_flag.hub.topics_old', 'journals_flag.pic_1.next', 'reader.paper2.next', 'chap1_finale_c', 'reader_flag.paper2.next', 'door_block_talk', 'journals_flag.pic_1.bingo', 'journals_flag.pic_2.next', 'journals_flag.pic_2.bingo', 'block_magnify', 'reader.paper0.prev', 'block', 'reader_flag.paper0.prev', 'block_0', 'door_block_clean', 'reader.paper2.prev', 'reader.paper1.prev', 'doorblock', 'tocloset', 'reader_flag.paper2.prev', 'reader_flag.paper1.prev', 'block_tomap2', 'journals_flag.pic_0_old.next', 'journals_flag.pic_1_old.next', 'block_tocollection', 'block_nelson', 'journals_flag.pic_2_old.next', 'block_tomap1', 'block_badge', 'need_glasses', 'block_badge_2', 'fox', 'block_1']\ntext_lists = ['tunic.historicalsociety.cage.confrontation', 'tunic.wildlife.center.crane_ranger.crane', 'tunic.historicalsociety.frontdesk.archivist.newspaper', 'tunic.historicalsociety.entry.groupconvo', 'tunic.wildlife.center.wells.nodeer', 'tunic.historicalsociety.frontdesk.archivist.have_glass', 'tunic.drycleaner.frontdesk.worker.hub', 'tunic.historicalsociety.closet_dirty.gramps.news', 'tunic.humanecology.frontdesk.worker.intro', 'tunic.historicalsociety.frontdesk.archivist_glasses.confrontation', 'tunic.historicalsociety.basement.seescratches', 'tunic.historicalsociety.collection.cs', 'tunic.flaghouse.entry.flag_girl.hello', 'tunic.historicalsociety.collection.gramps.found', 'tunic.historicalsociety.basement.ch3start', 'tunic.historicalsociety.entry.groupconvo_flag', 'tunic.library.frontdesk.worker.hello', 'tunic.library.frontdesk.worker.wells', 'tunic.historicalsociety.collection_flag.gramps.flag', 'tunic.historicalsociety.basement.savedteddy', 'tunic.library.frontdesk.worker.nelson', 'tunic.wildlife.center.expert.removed_cup', 'tunic.library.frontdesk.worker.flag', 'tunic.historicalsociety.frontdesk.archivist.hello', 'tunic.historicalsociety.closet.gramps.intro_0_cs_0', 'tunic.historicalsociety.entry.boss.flag', 'tunic.flaghouse.entry.flag_girl.symbol', 'tunic.historicalsociety.closet_dirty.trigger_scarf', 'tunic.drycleaner.frontdesk.worker.done', 'tunic.historicalsociety.closet_dirty.what_happened', 'tunic.wildlife.center.wells.animals', 'tunic.historicalsociety.closet.teddy.intro_0_cs_0', 'tunic.historicalsociety.cage.glasses.afterteddy', 'tunic.historicalsociety.cage.teddy.trapped', 'tunic.historicalsociety.cage.unlockdoor', 'tunic.historicalsociety.stacks.journals.pic_2.bingo', 'tunic.historicalsociety.entry.wells.flag', 'tunic.humanecology.frontdesk.worker.badger', 'tunic.historicalsociety.stacks.journals_flag.pic_0.bingo', 'tunic.historicalsociety.closet.intro', 'tunic.historicalsociety.closet.retirement_letter.hub', 'tunic.historicalsociety.entry.directory.closeup.archivist', 'tunic.historicalsociety.collection.tunic.slip', 'tunic.kohlcenter.halloffame.plaque.face.date', 'tunic.historicalsociety.closet_dirty.trigger_coffee', 'tunic.drycleaner.frontdesk.logbook.page.bingo', 'tunic.library.microfiche.reader.paper2.bingo', 'tunic.kohlcenter.halloffame.togrampa', 'tunic.capitol_2.hall.boss.haveyougotit', 'tunic.wildlife.center.wells.nodeer_recap', 'tunic.historicalsociety.cage.glasses.beforeteddy', 'tunic.historicalsociety.closet_dirty.gramps.helpclean', 'tunic.wildlife.center.expert.recap', 'tunic.historicalsociety.frontdesk.archivist.have_glass_recap', 'tunic.historicalsociety.stacks.journals_flag.pic_1.bingo', 'tunic.historicalsociety.cage.lockeddoor', 'tunic.historicalsociety.stacks.journals_flag.pic_2.bingo', 'tunic.historicalsociety.collection.gramps.lost', 'tunic.historicalsociety.closet.notebook', 'tunic.historicalsociety.frontdesk.magnify', 'tunic.humanecology.frontdesk.businesscards.card_bingo.bingo', 'tunic.wildlife.center.remove_cup', 'tunic.library.frontdesk.wellsbadge.hub', 'tunic.wildlife.center.tracks.hub.deer', 'tunic.historicalsociety.frontdesk.key', 'tunic.library.microfiche.reader_flag.paper2.bingo', 'tunic.flaghouse.entry.colorbook', 'tunic.wildlife.center.coffee', 'tunic.capitol_1.hall.boss.haveyougotit', 'tunic.historicalsociety.basement.janitor', 'tunic.historicalsociety.collection_flag.gramps.recap', 'tunic.wildlife.center.wells.animals2', 'tunic.flaghouse.entry.flag_girl.symbol_recap', 'tunic.historicalsociety.closet_dirty.photo', 'tunic.historicalsociety.stacks.outtolunch', 'tunic.library.frontdesk.worker.wells_recap', 'tunic.historicalsociety.frontdesk.archivist_glasses.confrontation_recap', 'tunic.capitol_0.hall.boss.talktogramps', 'tunic.historicalsociety.closet.photo', 'tunic.historicalsociety.collection.tunic', 'tunic.historicalsociety.closet.teddy.intro_0_cs_5', 'tunic.historicalsociety.closet_dirty.gramps.archivist', 'tunic.historicalsociety.closet_dirty.door_block_talk', 'tunic.historicalsociety.entry.boss.flag_recap', 'tunic.historicalsociety.frontdesk.archivist.need_glass_0', 'tunic.historicalsociety.entry.wells.talktogramps', 'tunic.historicalsociety.frontdesk.block_magnify', 'tunic.historicalsociety.frontdesk.archivist.foundtheodora', 'tunic.historicalsociety.closet_dirty.gramps.nothing', 'tunic.historicalsociety.closet_dirty.door_block_clean', 'tunic.capitol_1.hall.boss.writeitup', 'tunic.library.frontdesk.worker.nelson_recap', 'tunic.library.frontdesk.worker.hello_short', 'tunic.historicalsociety.stacks.block', 'tunic.historicalsociety.frontdesk.archivist.need_glass_1', 'tunic.historicalsociety.entry.boss.talktogramps', 'tunic.historicalsociety.frontdesk.archivist.newspaper_recap', 'tunic.historicalsociety.entry.wells.flag_recap', 'tunic.drycleaner.frontdesk.worker.done2', 'tunic.library.frontdesk.worker.flag_recap', 'tunic.humanecology.frontdesk.block_0', 'tunic.library.frontdesk.worker.preflag', 'tunic.historicalsociety.basement.gramps.seeyalater', 'tunic.flaghouse.entry.flag_girl.hello_recap', 'tunic.historicalsociety.closet.doorblock', 'tunic.drycleaner.frontdesk.worker.takealook', 'tunic.historicalsociety.basement.gramps.whatdo', 'tunic.library.frontdesk.worker.droppedbadge', 'tunic.historicalsociety.entry.block_tomap2', 'tunic.library.frontdesk.block_nelson', 'tunic.library.microfiche.block_0', 'tunic.historicalsociety.entry.block_tocollection', 'tunic.historicalsociety.entry.block_tomap1', 'tunic.historicalsociety.collection.gramps.look_0', 'tunic.library.frontdesk.block_badge', 'tunic.historicalsociety.cage.need_glasses', 'tunic.library.frontdesk.block_badge_2', 'tunic.kohlcenter.halloffame.block_0', 'tunic.capitol_0.hall.chap1_finale_c', 'tunic.capitol_1.hall.chap2_finale_c', 'tunic.capitol_2.hall.chap4_finale_c', 'tunic.wildlife.center.fox.concern', 'tunic.drycleaner.frontdesk.block_0', 'tunic.historicalsociety.entry.gramps.hub', 'tunic.humanecology.frontdesk.block_1', 'tunic.drycleaner.frontdesk.block_1']\nroom_lists = ['tunic.historicalsociety.entry', 'tunic.wildlife.center', 'tunic.historicalsociety.cage', 'tunic.library.frontdesk', 'tunic.historicalsociety.frontdesk', 'tunic.historicalsociety.stacks', 'tunic.historicalsociety.closet_dirty', 'tunic.humanecology.frontdesk', 'tunic.historicalsociety.basement', 'tunic.kohlcenter.halloffame', 'tunic.library.microfiche', 'tunic.drycleaner.frontdesk', 'tunic.historicalsociety.collection', 'tunic.historicalsociety.closet', 'tunic.flaghouse.entry', 'tunic.historicalsociety.collection_flag', 'tunic.capitol_1.hall', 'tunic.capitol_0.hall', 'tunic.capitol_2.hall']\nLEVELS = [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22]\nlevel_groups = [\"0-4\", \"5-12\", \"13-22\"]","metadata":{"execution":{"iopub.status.busy":"2023-05-27T07:25:42.442825Z","iopub.execute_input":"2023-05-27T07:25:42.443208Z","iopub.status.idle":"2023-05-27T07:25:42.469794Z","shell.execute_reply.started":"2023-05-27T07:25:42.443179Z","shell.execute_reply":"2023-05-27T07:25:42.468435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = [\n\n    pl.col(\"page\").cast(pl.Float32),\n    (\n        (pl.col(\"elapsed_time\") - pl.col(\"elapsed_time\").shift(1)) \n         .fill_null(0)\n         .clip(0, 1e9)\n         .over([\"session_id\", \"level_group\"])\n         .alias(\"elapsed_time_diff\")\n    ),\n    (\n        (pl.col(\"screen_coor_x\") - pl.col(\"screen_coor_x\").shift(1)) \n         .abs()\n         .over([\"session_id\", \"level_group\"])\n        .alias(\"location_x_diff\") \n    ),\n    (\n        (pl.col(\"screen_coor_y\") - pl.col(\"screen_coor_y\").shift(1)) \n         .abs()\n         .over([\"session_id\", \"level_group\"])\n        .alias(\"location_y_diff\") \n    ),\n    pl.col(\"fqid\").fill_null(\"fqid_None\"),\n    pl.col(\"text_fqid\").fill_null(\"text_fqid_None\")\n]","metadata":{"execution":{"iopub.status.busy":"2023-05-27T07:25:50.375672Z","iopub.execute_input":"2023-05-27T07:25:50.376075Z","iopub.status.idle":"2023-05-27T07:25:50.404124Z","shell.execute_reply.started":"2023-05-27T07:25:50.376044Z","shell.execute_reply":"2023-05-27T07:25:50.402934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage_pl(df):\n    \n    start_mem = df.estimated_size(\"mb\")\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    # pl.Uint8,pl.UInt16,pl.UInt32,pl.UInt64\n    Numeric_Int_types = [pl.Int8,pl.Int16,pl.Int32,pl.Int64]\n    Numeric_Float_types = [pl.Float32,pl.Float64]\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        c_min = df[col].min()\n        c_max = df[col].max()\n        if col_type in Numeric_Int_types:\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                df = df.with_columns(df[col].cast(pl.Int8))\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                df = df.with_columns(df[col].cast(pl.Int16))\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                df = df.with_columns(df[col].cast(pl.Int32))\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                df = df.with_columns(df[col].cast(pl.Int64))\n\n        elif col_type in Numeric_Float_types:\n            if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                df = df.with_columns(df[col].cast(pl.Float32))\n            else:\n                pass\n        elif col_type == pl.Utf8:\n            df = df.with_columns(df[col].cast(pl.Categorical))\n        else:\n            pass\n    mem_usg = df.estimated_size(\"mb\")\n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df\n\ndef feature_engineer_pl(x, grp, use_extra, feature_suffix):\n    aggs = [\n        pl.col(\"index\").count().alias(f\"session_number_{feature_suffix}\"),\n\n        *[pl.col('index').filter(pl.col('text').str.contains(c)).count().alias(f'word_{c}') for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).mean().alias(f'word_mean_{c}') for c in\n          DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).std().alias(f'word_std_{c}') for c in\n          DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).max().alias(f'word_max_{c}') for c in\n          DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).sum().alias(f'word_sum_{c}') for c in\n          DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).median().alias(f'word_median_{c}') for c\n          in DIALOGS],\n\n        *[pl.col(c).drop_nulls().n_unique().alias(f\"{c}_unique_{feature_suffix}\") for c in CATS],\n\n        *[pl.col(c).mean().alias(f\"{c}_mean_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).std().alias(f\"{c}_std_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).min().alias(f\"{c}_min_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).max().alias(f\"{c}_max_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).median().alias(f\"{c}_median_{feature_suffix}\") for c in NUMS],\n\n        *[pl.col(\"fqid\").filter(pl.col(\"fqid\") == c).count().alias(f\"{c}_fqid_counts{feature_suffix}\")\n          for c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for\n          c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\") for\n          c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in fqid_lists],\n\n        *[pl.col(\"text_fqid\").filter(pl.col(\"text_fqid\") == c).count().alias(f\"{c}_text_fqid_counts{feature_suffix}\")\n          for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\")\n          for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in text_lists],\n\n        *[pl.col(\"room_fqid\").filter(pl.col(\"room_fqid\") == c).count().alias(f\"{c}_room_fqid_counts{feature_suffix}\")\n          for c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for\n          c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\")\n          for\n          c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in room_lists],\n\n        *[pl.col(\"event_name\").filter(pl.col(\"event_name\") == c).count().alias(f\"{c}_event_name_counts{feature_suffix}\")\n          for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\")\n          for\n          c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).median().alias(\n            f\"{c}_ET_median_{feature_suffix}\") for\n          c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in event_name_feature],\n\n        *[pl.col(\"name\").filter(pl.col(\"name\") == c).count().alias(f\"{c}_name_counts{feature_suffix}\") for c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\") for\n          c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in\n          name_feature],\n\n        *[pl.col(\"level\").filter(pl.col(\"level\") == c).count().alias(f\"{c}_LEVEL_count{feature_suffix}\") for c in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c\n          in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for c in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\") for\n          c in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in\n          LEVELS],\n\n        *[pl.col(\"level_group\").filter(pl.col(\"level_group\") == c).count().alias(\n            f\"{c}_LEVEL_group_count{feature_suffix}\") for c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\")\n          for c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).median().alias(\n            f\"{c}_ET_median_{feature_suffix}\") for c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in\n          level_groups],\n\n    ]\n\n    df = x.groupby(['session_id'], maintain_order=True).agg(aggs).sort(\"session_id\")\n\n    if use_extra:\n        if grp == '5-12':\n            aggs = [\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Here's the log book.\")\n                                              | (pl.col(\"fqid\") == 'logbook.page.bingo'))\n                    .apply(lambda s: s.max() - s.min()).alias(\"logbook_bingo_duration\"),\n                pl.col(\"index\").filter(\n                    (pl.col(\"text\") == \"Here's the log book.\") | (pl.col(\"fqid\") == 'logbook.page.bingo')).apply(\n                    lambda s: s.max() - s.min()).alias(\"logbook_bingo_indexCount\"),\n                pl.col(\"elapsed_time\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader')) | (\n                            pl.col(\"fqid\") == \"reader.paper2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\n                    \"reader_bingo_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader')) | (\n                        pl.col(\"fqid\") == \"reader.paper2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\n                    \"reader_bingo_indexCount\"),\n                pl.col(\"elapsed_time\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals')) | (\n                            pl.col(\"fqid\") == \"journals.pic_2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\n                    \"journals_bingo_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals')) | (\n                        pl.col(\"fqid\") == \"journals.pic_2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\n                    \"journals_bingo_indexCount\"),\n            ]\n            tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n            df = df.join(tmp, on=\"session_id\", how='left')\n\n        if grp == '13-22':\n            aggs = [\n                pl.col(\"elapsed_time\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader_flag')) | (\n                            pl.col(\"fqid\") == \"tunic.library.microfiche.reader_flag.paper2.bingo\")).apply(\n                    lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"reader_flag_duration\"),\n                pl.col(\"index\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader_flag')) | (\n                            pl.col(\"fqid\") == \"tunic.library.microfiche.reader_flag.paper2.bingo\")).apply(\n                    lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"reader_flag_indexCount\"),\n                pl.col(\"elapsed_time\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals_flag')) | (\n                            pl.col(\"fqid\") == \"journals_flag.pic_0.bingo\")).apply(\n                    lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"journalsFlag_bingo_duration\"),\n                pl.col(\"index\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals_flag')) | (\n                            pl.col(\"fqid\") == \"journals_flag.pic_0.bingo\")).apply(\n                    lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"journalsFlag_bingo_indexCount\")\n            ]\n            tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n            df = df.join(tmp, on=\"session_id\", how='left')\n\n    return df.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T07:26:08.789730Z","iopub.execute_input":"2023-05-27T07:26:08.790126Z","iopub.status.idle":"2023-05-27T07:26:08.866788Z","shell.execute_reply.started":"2023-05-27T07:26:08.790097Z","shell.execute_reply":"2023-05-27T07:26:08.865600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Processing","metadata":{}},{"cell_type":"code","source":"%%time\n\n# we prepare the dataset for the training by level :\ndf = (pl.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\",dtypes=train_dtypes)\n      .drop([\"fullscreen\", \"hq\", \"music\"])\n      .with_columns(columns))\ndf = reduce_mem_usage_pl(df)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T07:27:48.850136Z","iopub.execute_input":"2023-05-27T07:27:48.850520Z","iopub.status.idle":"2023-05-27T07:28:48.370249Z","shell.execute_reply.started":"2023-05-27T07:27:48.850492Z","shell.execute_reply":"2023-05-27T07:28:48.369187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.with_columns(df[\"text\"].cast(pl.Utf8))","metadata":{"execution":{"iopub.status.busy":"2023-05-27T07:29:00.538510Z","iopub.execute_input":"2023-05-27T07:29:00.538906Z","iopub.status.idle":"2023-05-27T07:29:02.050361Z","shell.execute_reply.started":"2023-05-27T07:29:00.538877Z","shell.execute_reply":"2023-05-27T07:29:02.049299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 按level分组\ndf1 = df.filter(pl.col(\"level_group\")=='0-4')\ndf2 = df.filter(pl.col(\"level_group\")=='5-12')\ndf3 = df.filter(pl.col(\"level_group\")=='13-22')\ndf1.shape,df2.shape,df3.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-27T07:29:10.261198Z","iopub.execute_input":"2023-05-27T07:29:10.261583Z","iopub.status.idle":"2023-05-27T07:29:12.366513Z","shell.execute_reply.started":"2023-05-27T07:29:10.261552Z","shell.execute_reply":"2023-05-27T07:29:12.365716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T07:29:21.506697Z","iopub.execute_input":"2023-05-27T07:29:21.507073Z","iopub.status.idle":"2023-05-27T07:29:21.928539Z","shell.execute_reply.started":"2023-05-27T07:29:21.507044Z","shell.execute_reply":"2023-05-27T07:29:21.927694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf1 = feature_engineer_pl(df1, grp='0-4', use_extra=True, feature_suffix='')\nprint('df1 done',df1.shape)\ndf2 = feature_engineer_pl(df2, grp='5-12', use_extra=True, feature_suffix='')\nprint('df2 done',df2.shape)\ndf3 = feature_engineer_pl(df3, grp='13-22', use_extra=True, feature_suffix='')\nprint('df3 done',df3.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T07:29:33.769826Z","iopub.execute_input":"2023-05-27T07:29:33.770207Z","iopub.status.idle":"2023-05-27T07:31:36.672464Z","shell.execute_reply.started":"2023-05-27T07:29:33.770179Z","shell.execute_reply":"2023-05-27T07:31:36.671621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# some cleaning...\n# 这部分清洗可以直接用，实际上是把一些特征稀疏的样本去除掉\nnull1 = df1.isnull().sum().sort_values(ascending=False) / len(df1)\nnull2 = df2.isnull().sum().sort_values(ascending=False) / len(df1)\nnull3 = df3.isnull().sum().sort_values(ascending=False) / len(df1)\n\ndrop1 = list(null1[null1>0.9].index)\ndrop2 = list(null2[null2>0.9].index)\ndrop3 = list(null3[null3>0.9].index)\nprint(len(drop1), len(drop2), len(drop3))\n\nfor col in df1.columns:\n    if df1[col].nunique()==1:\n        print(col)\n        drop1.append(col)\nprint(\"*********df1 DONE*********\")\nfor col in df2.columns:\n    if df2[col].nunique()==1:\n        print(col)\n        drop2.append(col)\nprint(\"*********df2 DONE*********\")\nfor col in df3.columns:\n    if df3[col].nunique()==1:\n        print(col)\n        drop3.append(col)\nprint(\"*********df3 DONE*********\")","metadata":{"execution":{"iopub.status.busy":"2023-05-27T07:33:54.742143Z","iopub.execute_input":"2023-05-27T07:33:54.742509Z","iopub.status.idle":"2023-05-27T07:33:57.797503Z","shell.execute_reply.started":"2023-05-27T07:33:54.742482Z","shell.execute_reply":"2023-05-27T07:33:57.796696Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def time_feature(train):\n    \n    train[\"year\"] = train[\"session_id\"].apply(lambda x: int(str(x)[:2])).astype(np.uint8)\n    train[\"month\"] = train[\"session_id\"].apply(lambda x: int(str(x)[2:4])+1).astype(np.uint8)\n    train[\"day\"] = train[\"session_id\"].apply(lambda x: int(str(x)[4:6])).astype(np.uint8)\n    train[\"hour\"] = train[\"session_id\"].apply(lambda x: int(str(x)[6:8])).astype(np.uint8)\n    train[\"minute\"] = train[\"session_id\"].apply(lambda x: int(str(x)[8:10])).astype(np.uint8)\n    train[\"second\"] = train[\"session_id\"].apply(lambda x: int(str(x)[10:12])).astype(np.uint8)\n\n    return train","metadata":{"execution":{"iopub.status.busy":"2023-05-27T07:34:16.212481Z","iopub.execute_input":"2023-05-27T07:34:16.212878Z","iopub.status.idle":"2023-05-27T07:34:16.223340Z","shell.execute_reply.started":"2023-05-27T07:34:16.212847Z","shell.execute_reply":"2023-05-27T07:34:16.221852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 根据session生成时间特征\ndf1 = time_feature(df1)\ndf2 = time_feature(df2)\ndf3 = time_feature(df3)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T07:34:23.593355Z","iopub.execute_input":"2023-05-27T07:34:23.593761Z","iopub.status.idle":"2023-05-27T07:34:24.156950Z","shell.execute_reply.started":"2023-05-27T07:34:23.593729Z","shell.execute_reply":"2023-05-27T07:34:24.156053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 获取所有的feature\ndf1 = df1.set_index('session_id')\ndf2 = df2.set_index('session_id')\ndf3 = df3.set_index('session_id')\n\nFEATURES1 = [c for c in df1.columns if c not in drop1+['level_group']]\nFEATURES2 = [c for c in df2.columns if c not in drop2+['level_group']]\nFEATURES3 = [c for c in df3.columns if c not in drop3+['level_group']]\nprint('We will train with', len(FEATURES1), len(FEATURES2), len(FEATURES3) ,'features')\nALL_USERS = df1.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"execution":{"iopub.status.busy":"2023-05-27T07:34:30.984552Z","iopub.execute_input":"2023-05-27T07:34:30.984943Z","iopub.status.idle":"2023-05-27T07:34:31.694444Z","shell.execute_reply.started":"2023-05-27T07:34:30.984914Z","shell.execute_reply":"2023-05-27T07:34:31.693529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1 = open('FEATURES1.txt', 'wb')\npickle.dump(FEATURES1, f1)\nf2 = open('FEATURES2.txt', 'wb')\npickle.dump(FEATURES2, f2)\nf3 = open('FEATURES3.txt', 'wb')\npickle.dump(FEATURES3, f3)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T07:34:40.087551Z","iopub.execute_input":"2023-05-27T07:34:40.087937Z","iopub.status.idle":"2023-05-27T07:34:40.093785Z","shell.execute_reply.started":"2023-05-27T07:34:40.087908Z","shell.execute_reply":"2023-05-27T07:34:40.093042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 确定不同session的迭代次数\n# With previous training notebook (Kfold with 20 folds as performed in others notebooks) :\nestimators_lgb = [498, 448, 378, 364, 405, 495, 456, 249, 384, 405, 356, 262, 484, 381, 392, 248 ,248, 345]","metadata":{"execution":{"iopub.status.busy":"2023-05-27T07:35:40.295756Z","iopub.execute_input":"2023-05-27T07:35:40.296262Z","iopub.status.idle":"2023-05-27T07:35:40.302294Z","shell.execute_reply.started":"2023-05-27T07:35:40.296220Z","shell.execute_reply":"2023-05-27T07:35:40.301452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 模型训练参数\nlgb_params = {\n    'boosting_type': 'gbdt',\n    'objective': 'binary',\n    'metric': 'binary_logloss',\n    'learning_rate': 0.05,\n    'alpha': 8,\n    'max_depth': 4,\n    'subsample': 0.8,\n    'colsample_bytree': 0.5,\n    'random_state': 42\n}","metadata":{"execution":{"iopub.status.busy":"2023-05-27T07:35:57.045898Z","iopub.execute_input":"2023-05-27T07:35:57.046267Z","iopub.status.idle":"2023-05-27T07:35:57.051713Z","shell.execute_reply.started":"2023-05-27T07:35:57.046238Z","shell.execute_reply":"2023-05-27T07:35:57.050730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 获取label\nwarnings.filterwarnings(\"ignore\")\ntargets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\n# 给label补充session和q信息\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]))\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]))\npred_lgb = np.zeros((df1.shape[0], 18))\n\n#交叉验证\nn_splits = 5\nkf = KFold(n_splits=n_splits)\n\n#不懂level的训练信息获取\nfor q in range(1, 19):\n    # USE THIS TRAIN DATA WITH THESE QUESTIONS\n    if q <= 3:\n        grp = '0-4'\n        df = df1\n        FEATURES = FEATURES1\n    elif q <= 13:\n        grp = '5-12'\n        df = df2\n        FEATURES = FEATURES2\n    elif q <= 22:\n        grp = '13-22'\n        df = df3\n        FEATURES = FEATURES3\n\n    lgb_params['n_estimators'] = estimators_lgb[q - 1]\n\n    # TRAIN DATA\n    for fold, (train_idx, val_idx) in enumerate(kf.split(df)):\n        #确定当前样本的特征、用户和标签\n        df_train = df.iloc[train_idx] #.reset_index(drop=True)\n        train_users = df_train.index.values\n        train_y = targets[targets['session'].isin(list(train_users))].loc[targets.q == q].set_index('session')\n        \n        #验证集情况\n        df_val = df.iloc[val_idx] #.reset_index(drop=True)\n        val_users = df_val.index.values\n        val_y = targets[targets['session'].isin(list(val_users))].loc[targets.q == q].set_index('session')\n        \n        #LGBM模型训练\n        clf = LGBMClassifier(**lgb_params)\n        clf.fit(df_train[FEATURES].astype('float32'), train_y['correct'], verbose=0)\n\n        clf.booster_.save_model(f'LGBM_question{q}_fold{fold}.lgb')\n        print(f'Model saved for question {q} fold {fold} with iterations = {estimators_lgb[q-1]}')","metadata":{"execution":{"iopub.status.busy":"2023-05-27T07:38:15.956154Z","iopub.execute_input":"2023-05-27T07:38:15.956541Z","iopub.status.idle":"2023-05-27T08:08:06.543518Z","shell.execute_reply.started":"2023-05-27T07:38:15.956512Z","shell.execute_reply":"2023-05-27T08:08:06.542519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Test","metadata":{}},{"cell_type":"code","source":"import jo_wilder_310\nenv = jo_wilder_310.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T09:02:21.121048Z","iopub.execute_input":"2023-05-27T09:02:21.121422Z","iopub.status.idle":"2023-05-27T09:02:21.126697Z","shell.execute_reply.started":"2023-05-27T09:02:21.121372Z","shell.execute_reply":"2023-05-27T09:02:21.125501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models_list = [[Booster(model_file = f\"/kaggle/working/LGBM_question{q}_fold{fold}.lgb\") for fold in range(5)] for q in range(1, 19)]","metadata":{"execution":{"iopub.status.busy":"2023-05-27T08:16:50.245753Z","iopub.execute_input":"2023-05-27T08:16:50.246142Z","iopub.status.idle":"2023-05-27T08:16:50.253810Z","shell.execute_reply.started":"2023-05-27T08:16:50.246106Z","shell.execute_reply":"2023-05-27T08:16:50.252601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\ncount = 0\n\nfor (test, sample_submission) in iter_test:\n    session_id = test.session_id.values[0]\n    grp = test.level_group.values[0]\n    a, b = limits[grp]\n    preds = []\n\n    # ------------------- level 0-4 ---------------------------------\n    if a == 1:\n        FEATURES = FEATURES1\n        test = (pl.from_pandas(test)\n                .drop([\"fullscreen\", \"hq\", \"music\"])\n                .with_columns(columns))\n        test = feature_engineer_pl(test, grp, use_extra=True, feature_suffix='')\n        test = time_feature(test)\n        test = test[FEATURES]\n\n    # ------------------- level 5-12 ---------------------------------\n    elif a == 4:\n        FEATURES = FEATURES2\n        test = (pl.from_pandas(test)\n                .drop([\"fullscreen\", \"hq\", \"music\"])\n                .with_columns(columns))\n        test = feature_engineer_pl(test, grp, use_extra=True, feature_suffix='')\n        test = time_feature(test)\n        test = test[FEATURES]\n\n    # ------------------- level 13-22 ---------------------------------\n    elif a == 14:\n        FEATURES = FEATURES3\n        test = (pl.from_pandas(test)\n                .drop([\"fullscreen\", \"hq\", \"music\"])\n                .with_columns(columns))\n        test = feature_engineer_pl(test, grp, use_extra=True, feature_suffix='')\n        test = time_feature(test)\n        test = test[FEATURES]\n    \n    for q in range(a,b):\n        fold = 0\n        thresh = 0.63\n        model_0 = models_list[q-1][fold]\n        model_1 = models_list[q-1][fold+1]\n        model_2 = models_list[q-1][fold+2]\n        model_3 = models_list[q-1][fold+3]\n        model_4 = models_list[q-1][fold+4]\n        \n        pred_0 = model_0.predict(test[FEATURES].astype(np.float32))\n        pred_1 = model_1.predict(test[FEATURES].astype(np.float32))\n        pred_2 = model_2.predict(test[FEATURES].astype(np.float32))\n        pred_3 = model_3.predict(test[FEATURES].astype(np.float32))\n        pred_4 = model_4.predict(test[FEATURES].astype(np.float32))\n        \n        pred = (pred_0 + pred_1 + pred_2 + pred_3 + pred_4) / 5\n        preds.append(int(pred > thresh))\n\n    sample_submission[\"correct\"] = preds\n    env.predict(sample_submission)","metadata":{},"execution_count":null,"outputs":[]}]}