{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n","metadata":{}},{"cell_type":"markdown","source":"The notebook uses the preprocessing from From : https://www.kaggle.com/code/takanashihumbert/magic-bingo-train-part-lb-0-687\nand from https://www.kaggle.com/code/leehomhuang/catboost-baseline-with-lots-features-inference\nwith some new features.\n\nAnd ideas from https://www.kaggle.com/code/cdeotte/xgboost-baseline-0-676 as well.","metadata":{}},{"cell_type":"code","source":"!pip install /kaggle/input/polars-for-student/polars-0.16.9-cp37-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n!pip install /kaggle/input/polars-for-student/typing_extensions-4.5.0-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:11:10.993793Z","iopub.execute_input":"2023-05-22T03:11:10.994700Z","iopub.status.idle":"2023-05-22T03:12:16.455117Z","shell.execute_reply.started":"2023-05-22T03:11:10.994641Z","shell.execute_reply":"2023-05-22T03:12:16.453466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nimport os\n\nimport pandas as pd\nimport numpy as np\nimport warnings\nimport pickle\nimport polars as pl\n\nfrom collections import defaultdict\nfrom itertools import combinations\nimport pyarrow as pa\n\nimport lightgbm as lgb\nfrom lightgbm import LGBMClassifier\nfrom lightgbm import Booster\nfrom lightgbm import early_stopping\nfrom lightgbm import log_evaluation\n\nfrom sklearn.model_selection import GroupKFold, KFold, train_test_split\nfrom sklearn.metrics import roc_auc_score, f1_score\n\n\nimport matplotlib.pyplot as plt\nfrom colorama import Fore, Back, Style","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-22T03:12:16.465641Z","iopub.execute_input":"2023-05-22T03:12:16.466044Z","iopub.status.idle":"2023-05-22T03:12:18.791463Z","shell.execute_reply.started":"2023-05-22T03:12:16.465975Z","shell.execute_reply":"2023-05-22T03:12:18.789598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dtypes = {\"session_id\": pl.Int64,\"elapsed_time\": pl.Int64,\"event_name\": pl.Categorical,\n                \"name\": pl.Categorical,\"level\": pl.Int8,\"page\": pl.Float32,\n                \"room_coor_x\": pl.Float32,\"room_coor_y\": pl.Float32,\"screen_coor_x\": pl.Float32,\n                \"screen_coor_y\": pl.Float32,\"hover_duration\": pl.Float32,\"text\": pl.Utf8,\n                \"fqid\": pl.Utf8,\"room_fqid\": pl.Categorical,\"text_fqid\": pl.Utf8,\n                \"fullscreen\": pl.Int8,\"hq\": pl.Int8,\"music\": pl.Int8,\"level_group\": pl.Categorical\n               }\n\ntest_dtypes = {\"session_id\": pl.Int64,\"elapsed_time\": pl.Int64,\"event_name\": pl.Categorical,\n                \"name\": pl.Categorical,\"level\": pl.Int8,\"page\": pl.Float32,\n                \"room_coor_x\": pl.Float32,\"room_coor_y\": pl.Float32,\"screen_coor_x\": pl.Float32,\n                \"screen_coor_y\": pl.Float32,\"hover_duration\": pl.Float32,\"text\": pl.Utf8,\n                \"fqid\": pl.Utf8,\"room_fqid\": pl.Categorical,\"text_fqid\": pl.Utf8,\n                \"fullscreen\": pl.Int8,\"hq\": pl.Int8,\"music\": pl.Int8,\"level_group\": pl.Categorical,\n               \"session_level\":pl.Int8}","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:12:18.793288Z","iopub.execute_input":"2023-05-22T03:12:18.794945Z","iopub.status.idle":"2023-05-22T03:12:18.805488Z","shell.execute_reply.started":"2023-05-22T03:12:18.794877Z","shell.execute_reply":"2023-05-22T03:12:18.804352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage_pl(df):\n    \n    start_mem = df.estimated_size(\"mb\")\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    # pl.Uint8,pl.UInt16,pl.UInt32,pl.UInt64\n    Numeric_Int_types = [pl.Int8,pl.Int16,pl.Int32,pl.Int64]\n    Numeric_Float_types = [pl.Float32,pl.Float64]\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        c_min = df[col].min()\n        c_max = df[col].max()\n        if col_type in Numeric_Int_types:\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                df = df.with_columns(df[col].cast(pl.Int8))\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                df = df.with_columns(df[col].cast(pl.Int16))\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                df = df.with_columns(df[col].cast(pl.Int32))\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                df = df.with_columns(df[col].cast(pl.Int64))\n\n        elif col_type in Numeric_Float_types:\n            if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                df = df.with_columns(df[col].cast(pl.Float32))\n            else:\n                pass\n        elif col_type == pl.Utf8:\n            df = df.with_columns(df[col].cast(pl.Categorical))\n        else:\n            pass\n    mem_usg = df.estimated_size(\"mb\")\n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:12:18.808577Z","iopub.execute_input":"2023-05-22T03:12:18.809266Z","iopub.status.idle":"2023-05-22T03:12:18.824364Z","shell.execute_reply.started":"2023-05-22T03:12:18.809218Z","shell.execute_reply":"2023-05-22T03:12:18.823206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data preprocessing","metadata":{}},{"cell_type":"code","source":"CATS = ['event_name', 'name', 'fqid', 'room_fqid', 'text_fqid']\nNUMS = ['page', 'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y',\n        'hover_duration', 'elapsed_time_diff']\nDIALOGS = ['that', 'this', 'it', 'you','find','found','Found','notebook','Wells','wells','help','need', 'Oh','Ooh','Jo', 'flag', 'can','and','is','the','to']\n\nname_feature = ['basic', 'undefined', 'close', 'open', 'prev', 'next']\nevent_name_feature = ['cutscene_click', 'person_click', 'navigate_click',\n       'observation_click', 'notification_click', 'object_click',\n       'object_hover', 'map_hover', 'map_click', 'checkpoint',\n       'notebook_click']\n\n# from https://www.kaggle.com/code/leehomhuang/catboost-baseline-with-lots-features-inference :\nfqid_lists = ['worker', 'archivist', 'gramps', 'wells', 'toentry', 'confrontation', 'crane_ranger', 'groupconvo', 'flag_girl', 'tomap', 'tostacks', 'tobasement', 'archivist_glasses', 'boss', 'journals', 'seescratches', 'groupconvo_flag', 'cs', 'teddy', 'expert', 'businesscards', 'ch3start', 'tunic.historicalsociety', 'tofrontdesk', 'savedteddy', 'plaque', 'glasses', 'tunic.drycleaner', 'reader_flag', 'tunic.library', 'tracks', 'tunic.capitol_2', 'trigger_scarf', 'reader', 'directory', 'tunic.capitol_1', 'journals.pic_0.next', 'unlockdoor', 'tunic', 'what_happened', 'tunic.kohlcenter', 'tunic.humanecology', 'colorbook', 'logbook', 'businesscards.card_0.next', 'journals.hub.topics', 'logbook.page.bingo', 'journals.pic_1.next', 'journals_flag', 'reader.paper0.next', 'tracks.hub.deer', 'reader_flag.paper0.next', 'trigger_coffee', 'wellsbadge', 'journals.pic_2.next', 'tomicrofiche', 'journals_flag.pic_0.bingo', 'plaque.face.date', 'notebook', 'tocloset_dirty', 'businesscards.card_bingo.bingo', 'businesscards.card_1.next', 'tunic.wildlife', 'tunic.hub.slip', 'tocage', 'journals.pic_2.bingo', 'tocollectionflag', 'tocollection', 'chap4_finale_c', 'chap2_finale_c', 'lockeddoor', 'journals_flag.hub.topics', 'tunic.capitol_0', 'reader_flag.paper2.bingo', 'photo', 'tunic.flaghouse', 'reader.paper1.next', 'directory.closeup.archivist', 'intro', 'businesscards.card_bingo.next', 'reader.paper2.bingo', 'retirement_letter', 'remove_cup', 'journals_flag.pic_0.next', 'magnify', 'coffee', 'key', 'togrampa', 'reader_flag.paper1.next', 'janitor', 'tohallway', 'chap1_finale', 'report', 'outtolunch', 'journals_flag.hub.topics_old', 'journals_flag.pic_1.next', 'reader.paper2.next', 'chap1_finale_c', 'reader_flag.paper2.next', 'door_block_talk', 'journals_flag.pic_1.bingo', 'journals_flag.pic_2.next', 'journals_flag.pic_2.bingo', 'block_magnify', 'reader.paper0.prev', 'block', 'reader_flag.paper0.prev', 'block_0', 'door_block_clean', 'reader.paper2.prev', 'reader.paper1.prev', 'doorblock', 'tocloset', 'reader_flag.paper2.prev', 'reader_flag.paper1.prev', 'block_tomap2', 'journals_flag.pic_0_old.next', 'journals_flag.pic_1_old.next', 'block_tocollection', 'block_nelson', 'journals_flag.pic_2_old.next', 'block_tomap1', 'block_badge', 'need_glasses', 'block_badge_2', 'fox', 'block_1']\ntext_lists = ['tunic.historicalsociety.cage.confrontation', 'tunic.wildlife.center.crane_ranger.crane', 'tunic.historicalsociety.frontdesk.archivist.newspaper', 'tunic.historicalsociety.entry.groupconvo', 'tunic.wildlife.center.wells.nodeer', 'tunic.historicalsociety.frontdesk.archivist.have_glass', 'tunic.drycleaner.frontdesk.worker.hub', 'tunic.historicalsociety.closet_dirty.gramps.news', 'tunic.humanecology.frontdesk.worker.intro', 'tunic.historicalsociety.frontdesk.archivist_glasses.confrontation', 'tunic.historicalsociety.basement.seescratches', 'tunic.historicalsociety.collection.cs', 'tunic.flaghouse.entry.flag_girl.hello', 'tunic.historicalsociety.collection.gramps.found', 'tunic.historicalsociety.basement.ch3start', 'tunic.historicalsociety.entry.groupconvo_flag', 'tunic.library.frontdesk.worker.hello', 'tunic.library.frontdesk.worker.wells', 'tunic.historicalsociety.collection_flag.gramps.flag', 'tunic.historicalsociety.basement.savedteddy', 'tunic.library.frontdesk.worker.nelson', 'tunic.wildlife.center.expert.removed_cup', 'tunic.library.frontdesk.worker.flag', 'tunic.historicalsociety.frontdesk.archivist.hello', 'tunic.historicalsociety.closet.gramps.intro_0_cs_0', 'tunic.historicalsociety.entry.boss.flag', 'tunic.flaghouse.entry.flag_girl.symbol', 'tunic.historicalsociety.closet_dirty.trigger_scarf', 'tunic.drycleaner.frontdesk.worker.done', 'tunic.historicalsociety.closet_dirty.what_happened', 'tunic.wildlife.center.wells.animals', 'tunic.historicalsociety.closet.teddy.intro_0_cs_0', 'tunic.historicalsociety.cage.glasses.afterteddy', 'tunic.historicalsociety.cage.teddy.trapped', 'tunic.historicalsociety.cage.unlockdoor', 'tunic.historicalsociety.stacks.journals.pic_2.bingo', 'tunic.historicalsociety.entry.wells.flag', 'tunic.humanecology.frontdesk.worker.badger', 'tunic.historicalsociety.stacks.journals_flag.pic_0.bingo', 'tunic.historicalsociety.closet.intro', 'tunic.historicalsociety.closet.retirement_letter.hub', 'tunic.historicalsociety.entry.directory.closeup.archivist', 'tunic.historicalsociety.collection.tunic.slip', 'tunic.kohlcenter.halloffame.plaque.face.date', 'tunic.historicalsociety.closet_dirty.trigger_coffee', 'tunic.drycleaner.frontdesk.logbook.page.bingo', 'tunic.library.microfiche.reader.paper2.bingo', 'tunic.kohlcenter.halloffame.togrampa', 'tunic.capitol_2.hall.boss.haveyougotit', 'tunic.wildlife.center.wells.nodeer_recap', 'tunic.historicalsociety.cage.glasses.beforeteddy', 'tunic.historicalsociety.closet_dirty.gramps.helpclean', 'tunic.wildlife.center.expert.recap', 'tunic.historicalsociety.frontdesk.archivist.have_glass_recap', 'tunic.historicalsociety.stacks.journals_flag.pic_1.bingo', 'tunic.historicalsociety.cage.lockeddoor', 'tunic.historicalsociety.stacks.journals_flag.pic_2.bingo', 'tunic.historicalsociety.collection.gramps.lost', 'tunic.historicalsociety.closet.notebook', 'tunic.historicalsociety.frontdesk.magnify', 'tunic.humanecology.frontdesk.businesscards.card_bingo.bingo', 'tunic.wildlife.center.remove_cup', 'tunic.library.frontdesk.wellsbadge.hub', 'tunic.wildlife.center.tracks.hub.deer', 'tunic.historicalsociety.frontdesk.key', 'tunic.library.microfiche.reader_flag.paper2.bingo', 'tunic.flaghouse.entry.colorbook', 'tunic.wildlife.center.coffee', 'tunic.capitol_1.hall.boss.haveyougotit', 'tunic.historicalsociety.basement.janitor', 'tunic.historicalsociety.collection_flag.gramps.recap', 'tunic.wildlife.center.wells.animals2', 'tunic.flaghouse.entry.flag_girl.symbol_recap', 'tunic.historicalsociety.closet_dirty.photo', 'tunic.historicalsociety.stacks.outtolunch', 'tunic.library.frontdesk.worker.wells_recap', 'tunic.historicalsociety.frontdesk.archivist_glasses.confrontation_recap', 'tunic.capitol_0.hall.boss.talktogramps', 'tunic.historicalsociety.closet.photo', 'tunic.historicalsociety.collection.tunic', 'tunic.historicalsociety.closet.teddy.intro_0_cs_5', 'tunic.historicalsociety.closet_dirty.gramps.archivist', 'tunic.historicalsociety.closet_dirty.door_block_talk', 'tunic.historicalsociety.entry.boss.flag_recap', 'tunic.historicalsociety.frontdesk.archivist.need_glass_0', 'tunic.historicalsociety.entry.wells.talktogramps', 'tunic.historicalsociety.frontdesk.block_magnify', 'tunic.historicalsociety.frontdesk.archivist.foundtheodora', 'tunic.historicalsociety.closet_dirty.gramps.nothing', 'tunic.historicalsociety.closet_dirty.door_block_clean', 'tunic.capitol_1.hall.boss.writeitup', 'tunic.library.frontdesk.worker.nelson_recap', 'tunic.library.frontdesk.worker.hello_short', 'tunic.historicalsociety.stacks.block', 'tunic.historicalsociety.frontdesk.archivist.need_glass_1', 'tunic.historicalsociety.entry.boss.talktogramps', 'tunic.historicalsociety.frontdesk.archivist.newspaper_recap', 'tunic.historicalsociety.entry.wells.flag_recap', 'tunic.drycleaner.frontdesk.worker.done2', 'tunic.library.frontdesk.worker.flag_recap', 'tunic.humanecology.frontdesk.block_0', 'tunic.library.frontdesk.worker.preflag', 'tunic.historicalsociety.basement.gramps.seeyalater', 'tunic.flaghouse.entry.flag_girl.hello_recap', 'tunic.historicalsociety.closet.doorblock', 'tunic.drycleaner.frontdesk.worker.takealook', 'tunic.historicalsociety.basement.gramps.whatdo', 'tunic.library.frontdesk.worker.droppedbadge', 'tunic.historicalsociety.entry.block_tomap2', 'tunic.library.frontdesk.block_nelson', 'tunic.library.microfiche.block_0', 'tunic.historicalsociety.entry.block_tocollection', 'tunic.historicalsociety.entry.block_tomap1', 'tunic.historicalsociety.collection.gramps.look_0', 'tunic.library.frontdesk.block_badge', 'tunic.historicalsociety.cage.need_glasses', 'tunic.library.frontdesk.block_badge_2', 'tunic.kohlcenter.halloffame.block_0', 'tunic.capitol_0.hall.chap1_finale_c', 'tunic.capitol_1.hall.chap2_finale_c', 'tunic.capitol_2.hall.chap4_finale_c', 'tunic.wildlife.center.fox.concern', 'tunic.drycleaner.frontdesk.block_0', 'tunic.historicalsociety.entry.gramps.hub', 'tunic.humanecology.frontdesk.block_1', 'tunic.drycleaner.frontdesk.block_1']\nroom_lists = ['tunic.historicalsociety.entry', 'tunic.wildlife.center', 'tunic.historicalsociety.cage', 'tunic.library.frontdesk', 'tunic.historicalsociety.frontdesk', 'tunic.historicalsociety.stacks', 'tunic.historicalsociety.closet_dirty', 'tunic.humanecology.frontdesk', 'tunic.historicalsociety.basement', 'tunic.kohlcenter.halloffame', 'tunic.library.microfiche', 'tunic.drycleaner.frontdesk', 'tunic.historicalsociety.collection', 'tunic.historicalsociety.closet', 'tunic.flaghouse.entry', 'tunic.historicalsociety.collection_flag', 'tunic.capitol_1.hall', 'tunic.capitol_0.hall', 'tunic.capitol_2.hall']\nLEVELS = [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22]\nlevel_groups = [\"0-4\", \"5-12\", \"13-22\"]","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:12:18.825918Z","iopub.execute_input":"2023-05-22T03:12:18.826618Z","iopub.status.idle":"2023-05-22T03:12:18.852796Z","shell.execute_reply.started":"2023-05-22T03:12:18.826574Z","shell.execute_reply":"2023-05-22T03:12:18.851646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Some few updates of https://www.kaggle.com/code/takanashihumbert/magic-bingo-train-part-lb-0-687\n\ncolumns = [\n\n    pl.col(\"page\").cast(pl.Float32),\n    (\n        (pl.col(\"elapsed_time\") - pl.col(\"elapsed_time\").shift(1)) \n         .fill_null(0)\n         .clip(0, 1e9)\n         .over([\"session_id\", \"level_group\"])\n         .alias(\"elapsed_time_diff\")\n    ),\n    (\n        (pl.col(\"screen_coor_x\") - pl.col(\"screen_coor_x\").shift(1)) \n         .abs()\n         .over([\"session_id\", \"level_group\"])\n        .alias(\"location_x_diff\") \n    ),\n    (\n        (pl.col(\"screen_coor_y\") - pl.col(\"screen_coor_y\").shift(1)) \n         .abs()\n         .over([\"session_id\", \"level_group\"])\n        .alias(\"location_y_diff\") \n    ),\n    pl.col(\"fqid\").fill_null(\"fqid_None\"),\n    pl.col(\"text_fqid\").fill_null(\"text_fqid_None\")\n]","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:12:18.854492Z","iopub.execute_input":"2023-05-22T03:12:18.855425Z","iopub.status.idle":"2023-05-22T03:12:18.872746Z","shell.execute_reply.started":"2023-05-22T03:12:18.855382Z","shell.execute_reply":"2023-05-22T03:12:18.871692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer_pl(x, grp, use_extra, feature_suffix):\n    aggs = [\n        pl.col(\"index\").count().alias(f\"session_number_{feature_suffix}\"),\n\n        *[pl.col('index').filter(pl.col('text').str.contains(c)).count().alias(f'word_{c}') for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).mean().alias(f'word_mean_{c}') for c in\n          DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).std().alias(f'word_std_{c}') for c in\n          DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).max().alias(f'word_max_{c}') for c in\n          DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).sum().alias(f'word_sum_{c}') for c in\n          DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).median().alias(f'word_median_{c}') for c\n          in DIALOGS],\n\n        *[pl.col(c).drop_nulls().n_unique().alias(f\"{c}_unique_{feature_suffix}\") for c in CATS],\n\n        *[pl.col(c).mean().alias(f\"{c}_mean_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).std().alias(f\"{c}_std_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).min().alias(f\"{c}_min_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).max().alias(f\"{c}_max_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).median().alias(f\"{c}_median_{feature_suffix}\") for c in NUMS],\n\n        *[pl.col(\"fqid\").filter(pl.col(\"fqid\") == c).count().alias(f\"{c}_fqid_counts{feature_suffix}\")\n          for c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for\n          c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\") for\n          c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in fqid_lists],\n\n        *[pl.col(\"text_fqid\").filter(pl.col(\"text_fqid\") == c).count().alias(f\"{c}_text_fqid_counts{feature_suffix}\")\n          for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\")\n          for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in text_lists],\n\n        *[pl.col(\"room_fqid\").filter(pl.col(\"room_fqid\") == c).count().alias(f\"{c}_room_fqid_counts{feature_suffix}\")\n          for c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for\n          c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\")\n          for\n          c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in room_lists],\n\n        *[pl.col(\"event_name\").filter(pl.col(\"event_name\") == c).count().alias(f\"{c}_event_name_counts{feature_suffix}\")\n          for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\")\n          for\n          c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).median().alias(\n            f\"{c}_ET_median_{feature_suffix}\") for\n          c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in event_name_feature],\n\n        *[pl.col(\"name\").filter(pl.col(\"name\") == c).count().alias(f\"{c}_name_counts{feature_suffix}\") for c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\") for\n          c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in\n          name_feature],\n\n        *[pl.col(\"level\").filter(pl.col(\"level\") == c).count().alias(f\"{c}_LEVEL_count{feature_suffix}\") for c in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c\n          in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for c in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\") for\n          c in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in\n          LEVELS],\n\n        *[pl.col(\"level_group\").filter(pl.col(\"level_group\") == c).count().alias(\n            f\"{c}_LEVEL_group_count{feature_suffix}\") for c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\")\n          for c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).median().alias(\n            f\"{c}_ET_median_{feature_suffix}\") for c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in\n          level_groups],\n\n    ]\n\n    df = x.groupby(['session_id'], maintain_order=True).agg(aggs).sort(\"session_id\")\n\n    if use_extra:\n        if grp == '5-12':\n            aggs = [\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Here's the log book.\")\n                                              | (pl.col(\"fqid\") == 'logbook.page.bingo'))\n                    .apply(lambda s: s.max() - s.min()).alias(\"logbook_bingo_duration\"),\n                pl.col(\"index\").filter(\n                    (pl.col(\"text\") == \"Here's the log book.\") | (pl.col(\"fqid\") == 'logbook.page.bingo')).apply(\n                    lambda s: s.max() - s.min()).alias(\"logbook_bingo_indexCount\"),\n                pl.col(\"elapsed_time\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader')) | (\n                            pl.col(\"fqid\") == \"reader.paper2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\n                    \"reader_bingo_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader')) | (\n                        pl.col(\"fqid\") == \"reader.paper2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\n                    \"reader_bingo_indexCount\"),\n                pl.col(\"elapsed_time\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals')) | (\n                            pl.col(\"fqid\") == \"journals.pic_2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\n                    \"journals_bingo_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals')) | (\n                        pl.col(\"fqid\") == \"journals.pic_2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\n                    \"journals_bingo_indexCount\"),\n            ]\n            tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n            df = df.join(tmp, on=\"session_id\", how='left')\n\n        if grp == '13-22':\n            aggs = [\n                pl.col(\"elapsed_time\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader_flag')) | (\n                            pl.col(\"fqid\") == \"tunic.library.microfiche.reader_flag.paper2.bingo\")).apply(\n                    lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"reader_flag_duration\"),\n                pl.col(\"index\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader_flag')) | (\n                            pl.col(\"fqid\") == \"tunic.library.microfiche.reader_flag.paper2.bingo\")).apply(\n                    lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"reader_flag_indexCount\"),\n                pl.col(\"elapsed_time\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals_flag')) | (\n                            pl.col(\"fqid\") == \"journals_flag.pic_0.bingo\")).apply(\n                    lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"journalsFlag_bingo_duration\"),\n                pl.col(\"index\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals_flag')) | (\n                            pl.col(\"fqid\") == \"journals_flag.pic_0.bingo\")).apply(\n                    lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"journalsFlag_bingo_indexCount\")\n            ]\n            tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n            df = df.join(tmp, on=\"session_id\", how='left')\n\n    return df.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:12:18.874540Z","iopub.execute_input":"2023-05-22T03:12:18.875420Z","iopub.status.idle":"2023-05-22T03:12:18.940253Z","shell.execute_reply.started":"2023-05-22T03:12:18.875359Z","shell.execute_reply":"2023-05-22T03:12:18.938842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# we prepare the dataset for the training by level :\ndf = (pl.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\",dtypes=train_dtypes)\n      .drop([\"fullscreen\", \"hq\", \"music\"])\n      .with_columns(columns))\ndf = reduce_mem_usage_pl(df)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:12:18.942319Z","iopub.execute_input":"2023-05-22T03:12:18.942983Z","iopub.status.idle":"2023-05-22T03:13:44.384468Z","shell.execute_reply.started":"2023-05-22T03:12:18.942941Z","shell.execute_reply":"2023-05-22T03:13:44.381203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.with_columns(df[\"text\"].cast(pl.Utf8))","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:13:44.387004Z","iopub.execute_input":"2023-05-22T03:13:44.388571Z","iopub.status.idle":"2023-05-22T03:13:46.167963Z","shell.execute_reply.started":"2023-05-22T03:13:44.388503Z","shell.execute_reply":"2023-05-22T03:13:46.166691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:13:46.170283Z","iopub.execute_input":"2023-05-22T03:13:46.171290Z","iopub.status.idle":"2023-05-22T03:13:46.208823Z","shell.execute_reply.started":"2023-05-22T03:13:46.171247Z","shell.execute_reply":"2023-05-22T03:13:46.207758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = df.filter(pl.col(\"level_group\")=='0-4')\ndf2 = df.filter(pl.col(\"level_group\")=='5-12')\ndf3 = df.filter(pl.col(\"level_group\")=='13-22')\ndf1.shape,df2.shape,df3.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:13:46.210983Z","iopub.execute_input":"2023-05-22T03:13:46.211982Z","iopub.status.idle":"2023-05-22T03:13:49.491030Z","shell.execute_reply.started":"2023-05-22T03:13:46.211914Z","shell.execute_reply":"2023-05-22T03:13:49.489515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:13:49.492323Z","iopub.execute_input":"2023-05-22T03:13:49.492762Z","iopub.status.idle":"2023-05-22T03:13:49.678601Z","shell.execute_reply.started":"2023-05-22T03:13:49.492723Z","shell.execute_reply":"2023-05-22T03:13:49.677488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf1 = feature_engineer_pl(df1, grp='0-4', use_extra=True, feature_suffix='')\nprint('df1 done',df1.shape)\ndf2 = feature_engineer_pl(df2, grp='5-12', use_extra=True, feature_suffix='')\nprint('df2 done',df2.shape)\ndf3 = feature_engineer_pl(df3, grp='13-22', use_extra=True, feature_suffix='')\nprint('df3 done',df3.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:13:49.680162Z","iopub.execute_input":"2023-05-22T03:13:49.681451Z","iopub.status.idle":"2023-05-22T03:16:05.021047Z","shell.execute_reply.started":"2023-05-22T03:13:49.681403Z","shell.execute_reply":"2023-05-22T03:16:05.012216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# some cleaning...\nnull1 = df1.isnull().sum().sort_values(ascending=False) / len(df1)\nnull2 = df2.isnull().sum().sort_values(ascending=False) / len(df1)\nnull3 = df3.isnull().sum().sort_values(ascending=False) / len(df1)\n\ndrop1 = list(null1[null1>0.9].index)\ndrop2 = list(null2[null2>0.9].index)\ndrop3 = list(null3[null3>0.9].index)\nprint(len(drop1), len(drop2), len(drop3))\n\nfor col in df1.columns:\n    if df1[col].nunique()==1:\n        print(col)\n        drop1.append(col)\nprint(\"*********df1 DONE*********\")\nfor col in df2.columns:\n    if df2[col].nunique()==1:\n        print(col)\n        drop2.append(col)\nprint(\"*********df2 DONE*********\")\nfor col in df3.columns:\n    if df3[col].nunique()==1:\n        print(col)\n        drop3.append(col)\nprint(\"*********df3 DONE*********\")","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:16:05.026115Z","iopub.execute_input":"2023-05-22T03:16:05.028091Z","iopub.status.idle":"2023-05-22T03:16:08.294991Z","shell.execute_reply.started":"2023-05-22T03:16:05.028014Z","shell.execute_reply":"2023-05-22T03:16:08.293140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def time_feature(train):\n    \n    train[\"year\"] = train[\"session_id\"].apply(lambda x: int(str(x)[:2])).astype(np.uint8)\n    train[\"month\"] = train[\"session_id\"].apply(lambda x: int(str(x)[2:4])+1).astype(np.uint8)\n    train[\"day\"] = train[\"session_id\"].apply(lambda x: int(str(x)[4:6])).astype(np.uint8)\n    train[\"hour\"] = train[\"session_id\"].apply(lambda x: int(str(x)[6:8])).astype(np.uint8)\n    train[\"minute\"] = train[\"session_id\"].apply(lambda x: int(str(x)[8:10])).astype(np.uint8)\n    train[\"second\"] = train[\"session_id\"].apply(lambda x: int(str(x)[10:12])).astype(np.uint8)\n\n    return train","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:16:08.297536Z","iopub.execute_input":"2023-05-22T03:16:08.297965Z","iopub.status.idle":"2023-05-22T03:16:08.308808Z","shell.execute_reply.started":"2023-05-22T03:16:08.297903Z","shell.execute_reply":"2023-05-22T03:16:08.307262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = time_feature(df1)\ndf2 = time_feature(df2)\ndf3 = time_feature(df3)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:16:08.310889Z","iopub.execute_input":"2023-05-22T03:16:08.311314Z","iopub.status.idle":"2023-05-22T03:16:08.697428Z","shell.execute_reply.started":"2023-05-22T03:16:08.311276Z","shell.execute_reply":"2023-05-22T03:16:08.695566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = df1.set_index('session_id')\ndf2 = df2.set_index('session_id')\ndf3 = df3.set_index('session_id')\n\nFEATURES1 = [c for c in df1.columns if c not in drop1+['level_group']]\nFEATURES2 = [c for c in df2.columns if c not in drop2+['level_group']]\nFEATURES3 = [c for c in df3.columns if c not in drop3+['level_group']]\nprint('We will train with', len(FEATURES1), len(FEATURES2), len(FEATURES3) ,'features')\nALL_USERS = df1.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:16:08.699300Z","iopub.execute_input":"2023-05-22T03:16:08.700455Z","iopub.status.idle":"2023-05-22T03:16:09.852555Z","shell.execute_reply.started":"2023-05-22T03:16:08.700405Z","shell.execute_reply":"2023-05-22T03:16:09.851318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1 = open('FEATURES1.txt', 'wb')\npickle.dump(FEATURES1, f1)\nf2 = open('FEATURES2.txt', 'wb')\npickle.dump(FEATURES2, f2)\nf3 = open('FEATURES3.txt', 'wb')\npickle.dump(FEATURES3, f3)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:16:09.853883Z","iopub.execute_input":"2023-05-22T03:16:09.854817Z","iopub.status.idle":"2023-05-22T03:16:09.864058Z","shell.execute_reply.started":"2023-05-22T03:16:09.854771Z","shell.execute_reply":"2023-05-22T03:16:09.862852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# With previous training notebook (Kfold with 20 folds as performed in others notebooks) :\nestimators_lgb = [498, 448, 378, 364, 405, 495, 456, 249, 384, 405, 356, 262, 484, 381, 392, 248 ,248, 345]","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:16:09.865985Z","iopub.execute_input":"2023-05-22T03:16:09.867212Z","iopub.status.idle":"2023-05-22T03:16:09.878146Z","shell.execute_reply.started":"2023-05-22T03:16:09.867147Z","shell.execute_reply":"2023-05-22T03:16:09.876878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_params = {\n    'boosting_type': 'gbdt',\n    'objective': 'binary',\n    'metric': 'binary_logloss',\n    'learning_rate': 0.05,\n    'alpha': 8,\n    'max_depth': 4,\n    'subsample': 0.8,\n    'colsample_bytree': 0.5,\n    'random_state': 42\n}","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:16:09.880643Z","iopub.execute_input":"2023-05-22T03:16:09.881821Z","iopub.status.idle":"2023-05-22T03:16:09.892670Z","shell.execute_reply.started":"2023-05-22T03:16:09.881768Z","shell.execute_reply":"2023-05-22T03:16:09.891469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## We fit and store the models for predictions","metadata":{}},{"cell_type":"code","source":"warnings.filterwarnings(\"ignore\")\ntargets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]))\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]))\npred_lgb = np.zeros((df1.shape[0], 18))\nn_splits = 5\nkf = KFold(n_splits=n_splits)\n\nfor q in range(1, 19):\n    # USE THIS TRAIN DATA WITH THESE QUESTIONS\n    if q <= 3:\n        grp = '0-4'\n        df = df1\n        FEATURES = FEATURES1\n    elif q <= 13:\n        grp = '5-12'\n        df = df2\n        FEATURES = FEATURES2\n    elif q <= 22:\n        grp = '13-22'\n        df = df3\n        FEATURES = FEATURES3\n\n    lgb_params['n_estimators'] = estimators_lgb[q - 1]\n\n    # TRAIN DATA\n    for fold, (train_idx, val_idx) in enumerate(kf.split(df)):\n        df_train = df.iloc[train_idx] #.reset_index(drop=True)\n        train_users = df_train.index.values\n        train_y = targets[targets['session'].isin(list(train_users))].loc[targets.q == q].set_index('session')\n\n        df_val = df.iloc[val_idx] #.reset_index(drop=True)\n        val_users = df_val.index.values\n        val_y = targets[targets['session'].isin(list(val_users))].loc[targets.q == q].set_index('session')\n\n        clf = LGBMClassifier(**lgb_params)\n        clf.fit(df_train[FEATURES].astype('float32'), train_y['correct'], verbose=0)\n\n        clf.booster_.save_model(f'LGBM_question{q}_fold{fold}.lgb')\n        print(f'Model saved for question {q} fold {fold} with iterations = {estimators_lgb[q-1]}')","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:16:09.895137Z","iopub.execute_input":"2023-05-22T03:16:09.896399Z","iopub.status.idle":"2023-05-22T03:49:51.664709Z","shell.execute_reply.started":"2023-05-22T03:16:09.896323Z","shell.execute_reply":"2023-05-22T03:49:51.663293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"import jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:49:51.666508Z","iopub.execute_input":"2023-05-22T03:49:51.667237Z","iopub.status.idle":"2023-05-22T03:49:51.708742Z","shell.execute_reply.started":"2023-05-22T03:49:51.667190Z","shell.execute_reply":"2023-05-22T03:49:51.706986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models_list = [[Booster(model_file = f\"/kaggle/working/LGBM_question{q}_fold{fold}.lgb\"\n) for fold in range(5)] for q in range(1, 19)]","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:49:51.710704Z","iopub.execute_input":"2023-05-22T03:49:51.711159Z","iopub.status.idle":"2023-05-22T03:49:52.757623Z","shell.execute_reply.started":"2023-05-22T03:49:51.711116Z","shell.execute_reply":"2023-05-22T03:49:52.755913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\ncount = 0\n\nfor (test, sample_submission) in iter_test:\n    session_id = test.session_id.values[0]\n    grp = test.level_group.values[0]\n    a, b = limits[grp]\n    preds = []\n\n    # ------------------- level 0-4 ---------------------------------\n    if a == 1:\n        FEATURES = FEATURES1\n        test = (pl.from_pandas(test)\n                .drop([\"fullscreen\", \"hq\", \"music\"])\n                .with_columns(columns))\n        test = feature_engineer_pl(test, grp, use_extra=True, feature_suffix='')\n        test = time_feature(test)\n        test = test[FEATURES]\n\n    # ------------------- level 5-12 ---------------------------------\n    elif a == 4:\n        FEATURES = FEATURES2\n        test = (pl.from_pandas(test)\n                .drop([\"fullscreen\", \"hq\", \"music\"])\n                .with_columns(columns))\n        test = feature_engineer_pl(test, grp, use_extra=True, feature_suffix='')\n        test = time_feature(test)\n        test = test[FEATURES]\n\n    # ------------------- level 13-22 ---------------------------------\n    elif a == 14:\n        FEATURES = FEATURES3\n        test = (pl.from_pandas(test)\n                .drop([\"fullscreen\", \"hq\", \"music\"])\n                .with_columns(columns))\n        test = feature_engineer_pl(test, grp, use_extra=True, feature_suffix='')\n        test = time_feature(test)\n        test = test[FEATURES]\n    \n    for q in range(a,b):\n        fold = 0\n        thresh = 0.63\n        model_0 = models_list[q-1][fold]\n        model_1 = models_list[q-1][fold+1]\n        model_2 = models_list[q-1][fold+2]\n        model_3 = models_list[q-1][fold+3]\n        model_4 = models_list[q-1][fold+4]\n        \n        pred_0 = model_0.predict(test[FEATURES].astype(np.float32))\n        pred_1 = model_1.predict(test[FEATURES].astype(np.float32))\n        pred_2 = model_2.predict(test[FEATURES].astype(np.float32))\n        pred_3 = model_3.predict(test[FEATURES].astype(np.float32))\n        pred_4 = model_4.predict(test[FEATURES].astype(np.float32))\n        \n        pred = (pred_0 + pred_1 + pred_2 + pred_3 + pred_4) / 5\n        preds.append(int(pred > thresh))\n\n    sample_submission[\"correct\"] = preds\n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:49:52.759406Z","iopub.execute_input":"2023-05-22T03:49:52.759803Z","iopub.status.idle":"2023-05-22T03:49:56.693805Z","shell.execute_reply.started":"2023-05-22T03:49:52.759765Z","shell.execute_reply":"2023-05-22T03:49:56.692627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('submission.csv').head(10)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:49:56.695353Z","iopub.execute_input":"2023-05-22T03:49:56.696025Z","iopub.status.idle":"2023-05-22T03:49:56.738313Z","shell.execute_reply.started":"2023-05-22T03:49:56.695983Z","shell.execute_reply":"2023-05-22T03:49:56.736670Z"},"trusted":true},"execution_count":null,"outputs":[]}]}