{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-21T06:36:49.159035Z","iopub.execute_input":"2023-04-21T06:36:49.160135Z","iopub.status.idle":"2023-04-21T06:36:49.190108Z","shell.execute_reply.started":"2023-04-21T06:36:49.160083Z","shell.execute_reply":"2023-04-21T06:36:49.188375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nimport json\nimport pickle\nimport warnings\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport seaborn as sns\n\nfrom sklearn.model_selection import KFold\n\nfrom xgboost import XGBClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.linear_model import LogisticRegression\n\nfrom sklearn.metrics import f1_score","metadata":{"execution":{"iopub.status.busy":"2023-04-21T06:36:49.196000Z","iopub.execute_input":"2023-04-21T06:36:49.196812Z","iopub.status.idle":"2023-04-21T06:36:49.920202Z","shell.execute_reply.started":"2023-04-21T06:36:49.196772Z","shell.execute_reply":"2023-04-21T06:36:49.919062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## feature engineer (XGB)","metadata":{}},{"cell_type":"code","source":"CATS_XGB = ['event_name', 'name', 'fqid', 'room_fqid', 'text_fqid']\nNUMS_XGB = ['page', 'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y',\n        'hover_duration', 'elapsed_time_diff']\n\nname_feature_xgb = ['basic', 'undefined', 'close', 'open', 'prev', 'next']\nevent_name_feature_xgb = ['cutscene_click', 'person_click', 'navigate_click',\n       'observation_click', 'notification_click', 'object_click',\n       'object_hover', 'map_hover', 'map_click', 'checkpoint',\n       'notebook_click']\n\n# from https://www.kaggle.com/code/leehomhuang/catboost-baseline-with-lots-features-inference :\nfqid_lists_xgb = ['worker', 'archivist', 'gramps', 'wells', 'toentry', 'confrontation', 'crane_ranger', 'groupconvo', 'flag_girl', 'tomap', 'tostacks', 'tobasement', 'archivist_glasses', 'boss', 'journals', 'seescratches', 'groupconvo_flag', 'cs', 'teddy', 'expert', 'businesscards', 'ch3start', 'tunic.historicalsociety', 'tofrontdesk', 'savedteddy', 'plaque', 'glasses', 'tunic.drycleaner', 'reader_flag', 'tunic.library', 'tracks', 'tunic.capitol_2', 'trigger_scarf', 'reader', 'directory', 'tunic.capitol_1', 'journals.pic_0.next', 'unlockdoor', 'tunic', 'what_happened', 'tunic.kohlcenter', 'tunic.humanecology', 'colorbook', 'logbook', 'businesscards.card_0.next', 'journals.hub.topics', 'logbook.page.bingo', 'journals.pic_1.next', 'journals_flag', 'reader.paper0.next', 'tracks.hub.deer', 'reader_flag.paper0.next', 'trigger_coffee', 'wellsbadge', 'journals.pic_2.next', 'tomicrofiche', 'journals_flag.pic_0.bingo', 'plaque.face.date', 'notebook', 'tocloset_dirty', 'businesscards.card_bingo.bingo', 'businesscards.card_1.next', 'tunic.wildlife', 'tunic.hub.slip', 'tocage', 'journals.pic_2.bingo', 'tocollectionflag', 'tocollection', 'chap4_finale_c', 'chap2_finale_c', 'lockeddoor', 'journals_flag.hub.topics', 'tunic.capitol_0', 'reader_flag.paper2.bingo', 'photo', 'tunic.flaghouse', 'reader.paper1.next', 'directory.closeup.archivist', 'intro', 'businesscards.card_bingo.next', 'reader.paper2.bingo', 'retirement_letter', 'remove_cup', 'journals_flag.pic_0.next', 'magnify', 'coffee', 'key', 'togrampa', 'reader_flag.paper1.next', 'janitor', 'tohallway', 'chap1_finale', 'report', 'outtolunch', 'journals_flag.hub.topics_old', 'journals_flag.pic_1.next', 'reader.paper2.next', 'chap1_finale_c', 'reader_flag.paper2.next', 'door_block_talk', 'journals_flag.pic_1.bingo', 'journals_flag.pic_2.next', 'journals_flag.pic_2.bingo', 'block_magnify', 'reader.paper0.prev', 'block', 'reader_flag.paper0.prev', 'block_0', 'door_block_clean', 'reader.paper2.prev', 'reader.paper1.prev', 'doorblock', 'tocloset', 'reader_flag.paper2.prev', 'reader_flag.paper1.prev', 'block_tomap2', 'journals_flag.pic_0_old.next', 'journals_flag.pic_1_old.next', 'block_tocollection', 'block_nelson', 'journals_flag.pic_2_old.next', 'block_tomap1', 'block_badge', 'need_glasses', 'block_badge_2', 'fox', 'block_1']\n\ntext_lists_xgb = ['tunic.historicalsociety.cage.confrontation', 'tunic.wildlife.center.crane_ranger.crane', 'tunic.historicalsociety.frontdesk.archivist.newspaper', 'tunic.historicalsociety.entry.groupconvo', 'tunic.wildlife.center.wells.nodeer', 'tunic.historicalsociety.frontdesk.archivist.have_glass', 'tunic.drycleaner.frontdesk.worker.hub', 'tunic.historicalsociety.closet_dirty.gramps.news', 'tunic.humanecology.frontdesk.worker.intro', 'tunic.historicalsociety.frontdesk.archivist_glasses.confrontation', 'tunic.historicalsociety.basement.seescratches', 'tunic.historicalsociety.collection.cs', 'tunic.flaghouse.entry.flag_girl.hello', 'tunic.historicalsociety.collection.gramps.found', 'tunic.historicalsociety.basement.ch3start', 'tunic.historicalsociety.entry.groupconvo_flag', 'tunic.library.frontdesk.worker.hello', 'tunic.library.frontdesk.worker.wells', 'tunic.historicalsociety.collection_flag.gramps.flag', 'tunic.historicalsociety.basement.savedteddy', 'tunic.library.frontdesk.worker.nelson', 'tunic.wildlife.center.expert.removed_cup', 'tunic.library.frontdesk.worker.flag', 'tunic.historicalsociety.frontdesk.archivist.hello', 'tunic.historicalsociety.closet.gramps.intro_0_cs_0', 'tunic.historicalsociety.entry.boss.flag', 'tunic.flaghouse.entry.flag_girl.symbol', 'tunic.historicalsociety.closet_dirty.trigger_scarf', 'tunic.drycleaner.frontdesk.worker.done', 'tunic.historicalsociety.closet_dirty.what_happened', 'tunic.wildlife.center.wells.animals', 'tunic.historicalsociety.closet.teddy.intro_0_cs_0', 'tunic.historicalsociety.cage.glasses.afterteddy', 'tunic.historicalsociety.cage.teddy.trapped', 'tunic.historicalsociety.cage.unlockdoor', 'tunic.historicalsociety.stacks.journals.pic_2.bingo', 'tunic.historicalsociety.entry.wells.flag', 'tunic.humanecology.frontdesk.worker.badger', 'tunic.historicalsociety.stacks.journals_flag.pic_0.bingo', 'tunic.historicalsociety.closet.intro', 'tunic.historicalsociety.closet.retirement_letter.hub', 'tunic.historicalsociety.entry.directory.closeup.archivist', 'tunic.historicalsociety.collection.tunic.slip', 'tunic.kohlcenter.halloffame.plaque.face.date', 'tunic.historicalsociety.closet_dirty.trigger_coffee', 'tunic.drycleaner.frontdesk.logbook.page.bingo', 'tunic.library.microfiche.reader.paper2.bingo', 'tunic.kohlcenter.halloffame.togrampa', 'tunic.capitol_2.hall.boss.haveyougotit', 'tunic.wildlife.center.wells.nodeer_recap', 'tunic.historicalsociety.cage.glasses.beforeteddy', 'tunic.historicalsociety.closet_dirty.gramps.helpclean', 'tunic.wildlife.center.expert.recap', 'tunic.historicalsociety.frontdesk.archivist.have_glass_recap', 'tunic.historicalsociety.stacks.journals_flag.pic_1.bingo', 'tunic.historicalsociety.cage.lockeddoor', 'tunic.historicalsociety.stacks.journals_flag.pic_2.bingo', 'tunic.historicalsociety.collection.gramps.lost', 'tunic.historicalsociety.closet.notebook', 'tunic.historicalsociety.frontdesk.magnify', 'tunic.humanecology.frontdesk.businesscards.card_bingo.bingo', 'tunic.wildlife.center.remove_cup', 'tunic.library.frontdesk.wellsbadge.hub', 'tunic.wildlife.center.tracks.hub.deer', 'tunic.historicalsociety.frontdesk.key', 'tunic.library.microfiche.reader_flag.paper2.bingo', 'tunic.flaghouse.entry.colorbook', 'tunic.wildlife.center.coffee', 'tunic.capitol_1.hall.boss.haveyougotit', 'tunic.historicalsociety.basement.janitor', 'tunic.historicalsociety.collection_flag.gramps.recap', 'tunic.wildlife.center.wells.animals2', 'tunic.flaghouse.entry.flag_girl.symbol_recap', 'tunic.historicalsociety.closet_dirty.photo', 'tunic.historicalsociety.stacks.outtolunch', 'tunic.library.frontdesk.worker.wells_recap', 'tunic.historicalsociety.frontdesk.archivist_glasses.confrontation_recap', 'tunic.capitol_0.hall.boss.talktogramps', 'tunic.historicalsociety.closet.photo', 'tunic.historicalsociety.collection.tunic', 'tunic.historicalsociety.closet.teddy.intro_0_cs_5', 'tunic.historicalsociety.closet_dirty.gramps.archivist', 'tunic.historicalsociety.closet_dirty.door_block_talk', 'tunic.historicalsociety.entry.boss.flag_recap', 'tunic.historicalsociety.frontdesk.archivist.need_glass_0', 'tunic.historicalsociety.entry.wells.talktogramps', 'tunic.historicalsociety.frontdesk.block_magnify', 'tunic.historicalsociety.frontdesk.archivist.foundtheodora', 'tunic.historicalsociety.closet_dirty.gramps.nothing', 'tunic.historicalsociety.closet_dirty.door_block_clean', 'tunic.capitol_1.hall.boss.writeitup', 'tunic.library.frontdesk.worker.nelson_recap', 'tunic.library.frontdesk.worker.hello_short', 'tunic.historicalsociety.stacks.block', 'tunic.historicalsociety.frontdesk.archivist.need_glass_1', 'tunic.historicalsociety.entry.boss.talktogramps', 'tunic.historicalsociety.frontdesk.archivist.newspaper_recap', 'tunic.historicalsociety.entry.wells.flag_recap', 'tunic.drycleaner.frontdesk.worker.done2', 'tunic.library.frontdesk.worker.flag_recap', 'tunic.humanecology.frontdesk.block_0', 'tunic.library.frontdesk.worker.preflag', 'tunic.historicalsociety.basement.gramps.seeyalater', 'tunic.flaghouse.entry.flag_girl.hello_recap', 'tunic.historicalsociety.closet.doorblock', 'tunic.drycleaner.frontdesk.worker.takealook', 'tunic.historicalsociety.basement.gramps.whatdo', 'tunic.library.frontdesk.worker.droppedbadge', 'tunic.historicalsociety.entry.block_tomap2', 'tunic.library.frontdesk.block_nelson', 'tunic.library.microfiche.block_0', 'tunic.historicalsociety.entry.block_tocollection', 'tunic.historicalsociety.entry.block_tomap1', 'tunic.historicalsociety.collection.gramps.look_0', 'tunic.library.frontdesk.block_badge', 'tunic.historicalsociety.cage.need_glasses', 'tunic.library.frontdesk.block_badge_2', 'tunic.kohlcenter.halloffame.block_0', 'tunic.capitol_0.hall.chap1_finale_c', 'tunic.capitol_1.hall.chap2_finale_c', 'tunic.capitol_2.hall.chap4_finale_c', 'tunic.wildlife.center.fox.concern', 'tunic.drycleaner.frontdesk.block_0', 'tunic.historicalsociety.entry.gramps.hub', 'tunic.humanecology.frontdesk.block_1', 'tunic.drycleaner.frontdesk.block_1']\nroom_lists_xgb = ['tunic.historicalsociety.entry', 'tunic.wildlife.center', 'tunic.historicalsociety.cage', 'tunic.library.frontdesk', 'tunic.historicalsociety.frontdesk', 'tunic.historicalsociety.stacks', 'tunic.historicalsociety.closet_dirty', 'tunic.humanecology.frontdesk', 'tunic.historicalsociety.basement', 'tunic.kohlcenter.halloffame', 'tunic.library.microfiche', 'tunic.drycleaner.frontdesk', 'tunic.historicalsociety.collection', 'tunic.historicalsociety.closet', 'tunic.flaghouse.entry', 'tunic.historicalsociety.collection_flag', 'tunic.capitol_1.hall', 'tunic.capitol_0.hall', 'tunic.capitol_2.hall']","metadata":{"execution":{"iopub.status.busy":"2023-04-21T06:36:49.922327Z","iopub.execute_input":"2023-04-21T06:36:49.923045Z","iopub.status.idle":"2023-04-21T06:36:49.943343Z","shell.execute_reply.started":"2023-04-21T06:36:49.923003Z","shell.execute_reply":"2023-04-21T06:36:49.942122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = [\n\n    pl.col(\"page\").cast(pl.Float32),\n    (\n        (pl.col(\"elapsed_time\") - pl.col(\"elapsed_time\").shift(1)) \n         .fill_null(0)\n         .clip(0, 1e9)\n         .over([\"session_id\", \"level_group\"])\n         .alias(\"elapsed_time_diff\")\n    ),\n    (\n        (pl.col(\"screen_coor_x\") - pl.col(\"screen_coor_x\").shift(1)) \n         .abs()\n         .over([\"session_id\", \"level_group\"])\n        .alias(\"location_x_diff\") \n    ),\n    (\n        (pl.col(\"screen_coor_y\") - pl.col(\"screen_coor_y\").shift(1)) \n         .abs()\n         .over([\"session_id\", \"level_group\"])\n        .alias(\"location_y_diff\") \n    ),\n    pl.col(\"fqid\").fill_null(\"fqid_None\"),\n    pl.col(\"text_fqid\").fill_null(\"text_fqid_None\")\n]","metadata":{"execution":{"iopub.status.busy":"2023-04-21T06:36:49.946482Z","iopub.execute_input":"2023-04-21T06:36:49.947505Z","iopub.status.idle":"2023-04-21T06:36:49.963066Z","shell.execute_reply.started":"2023-04-21T06:36:49.947454Z","shell.execute_reply":"2023-04-21T06:36:49.962004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer_pl(x, grp, use_extra, feature_suffix):\n        \n    aggs = [\n        pl.col(\"index\").count().alias(f\"session_number_{feature_suffix}\"),\n        \n        *[pl.col(c).drop_nulls().n_unique().alias(f\"{c}_unique_{feature_suffix}\") for c in CATS_XGB],\n        [pl.col(c).quantile(0.1, \"nearest\").alias(f\"{c}_quantile1_{feature_suffix}\") for c in NUMS_XGB],\n        *[pl.col(c).quantile(0.2, \"nearest\").alias(f\"{c}_quantile2_{feature_suffix}\") for c in NUMS_XGB],\n        *[pl.col(c).quantile(0.4, \"nearest\").alias(f\"{c}_quantile4_{feature_suffix}\") for c in NUMS_XGB],\n        *[pl.col(c).quantile(0.6, \"nearest\").alias(f\"{c}_quantile6_{feature_suffix}\") for c in NUMS_XGB],\n        *[pl.col(c).quantile(0.8, \"nearest\").alias(f\"{c}_quantile8_{feature_suffix}\") for c in NUMS_XGB],\n        *[pl.col(c).quantile(0.9, \"nearest\").alias(f\"{c}_quantile9_{feature_suffix}\") for c in NUMS_XGB],\n        \n        *[pl.col(c).mean().alias(f\"{c}_mean_{feature_suffix}\") for c in NUMS_XGB],\n        *[pl.col(c).std().alias(f\"{c}_std_{feature_suffix}\") for c in NUMS_XGB],\n        *[pl.col(c).min().alias(f\"{c}_min_{feature_suffix}\") for c in NUMS_XGB],\n        *[pl.col(c).max().alias(f\"{c}_max_{feature_suffix}\") for c in NUMS_XGB],\n        *[pl.col(\"event_name\").filter(pl.col(\"event_name\") == c).count().alias(f\"{c}_event_name_counts{feature_suffix}\")for c in event_name_feature_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).quantile(0.1, \"nearest\").alias(f\"{c}_ET_quantile1_{feature_suffix}\") for c in event_name_feature_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).quantile(0.2, \"nearest\").alias(f\"{c}_ET_quantile2_{feature_suffix}\") for c in event_name_feature_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).quantile(0.4, \"nearest\").alias(f\"{c}_ET_quantile4_{feature_suffix}\") for c in event_name_feature_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).quantile(0.6, \"nearest\").alias(f\"{c}_ET_quantile6_{feature_suffix}\") for c in event_name_feature_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).quantile(0.8, \"nearest\").alias(f\"{c}_ET_quantile8_{feature_suffix}\") for c in event_name_feature_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).quantile(0.9, \"nearest\").alias(f\"{c}_ET_quantile9_{feature_suffix}\") for c in event_name_feature_xgb],      \n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c in event_name_feature_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in event_name_feature_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in event_name_feature_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).min().alias(f\"{c}_ET_min_{feature_suffix}\") for c in event_name_feature_xgb],\n                *[pl.col(\"name\").filter(pl.col(\"name\") == c).count().alias(f\"{c}_name_counts{feature_suffix}\")for c in name_feature_xgb],   \n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\")==c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c in name_feature_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\")==c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in name_feature_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\")==c).min().alias(f\"{c}_ET_min_{feature_suffix}\") for c in name_feature_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\")==c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in name_feature_xgb],  \n        \n        *[pl.col(\"room_fqid\").filter(pl.col(\"room_fqid\") == c).count().alias(f\"{c}_room_fqid_counts{feature_suffix}\")for c in room_lists_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in room_lists_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c in room_lists_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in room_lists_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).min().alias(f\"{c}_ET_min_{feature_suffix}\") for c in room_lists_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for c in room_lists_xgb],\n\n        *[pl.col(\"fqid\").filter(pl.col(\"fqid\") == c).count().alias(f\"{c}_fqid_counts{feature_suffix}\")for c in fqid_lists_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in fqid_lists_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c in fqid_lists_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in fqid_lists_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).min().alias(f\"{c}_ET_min_{feature_suffix}\") for c in fqid_lists_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for c in fqid_lists_xgb],\n       \n        *[pl.col(\"text_fqid\").filter(pl.col(\"text_fqid\") == c).count().alias(f\"{c}_text_fqid_counts{feature_suffix}\") for c in text_lists_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in text_lists_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c in text_lists_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in text_lists_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).min().alias(f\"{c}_ET_min_{feature_suffix}\") for c in text_lists_xgb],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for c in text_lists_xgb],\n         \n        *[pl.col(\"location_x_diff\").filter(pl.col(\"event_name\")==c).mean().alias(f\"{c}_ET_mean_x{feature_suffix}\") for c in event_name_feature_xgb],\n        *[pl.col(\"location_x_diff\").filter(pl.col(\"event_name\")==c).std().alias(f\"{c}_ET_std_x{feature_suffix}\") for c in event_name_feature_xgb],\n        *[pl.col(\"location_x_diff\").filter(pl.col(\"event_name\")==c).max().alias(f\"{c}_ET_max_x{feature_suffix}\") for c in event_name_feature_xgb],\n        *[pl.col(\"location_x_diff\").filter(pl.col(\"event_name\")==c).min().alias(f\"{c}_ET_min_x{feature_suffix}\") for c in event_name_feature_xgb],\n    ]\n\n    df = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n  \n    if use_extra:\n        if grp=='5-12':\n            aggs = [\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\")==\"Here's the log book.\")|(pl.col(\"fqid\")=='logbook.page.bingo')).apply(lambda s: s.max()-s.min()).alias(\"logbook_bingo_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\")==\"Here's the log book.\")|(pl.col(\"fqid\")=='logbook.page.bingo')).apply(lambda s: s.max()-s.min()).alias(\"logbook_bingo_indexCount\"),\n                pl.col(\"elapsed_time\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='reader'))|(pl.col(\"fqid\")==\"reader.paper2.bingo\")).apply(lambda s: s.max()-s.min()).alias(\"reader_bingo_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='reader'))|(pl.col(\"fqid\")==\"reader.paper2.bingo\")).apply(lambda s: s.max()-s.min()).alias(\"reader_bingo_indexCount\"),\n                pl.col(\"elapsed_time\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='journals'))|(pl.col(\"fqid\")==\"journals.pic_2.bingo\")).apply(lambda s: s.max()-s.min()).alias(\"journals_bingo_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='journals'))|(pl.col(\"fqid\")==\"journals.pic_2.bingo\")).apply(lambda s: s.max()-s.min()).alias(\"journals_bingo_indexCount\"),\n            ]\n            tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n            df = df.join(tmp, on=\"session_id\", how='left')\n        if grp=='13-22':\n            aggs = [\n                pl.col(\"elapsed_time\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='reader_flag'))|(pl.col(\"fqid\")==\"tunic.library.microfiche.reader_flag.paper2.bingo\")).apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(\"reader_flag_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='reader_flag'))|(pl.col(\"fqid\")==\"tunic.library.microfiche.reader_flag.paper2.bingo\")).apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(\"reader_flag_indexCount\"),\n                pl.col(\"elapsed_time\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='journals_flag'))|(pl.col(\"fqid\")==\"journals_flag.pic_0.bingo\")).apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(\"journalsFlag_bingo_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='journals_flag'))|(pl.col(\"fqid\")==\"journals_flag.pic_0.bingo\")).apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(\"journalsFlag_bingo_indexCount\"),\n            ]\n            tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n            df = df.join(tmp, on=\"session_id\", how='left')\n            \n    return df.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2023-04-21T06:36:49.964875Z","iopub.execute_input":"2023-04-21T06:36:49.965606Z","iopub.status.idle":"2023-04-21T06:36:50.024446Z","shell.execute_reply.started":"2023-04-21T06:36:49.965567Z","shell.execute_reply":"2023-04-21T06:36:50.022429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## feature engineer (RF, NN)","metadata":{}},{"cell_type":"code","source":"CATS_RF = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMS_RF = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']","metadata":{"execution":{"iopub.status.busy":"2023-04-21T06:36:50.026370Z","iopub.execute_input":"2023-04-21T06:36:50.026837Z","iopub.status.idle":"2023-04-21T06:36:50.044157Z","shell.execute_reply.started":"2023-04-21T06:36:50.026774Z","shell.execute_reply":"2023-04-21T06:36:50.042801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer_rf(train):\n    dfs = []\n    for c in CATS_RF:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS_RF:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        dfs.append(tmp)\n    for c in NUMS_RF:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n#     df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-21T06:36:50.046241Z","iopub.execute_input":"2023-04-21T06:36:50.046650Z","iopub.status.idle":"2023-04-21T06:36:50.056864Z","shell.execute_reply.started":"2023-04-21T06:36:50.046613Z","shell.execute_reply":"2023-04-21T06:36:50.055537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## params","metadata":{}},{"cell_type":"code","source":"# With previous training notebook (Kfold with 20 folds as performed in others notebooks) :\nestimators_xgb = [498, 448, 378, 364, 405, 495, 456, 249, 384, 405, 356, 262, 484, 381, 392, 248 ,248, 345]","metadata":{"execution":{"iopub.status.busy":"2023-04-21T06:36:50.059682Z","iopub.execute_input":"2023-04-21T06:36:50.060161Z","iopub.status.idle":"2023-04-21T06:36:50.075347Z","shell.execute_reply.started":"2023-04-21T06:36:50.060120Z","shell.execute_reply":"2023-04-21T06:36:50.073883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_params = {\n        'booster': 'gbtree',\n        'tree_method': 'hist',\n        'objective': 'binary:logistic',\n        'eval_metric':'logloss',\n        'learning_rate': 0.02,\n        'alpha': 8,\n        'max_depth': 4,\n        'subsample':0.8,\n        'colsample_bytree': 0.5,\n        'seed': 42\n        }","metadata":{"execution":{"iopub.status.busy":"2023-04-21T06:36:50.077686Z","iopub.execute_input":"2023-04-21T06:36:50.078092Z","iopub.status.idle":"2023-04-21T06:36:50.088480Z","shell.execute_reply.started":"2023-04-21T06:36:50.078057Z","shell.execute_reply":"2023-04-21T06:36:50.086937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf_params = {\n}","metadata":{"execution":{"iopub.status.busy":"2023-04-21T06:36:50.090453Z","iopub.execute_input":"2023-04-21T06:36:50.091238Z","iopub.status.idle":"2023-04-21T06:36:50.101670Z","shell.execute_reply.started":"2023-04-21T06:36:50.091176Z","shell.execute_reply":"2023-04-21T06:36:50.100367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn_params = {\n}","metadata":{"execution":{"iopub.status.busy":"2023-04-21T06:36:50.103200Z","iopub.execute_input":"2023-04-21T06:36:50.103658Z","iopub.status.idle":"2023-04-21T06:36:50.115642Z","shell.execute_reply.started":"2023-04-21T06:36:50.103622Z","shell.execute_reply":"2023-04-21T06:36:50.113415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## submission","metadata":{}},{"cell_type":"code","source":"# IMPORT KAGGLE API\nimport jo_wilder\njo_wilder.make_env.__called__ = False\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n\n# CLEAR MEMORY\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-21T06:36:50.117599Z","iopub.execute_input":"2023-04-21T06:36:50.119276Z","iopub.status.idle":"2023-04-21T06:36:50.272549Z","shell.execute_reply.started":"2023-04-21T06:36:50.119201Z","shell.execute_reply":"2023-04-21T06:36:50.271465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## features\nwith open(\"/kaggle/input/features/features_1.json\", mode=\"r\") as fp:\n    FEATURES1 = json.load(fp)[\"features\"]\nwith open(\"/kaggle/input/features/features_2.json\", mode=\"r\") as fp:\n    FEATURES2 = json.load(fp)[\"features\"]\nwith open(\"/kaggle/input/features/features_3.json\", mode=\"r\") as fp:\n    FEATURES3 = json.load(fp)[\"features\"]\n    \nwith open(\"/kaggle/input/features/features_rf.json\", mode=\"r\") as fp:\n    FEATURES_RF = json.load(fp)[\"features\"]","metadata":{"execution":{"iopub.status.busy":"2023-04-21T06:36:50.276895Z","iopub.execute_input":"2023-04-21T06:36:50.277561Z","iopub.status.idle":"2023-04-21T06:36:50.291004Z","shell.execute_reply.started":"2023-04-21T06:36:50.277520Z","shell.execute_reply":"2023-04-21T06:36:50.289384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\nN_FOLDS = 5\nbest_threshold = 0.635\n\nfor idx, (test, sample_submission) in enumerate(iter_test):\n    user_id = int(sample_submission.session_id.values[0].split(\"_\")[0])\n    print(user_id)\n    session_id = test.session_id.values[0]\n    grp = test.level_group.values[0]\n    a, b = limits[grp]\n\n    # ------------------- level 0-4 ---------------------------------\n    if a == 1:\n        FEATURES_XGB = FEATURES1\n        test_xgb = (pl.from_pandas(test)\n              .drop([\"fullscreen\", \"hq\", \"music\"])\n              .with_columns(columns))\n\n\n    # ------------------- level 5-12 ---------------------------------\n    elif a == 4:\n        FEATURES_XGB = FEATURES2\n        test_xgb = (pl.from_pandas(test)\n              .drop([\"fullscreen\", \"hq\", \"music\"])\n              .with_columns(columns))\n\n    # ------------------- level 13-22 ---------------------------------    \n    elif a == 14:\n        FEATURES_XGB = FEATURES3\n        test_xgb = (pl.from_pandas(test)\n              .drop([\"fullscreen\", \"hq\", \"music\"])\n              .with_columns(columns))\n\n    test_xgb = feature_engineer_pl(test_xgb, grp, use_extra=True, feature_suffix='')\n    test_xgb = test_xgb.loc[test_xgb.session_id==user_id]\n    test_xgb = test_xgb[FEATURES_XGB].astype('float32')\n    test_rf = feature_engineer_rf(test)\n    test_rf = test_rf.loc[test_rf.session_id==user_id]\n    test_rf = test_rf[FEATURES_RF].astype('float32')\n\n    for q in range(a, b):\n        xgb_preds = []\n        rf_preds = []\n        nn_preds = []\n        lr_preds = []\n        for f in range(N_FOLDS):\n            ### load models\n            xgb = XGBClassifier()\n            xgb.load_model(f'/kaggle/input/exp009/XGB_question{q}_{f}.xgb')\n            with open(f\"/kaggle/input/exp015/RF_question{q}_{f}.pickle\", mode=\"rb\") as fp:\n                rf = pickle.load(fp)\n            with open(f\"/kaggle/input/exp015/NN_question{q}_{f}.pickle\", mode=\"rb\") as fp:\n                nn = pickle.load(fp)\n\n            p1 = xgb.predict_proba(test_xgb)[:, 1]\n            xgb_preds.append(p1)\n            p2 = rf.predict_proba(test_rf)[:, 1]\n            rf_preds.append(p2)\n            p3 = nn.predict_proba(test_rf)[:, 1]\n            nn_preds.append(p3)\n        p1 = np.mean(xgb_preds, axis=0)\n        p2 = np.mean(rf_preds, axis=0)\n        p3 = np.mean(nn_preds, axis=0)\n        pp = np.concatenate([p1.reshape(-1, 1), p2.reshape(-1, 1), p3.reshape(-1, 1)], axis=1)\n        for f in range(N_FOLDS):\n            with open(f\"/kaggle/input/exp015/LR_question{q}_{f}.pickle\", mode=\"rb\") as fp:\n                    lr = pickle.load(fp)\n            p = lr.predict_proba(pp)[:, 1]\n            lr_preds.append(p)\n        p = np.mean(lr_preds)\n\n        mask = sample_submission.session_id.str.contains(f'q{q}')\n        print(sample_submission.loc[mask,'correct'])\n        sample_submission.loc[mask,'correct'] = (p > best_threshold).astype(\"int\")\n\n    env.predict(sample_submission)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-21T06:36:50.293214Z","iopub.execute_input":"2023-04-21T06:36:50.294091Z","iopub.status.idle":"2023-04-21T06:36:58.594324Z","shell.execute_reply.started":"2023-04-21T06:36:50.294047Z","shell.execute_reply":"2023-04-21T06:36:58.592259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv(\"submission.csv\").head(60)","metadata":{"execution":{"iopub.status.busy":"2023-04-21T06:36:58.595833Z","iopub.execute_input":"2023-04-21T06:36:58.596199Z","iopub.status.idle":"2023-04-21T06:36:58.626175Z","shell.execute_reply.started":"2023-04-21T06:36:58.596165Z","shell.execute_reply":"2023-04-21T06:36:58.624897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}