{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n**A mixture of 2 models:**\n - https://www.kaggle.com/code/vadimkamaev/catboost\n - https://www.kaggle.com/code/leehomhuang/catboost-baseline-with-lots-features-inference","metadata":{"papermill":{"duration":0.011465,"end_time":"2023-06-18T11:17:37.121257","exception":false,"start_time":"2023-06-18T11:17:37.109792","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import os\nimport gc\nimport sys\nimport datetime\nimport pickle\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nfrom catboost import CatBoostClassifier, Pool\n\n# Feature Lists\nCATS = ['event_name', 'name', 'fqid', 'room_fqid', 'text_fqid']\nNUMS = ['page', 'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y',\n        'hover_duration', 'elapsed_time_diff']\nDIALOGS = ['that', 'this', 'it', 'you','find','found','Found','notebook','Wells','wells','help','need', 'Oh','Ooh','Jo', 'flag', 'can','and','is','the','to']\n\nname_feature = ['basic', 'undefined', 'close', 'open', 'prev', 'next']\nevent_name_feature = ['cutscene_click', 'person_click', 'navigate_click',\n       'observation_click', 'notification_click', 'object_click',\n       'object_hover', 'map_hover', 'map_click', 'checkpoint',\n       'notebook_click']\n\n# What you can interact\nsub_fqid_lists = {'0-4': ['gramps',\n 'wells',\n 'toentry',\n 'groupconvo',\n 'tomap',\n 'tostacks',\n 'tobasement',\n 'boss',\n 'cs',\n 'teddy',\n 'tunic.historicalsociety',\n 'plaque',\n 'directory',\n 'tunic',\n 'tunic.kohlcenter',\n 'plaque.face.date',\n 'notebook',\n 'tunic.hub.slip',\n 'tocollection',\n 'tunic.capitol_0',\n 'photo',\n 'intro',\n 'retirement_letter',\n 'togrampa',\n 'janitor',\n 'chap1_finale',\n 'report',\n 'outtolunch',\n 'chap1_finale_c',\n 'block_0',\n 'doorblock',\n 'tocloset',\n 'block_tomap2',\n 'block_tocollection',\n 'block_tomap1'],\n                  '5-12': ['worker',\n 'archivist',\n 'gramps',\n 'toentry',\n 'tomap',\n 'tostacks',\n 'tobasement',\n 'boss',\n 'journals',\n 'businesscards',\n 'tunic.historicalsociety',\n 'tofrontdesk',\n 'plaque',\n 'tunic.drycleaner',\n 'tunic.library',\n 'trigger_scarf',\n 'reader',\n 'directory',\n 'tunic.capitol_1',\n 'journals.pic_0.next',\n 'tunic',\n 'what_happened',\n 'tunic.kohlcenter',\n 'tunic.humanecology',\n 'logbook',\n 'businesscards.card_0.next',\n 'journals.hub.topics',\n 'logbook.page.bingo',\n 'journals.pic_1.next',\n 'reader.paper0.next',\n 'trigger_coffee',\n 'wellsbadge',\n 'journals.pic_2.next',\n 'tomicrofiche',\n 'tocloset_dirty',\n 'businesscards.card_bingo.bingo',\n 'businesscards.card_1.next',\n 'tunic.hub.slip',\n 'journals.pic_2.bingo',\n 'tocollection',\n 'chap2_finale_c',\n 'tunic.capitol_0',\n 'photo',\n 'reader.paper1.next',\n 'businesscards.card_bingo.next',\n 'reader.paper2.bingo',\n 'magnify',\n 'janitor',\n 'tohallway',\n 'outtolunch',\n 'reader.paper2.next',\n 'door_block_talk',\n 'block_magnify',\n 'reader.paper0.prev',\n 'block',\n 'block_0',\n 'door_block_clean',\n 'reader.paper2.prev',\n 'reader.paper1.prev',\n 'block_badge',\n 'block_badge_2',\n 'block_1'],\n                  '13-22': ['worker',\n 'gramps',\n 'wells',\n 'toentry',\n 'confrontation',\n 'crane_ranger',\n 'flag_girl',\n 'tomap',\n 'tostacks',\n 'tobasement',\n 'archivist_glasses',\n 'boss',\n 'journals',\n 'seescratches',\n 'groupconvo_flag',\n 'teddy',\n 'expert',\n 'businesscards',\n 'ch3start',\n 'tunic.historicalsociety',\n 'tofrontdesk',\n 'savedteddy',\n 'plaque',\n 'glasses',\n 'tunic.drycleaner',\n 'reader_flag',\n 'tunic.library',\n 'tracks',\n 'tunic.capitol_2',\n 'reader',\n 'directory',\n 'tunic.capitol_1',\n 'journals.pic_0.next',\n 'unlockdoor',\n 'tunic',\n 'tunic.kohlcenter',\n 'tunic.humanecology',\n 'colorbook',\n 'logbook',\n 'businesscards.card_0.next',\n 'journals.hub.topics',\n 'journals.pic_1.next',\n 'journals_flag',\n 'reader.paper0.next',\n 'tracks.hub.deer',\n 'reader_flag.paper0.next',\n 'journals.pic_2.next',\n 'tomicrofiche',\n 'journals_flag.pic_0.bingo',\n 'tocloset_dirty',\n 'businesscards.card_1.next',\n 'tunic.wildlife',\n 'tunic.hub.slip',\n 'tocage',\n 'journals.pic_2.bingo',\n 'tocollectionflag',\n 'tocollection',\n 'chap4_finale_c',\n 'lockeddoor',\n 'journals_flag.hub.topics',\n 'reader_flag.paper2.bingo',\n 'photo',\n 'tunic.flaghouse',\n 'reader.paper1.next',\n 'directory.closeup.archivist',\n 'businesscards.card_bingo.next',\n 'remove_cup',\n 'journals_flag.pic_0.next',\n 'coffee',\n 'key',\n 'reader_flag.paper1.next',\n 'tohallway',\n 'outtolunch',\n 'journals_flag.hub.topics_old',\n 'journals_flag.pic_1.next',\n 'reader.paper2.next',\n 'reader_flag.paper2.next',\n 'journals_flag.pic_1.bingo',\n 'journals_flag.pic_2.next',\n 'journals_flag.pic_2.bingo',\n 'reader.paper0.prev',\n 'reader_flag.paper0.prev',\n 'reader.paper2.prev',\n 'reader.paper1.prev',\n 'reader_flag.paper2.prev',\n 'reader_flag.paper1.prev',\n 'journals_flag.pic_0_old.next',\n 'journals_flag.pic_1_old.next',\n 'block_nelson',\n 'journals_flag.pic_2_old.next',\n 'need_glasses',\n 'fox'],\n                 }\n\n# What is in the room\nsub_room_lists = {'0-4': ['tunic.historicalsociety.entry',\n 'tunic.historicalsociety.stacks',\n 'tunic.historicalsociety.basement',\n 'tunic.kohlcenter.halloffame',\n 'tunic.historicalsociety.collection',\n 'tunic.historicalsociety.closet',\n 'tunic.capitol_0.hall'],\n                  '5-12': ['tunic.historicalsociety.entry',\n 'tunic.library.frontdesk',\n 'tunic.historicalsociety.frontdesk',\n 'tunic.historicalsociety.stacks',\n 'tunic.historicalsociety.closet_dirty',\n 'tunic.humanecology.frontdesk',\n 'tunic.historicalsociety.basement',\n 'tunic.kohlcenter.halloffame',\n 'tunic.library.microfiche',\n 'tunic.drycleaner.frontdesk',\n 'tunic.historicalsociety.collection',\n 'tunic.capitol_1.hall',\n 'tunic.capitol_0.hall'],\n                  '13-22': ['tunic.historicalsociety.entry',\n 'tunic.wildlife.center',\n 'tunic.historicalsociety.cage',\n 'tunic.library.frontdesk',\n 'tunic.historicalsociety.frontdesk',\n 'tunic.historicalsociety.stacks',\n 'tunic.historicalsociety.closet_dirty',\n 'tunic.humanecology.frontdesk',\n 'tunic.historicalsociety.basement',\n 'tunic.kohlcenter.halloffame',\n 'tunic.library.microfiche',\n 'tunic.drycleaner.frontdesk',\n 'tunic.historicalsociety.collection',\n 'tunic.flaghouse.entry',\n 'tunic.historicalsociety.collection_flag',\n 'tunic.capitol_1.hall',\n 'tunic.capitol_2.hall'],\n                 }\n\n\nsub_text_lists = {'0-4': ['tunic.historicalsociety.entry.groupconvo',\n 'tunic.historicalsociety.collection.cs',\n 'tunic.historicalsociety.collection.gramps.found',\n 'tunic.historicalsociety.closet.gramps.intro_0_cs_0',\n 'tunic.historicalsociety.closet.teddy.intro_0_cs_0',\n 'tunic.historicalsociety.closet.intro',\n 'tunic.historicalsociety.closet.retirement_letter.hub',\n 'tunic.historicalsociety.collection.tunic.slip',\n 'tunic.kohlcenter.halloffame.plaque.face.date',\n 'tunic.kohlcenter.halloffame.togrampa',\n 'tunic.historicalsociety.collection.gramps.lost',\n 'tunic.historicalsociety.closet.notebook',\n 'tunic.historicalsociety.basement.janitor',\n 'tunic.historicalsociety.stacks.outtolunch',\n 'tunic.historicalsociety.closet.photo',\n 'tunic.historicalsociety.collection.tunic',\n 'tunic.historicalsociety.closet.teddy.intro_0_cs_5',\n 'tunic.historicalsociety.entry.wells.talktogramps',\n 'tunic.historicalsociety.entry.boss.talktogramps',\n 'tunic.historicalsociety.closet.doorblock',\n 'tunic.historicalsociety.entry.block_tomap2',\n 'tunic.historicalsociety.entry.block_tocollection',\n 'tunic.historicalsociety.entry.block_tomap1',\n 'tunic.historicalsociety.collection.gramps.look_0',\n 'tunic.kohlcenter.halloffame.block_0',\n 'tunic.capitol_0.hall.chap1_finale_c',\n 'tunic.historicalsociety.entry.gramps.hub'],\n               '5-12': ['tunic.historicalsociety.frontdesk.archivist.newspaper',\n 'tunic.historicalsociety.frontdesk.archivist.have_glass',\n 'tunic.drycleaner.frontdesk.worker.hub',\n 'tunic.historicalsociety.closet_dirty.gramps.news',\n 'tunic.humanecology.frontdesk.worker.intro',\n 'tunic.library.frontdesk.worker.hello',\n 'tunic.library.frontdesk.worker.wells',\n 'tunic.historicalsociety.frontdesk.archivist.hello',\n 'tunic.historicalsociety.closet_dirty.trigger_scarf',\n 'tunic.drycleaner.frontdesk.worker.done',\n 'tunic.historicalsociety.closet_dirty.what_happened',\n 'tunic.historicalsociety.stacks.journals.pic_2.bingo',\n 'tunic.humanecology.frontdesk.worker.badger',\n 'tunic.historicalsociety.closet_dirty.trigger_coffee',\n 'tunic.drycleaner.frontdesk.logbook.page.bingo',\n 'tunic.library.microfiche.reader.paper2.bingo',\n 'tunic.historicalsociety.closet_dirty.gramps.helpclean',\n 'tunic.historicalsociety.frontdesk.archivist.have_glass_recap',\n 'tunic.historicalsociety.frontdesk.magnify',\n 'tunic.humanecology.frontdesk.businesscards.card_bingo.bingo',\n 'tunic.library.frontdesk.wellsbadge.hub',\n 'tunic.capitol_1.hall.boss.haveyougotit',\n 'tunic.historicalsociety.basement.janitor',\n 'tunic.historicalsociety.closet_dirty.photo',\n 'tunic.historicalsociety.stacks.outtolunch',\n 'tunic.library.frontdesk.worker.wells_recap',\n 'tunic.capitol_0.hall.boss.talktogramps',\n 'tunic.historicalsociety.closet_dirty.gramps.archivist',\n 'tunic.historicalsociety.closet_dirty.door_block_talk',\n 'tunic.historicalsociety.frontdesk.archivist.need_glass_0',\n 'tunic.historicalsociety.frontdesk.block_magnify',\n 'tunic.historicalsociety.frontdesk.archivist.foundtheodora',\n 'tunic.historicalsociety.closet_dirty.gramps.nothing',\n 'tunic.historicalsociety.closet_dirty.door_block_clean',\n 'tunic.library.frontdesk.worker.hello_short',\n 'tunic.historicalsociety.stacks.block',\n 'tunic.historicalsociety.frontdesk.archivist.need_glass_1',\n 'tunic.historicalsociety.frontdesk.archivist.newspaper_recap',\n 'tunic.drycleaner.frontdesk.worker.done2',\n 'tunic.humanecology.frontdesk.block_0',\n 'tunic.library.frontdesk.worker.preflag',\n 'tunic.drycleaner.frontdesk.worker.takealook',\n 'tunic.library.frontdesk.worker.droppedbadge',\n 'tunic.library.microfiche.block_0',\n 'tunic.library.frontdesk.block_badge',\n 'tunic.library.frontdesk.block_badge_2',\n 'tunic.capitol_1.hall.chap2_finale_c',\n 'tunic.drycleaner.frontdesk.block_0',\n 'tunic.humanecology.frontdesk.block_1',\n 'tunic.drycleaner.frontdesk.block_1'],\n               '13-22': ['tunic.historicalsociety.cage.confrontation',\n 'tunic.wildlife.center.crane_ranger.crane',\n 'tunic.wildlife.center.wells.nodeer',\n 'tunic.historicalsociety.frontdesk.archivist_glasses.confrontation',\n 'tunic.historicalsociety.basement.seescratches',\n 'tunic.flaghouse.entry.flag_girl.hello',\n 'tunic.historicalsociety.basement.ch3start',\n 'tunic.historicalsociety.entry.groupconvo_flag',\n 'tunic.historicalsociety.collection_flag.gramps.flag',\n 'tunic.historicalsociety.basement.savedteddy',\n 'tunic.library.frontdesk.worker.nelson',\n 'tunic.wildlife.center.expert.removed_cup',\n 'tunic.library.frontdesk.worker.flag',\n 'tunic.historicalsociety.entry.boss.flag',\n 'tunic.flaghouse.entry.flag_girl.symbol',\n 'tunic.wildlife.center.wells.animals',\n 'tunic.historicalsociety.cage.glasses.afterteddy',\n 'tunic.historicalsociety.cage.teddy.trapped',\n 'tunic.historicalsociety.cage.unlockdoor',\n 'tunic.historicalsociety.stacks.journals.pic_2.bingo',\n 'tunic.historicalsociety.entry.wells.flag',\n 'tunic.humanecology.frontdesk.worker.badger',\n 'tunic.historicalsociety.stacks.journals_flag.pic_0.bingo',\n 'tunic.historicalsociety.entry.directory.closeup.archivist',\n 'tunic.capitol_2.hall.boss.haveyougotit',\n 'tunic.wildlife.center.wells.nodeer_recap',\n 'tunic.historicalsociety.cage.glasses.beforeteddy',\n 'tunic.wildlife.center.expert.recap',\n 'tunic.historicalsociety.stacks.journals_flag.pic_1.bingo',\n 'tunic.historicalsociety.cage.lockeddoor',\n 'tunic.historicalsociety.stacks.journals_flag.pic_2.bingo',\n 'tunic.wildlife.center.remove_cup',\n 'tunic.wildlife.center.tracks.hub.deer',\n 'tunic.historicalsociety.frontdesk.key',\n 'tunic.library.microfiche.reader_flag.paper2.bingo',\n 'tunic.flaghouse.entry.colorbook',\n 'tunic.wildlife.center.coffee',\n 'tunic.historicalsociety.collection_flag.gramps.recap',\n 'tunic.wildlife.center.wells.animals2',\n 'tunic.flaghouse.entry.flag_girl.symbol_recap',\n 'tunic.historicalsociety.closet_dirty.photo',\n 'tunic.historicalsociety.stacks.outtolunch',\n 'tunic.historicalsociety.frontdesk.archivist_glasses.confrontation_recap',\n 'tunic.historicalsociety.entry.boss.flag_recap',\n 'tunic.capitol_1.hall.boss.writeitup',\n 'tunic.library.frontdesk.worker.nelson_recap',\n 'tunic.historicalsociety.entry.wells.flag_recap',\n 'tunic.drycleaner.frontdesk.worker.done2',\n 'tunic.library.frontdesk.worker.flag_recap',\n 'tunic.library.frontdesk.worker.preflag',\n 'tunic.historicalsociety.basement.gramps.seeyalater',\n 'tunic.flaghouse.entry.flag_girl.hello_recap',\n 'tunic.historicalsociety.basement.gramps.whatdo',\n 'tunic.library.frontdesk.block_nelson',\n 'tunic.historicalsociety.cage.need_glasses',\n 'tunic.capitol_2.hall.chap4_finale_c',\n 'tunic.wildlife.center.fox.concern']\n              }\n\n\nSUB_LEVELS = {'0-4': [1, 2, 3, 4],\n              '5-12': [5, 6, 7, 8, 9, 10, 11, 12],\n              '13-22': [13, 14, 15, 16, 17, 18, 19, 20, 21, 22]}\nlevel_groups = [\"0-4\", \"5-12\", \"13-22\"]","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":1.624507,"end_time":"2023-06-18T11:17:38.753283","exception":false,"start_time":"2023-06-18T11:17:37.128776","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T14:03:12.247972Z","iopub.execute_input":"2023-06-18T14:03:12.248400Z","iopub.status.idle":"2023-06-18T14:03:12.297027Z","shell.execute_reply.started":"2023-06-18T14:03:12.248366Z","shell.execute_reply":"2023-06-18T14:03:12.295383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the difference of elapsed time, move on (x,y) and fill null in fqid/text_fqid\ncolumns = [\n    pl.col(\"page\").cast(pl.Float32),\n    (\n        (pl.col(\"elapsed_time\") - pl.col(\"elapsed_time\").shift(1))\n        .fill_null(0)\n        .clip(0, 1e9)\n        .over([\"session_id\", \"level\"])\n        .alias(\"elapsed_time_diff\")\n    ),\n    (\n        (pl.col(\"screen_coor_x\") - pl.col(\"screen_coor_x\").shift(1))\n        .abs()\n        .over([\"session_id\", \"level\"])\n    ),\n    (\n        (pl.col(\"screen_coor_y\") - pl.col(\"screen_coor_y\").shift(1))\n        .abs()\n        .over([\"session_id\", \"level\"])\n    ),\n    pl.col(\"fqid\").fill_null(\"fqid_None\"),\n    pl.col(\"text_fqid\").fill_null(\"text_fqid_None\")\n\n]\n\n\n# Load pretrained model of CatBoost\nmodels_list = [[CatBoostClassifier().load_model(\n    f\"/kaggle/input/catboost-predict/fold{fold}_q{q}.cbm\"\n) for fold in range(5)] for q in range(1, 19)]","metadata":{"papermill":{"duration":1.820884,"end_time":"2023-06-18T11:17:40.582282","exception":false,"start_time":"2023-06-18T11:17:38.761398","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T14:03:12.406261Z","iopub.execute_input":"2023-06-18T14:03:12.406689Z","iopub.status.idle":"2023-06-18T14:03:12.676797Z","shell.execute_reply.started":"2023-06-18T14:03:12.406648Z","shell.execute_reply":"2023-06-18T14:03:12.675414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(x, grp, use_extra, feature_suffix):\n    LEVELS = SUB_LEVELS[grp]\n    text_lists = sub_text_lists[grp]\n    room_lists = sub_room_lists[grp]\n    fqid_lists = sub_fqid_lists[grp]\n    # Statistical feature: mean, std_var, max, min, sum, median\n    aggs = [\n        pl.col(\"index\").count().alias(f\"session_number_{feature_suffix}\"),\n\n        *[pl.col('index').filter(pl.col('text').str.contains(c)).count().alias(f'word_{c}') for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).mean().alias(f'word_mean_{c}') for c in\n          DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).std().alias(f'word_std_{c}') for c in\n          DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).max().alias(f'word_max_{c}') for c in\n          DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).sum().alias(f'word_sum_{c}') for c in\n          DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).median().alias(f'word_median_{c}') for c\n          in DIALOGS],\n\n        *[pl.col(c).drop_nulls().n_unique().alias(f\"{c}_unique_{feature_suffix}\") for c in CATS],\n\n        *[pl.col(c).mean().alias(f\"{c}_mean_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).std().alias(f\"{c}_std_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).min().alias(f\"{c}_min_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).max().alias(f\"{c}_max_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).median().alias(f\"{c}_median_{feature_suffix}\") for c in NUMS],\n\n        *[pl.col(\"fqid\").filter(pl.col(\"fqid\") == c).count().alias(f\"{c}_fqid_counts{feature_suffix}\")\n          for c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for\n          c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\") for\n          c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in fqid_lists],\n\n        *[pl.col(\"text_fqid\").filter(pl.col(\"text_fqid\") == c).count().alias(f\"{c}_text_fqid_counts{feature_suffix}\")\n          for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\")\n          for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in text_lists],\n\n        *[pl.col(\"room_fqid\").filter(pl.col(\"room_fqid\") == c).count().alias(f\"{c}_room_fqid_counts{feature_suffix}\")\n          for c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for\n          c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\")\n          for\n          c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in room_lists],\n\n        *[pl.col(\"event_name\").filter(pl.col(\"event_name\") == c).count().alias(f\"{c}_event_name_counts{feature_suffix}\")\n          for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\")\n          for\n          c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).median().alias(\n            f\"{c}_ET_median_{feature_suffix}\") for\n          c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in event_name_feature],\n\n        *[pl.col(\"name\").filter(pl.col(\"name\") == c).count().alias(f\"{c}_name_counts{feature_suffix}\") for c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\") for\n          c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in\n          name_feature],\n\n        *[pl.col(\"level\").filter(pl.col(\"level\") == c).count().alias(f\"{c}_LEVEL_count{feature_suffix}\") for c in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c\n          in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for c in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\") for\n          c in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in\n          LEVELS],\n\n        *[pl.col(\"level_group\").filter(pl.col(\"level_group\") == c).count().alias(\n            f\"{c}_LEVEL_group_count{feature_suffix}\") for c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\")\n          for c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).median().alias(\n            f\"{c}_ET_median_{feature_suffix}\") for c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in\n          level_groups],\n\n    ]\n\n    # Group by session_id and aggregate the statistics(min, mac, etc.) of elapsed_time_diff between different operations/interactions\n    df = x.groupby(['session_id'], maintain_order=True).agg(aggs).sort(\"session_id\")\n\n    # Use the 'bingo' features that students navigate to click and get bingo prompt\n    # Get the count and duration(elapsed time) of activate bingo prompt\n    if use_extra:\n        if grp == '5-12':\n            aggs = [\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Here's the log book.\")\n                                              | (pl.col(\"fqid\") == 'logbook.page.bingo'))\n                    .apply(lambda s: s.max() - s.min()).alias(\"logbook_bingo_duration\"),\n                pl.col(\"index\").filter(\n                    (pl.col(\"text\") == \"Here's the log book.\") | (pl.col(\"fqid\") == 'logbook.page.bingo')).apply(\n                    lambda s: s.max() - s.min()).alias(\"logbook_bingo_indexCount\"),\n                pl.col(\"elapsed_time\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader')) | (\n                            pl.col(\"fqid\") == \"reader.paper2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\n                    \"reader_bingo_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader')) | (\n                        pl.col(\"fqid\") == \"reader.paper2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\n                    \"reader_bingo_indexCount\"),\n                pl.col(\"elapsed_time\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals')) | (\n                            pl.col(\"fqid\") == \"journals.pic_2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\n                    \"journals_bingo_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals')) | (\n                        pl.col(\"fqid\") == \"journals.pic_2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\n                    \"journals_bingo_indexCount\"),\n            ]\n            tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n            df = df.join(tmp, on=\"session_id\", how='left')\n\n        if grp == '13-22':\n            aggs = [\n                pl.col(\"elapsed_time\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader_flag')) | (\n                            pl.col(\"fqid\") == \"tunic.library.microfiche.reader_flag.paper2.bingo\")).apply(\n                    lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"reader_flag_duration\"),\n                pl.col(\"index\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader_flag')) | (\n                            pl.col(\"fqid\") == \"tunic.library.microfiche.reader_flag.paper2.bingo\")).apply(\n                    lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"reader_flag_indexCount\"),\n                pl.col(\"elapsed_time\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals_flag')) | (\n                            pl.col(\"fqid\") == \"journals_flag.pic_0.bingo\")).apply(\n                    lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"journalsFlag_bingo_duration\"),\n                pl.col(\"index\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals_flag')) | (\n                            pl.col(\"fqid\") == \"journals_flag.pic_0.bingo\")).apply(\n                    lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"journalsFlag_bingo_indexCount\")\n            ]\n            tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n            df = df.join(tmp, on=\"session_id\", how='left')\n\n    return df.to_pandas()","metadata":{"papermill":{"duration":0.096478,"end_time":"2023-06-18T11:17:40.686254","exception":false,"start_time":"2023-06-18T11:17:40.589776","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T14:03:12.679502Z","iopub.execute_input":"2023-06-18T14:03:12.680570Z","iopub.status.idle":"2023-06-18T14:03:12.768678Z","shell.execute_reply.started":"2023-06-18T14:03:12.680524Z","shell.execute_reply":"2023-06-18T14:03:12.767128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def time_feature(train):\n    train[\"year\"] = train[\"session_id\"].apply(lambda x: int(str(x)[:2])).astype(np.uint8)\n    train[\"month\"] = train[\"session_id\"].apply(lambda x: int(str(x)[2:4])+1).astype(np.uint8)\n    train[\"day\"] = train[\"session_id\"].apply(lambda x: int(str(x)[4:6])).astype(np.uint8)\n    train[\"hour\"] = train[\"session_id\"].apply(lambda x: int(str(x)[6:8])).astype(np.uint8)\n    train[\"minute\"] = train[\"session_id\"].apply(lambda x: int(str(x)[8:10])).astype(np.uint8)\n    train[\"second\"] = train[\"session_id\"].apply(lambda x: int(str(x)[10:12])).astype(np.uint8)\n    return train","metadata":{"papermill":{"duration":0.024275,"end_time":"2023-06-18T11:17:40.717956","exception":false,"start_time":"2023-06-18T11:17:40.693681","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T14:03:12.770949Z","iopub.execute_input":"2023-06-18T14:03:12.771348Z","iopub.status.idle":"2023-06-18T14:03:12.783192Z","shell.execute_reply.started":"2023-06-18T14:03:12.771312Z","shell.execute_reply":"2023-06-18T14:03:12.781741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def delt_time_def(df):\n    df.sort_values(by=['session_id', 'elapsed_time'], inplace=True)\n    df['d_time'] = df['elapsed_time'].diff(1)\n    df['d_time'].fillna(0, inplace=True)\n    df['delt_time'] = df['d_time'].clip(0, 103000)  \n    return df","metadata":{"papermill":{"duration":0.018419,"end_time":"2023-06-18T11:17:40.744037","exception":false,"start_time":"2023-06-18T11:17:40.725618","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T14:03:12.785017Z","iopub.execute_input":"2023-06-18T14:03:12.785392Z","iopub.status.idle":"2023-06-18T14:03:12.803475Z","shell.execute_reply.started":"2023-06-18T14:03:12.785359Z","shell.execute_reply":"2023-06-18T14:03:12.802027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer_2(train):\n    CATS = ['event_name', 'fqid', 'room_fqid', 'text_fqid', 'page']\n    NUMS = ['delt_time', 'hover_duration']\n    EV_NAME = ['checkpoint','observation_click', 'cutscene_click', 'notification_click', 'person_click',\n               'object_click', 'map_click', 'object_hover']    \n    new_train = pd.DataFrame(index=train['session_id'].unique(), columns=[])\n    for c in EV_NAME:\n        new_train['l_ev_name_' + c] = train[train['event_name'] == c].groupby(['session_id'])['index'].count()\n        new_train['t_ev_name_' + c] = train[train['event_name'] == c].groupby(['session_id'])['delt_time'].sum()\n    maska = train['name'] == 'basic'\n    \n    \n    qvant = train.groupby(['session_id'])['d_time'].quantile(q=0.3)\n    qvant.name = 'qvant1_0_3'\n    new_train = new_train.join(qvant)\n\n    qvant = train.groupby(['session_id'])['d_time'].quantile(q=0.8)\n    qvant.name = 'qvant2_0_8'\n    new_train = new_train.join(qvant)\n\n    qvant = train.groupby(['session_id'])['d_time'].quantile(q=0.5)\n    qvant.name = 'qvant3_0_5'\n    new_train = new_train.join(qvant)\n\n    qvant = train.groupby(['session_id'])['d_time'].quantile(q=0.65)\n    qvant.name = 'qvant4_0_65'\n    new_train = new_train.join(qvant)\n    \n    new_train['finish'] = train[maska].groupby(['session_id'])['elapsed_time'].last(1)  \n    new_train['len'] = train[maska].groupby(['session_id'])['index'].count()\n    for c in CATS:\n        tmp = train[maska].groupby(['session_id'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        new_train = new_train.join(tmp)\n    for c in NUMS:\n        tmp = train[maska].groupby(['session_id'])[c].agg('mean')\n        new_train = new_train.join(tmp)\n    for c in NUMS:\n        tmp = train[maska].groupby(['session_id'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        new_train = new_train.join(tmp)\n    new_train[\"session_id\"] = new_train.index\n    new_train = time_feature(new_train)\n    new_train.drop(columns=[\"session_id\"], inplace=True)\n    new_train = new_train.fillna(-1)\n    return new_train","metadata":{"papermill":{"duration":0.030427,"end_time":"2023-06-18T11:17:40.781999","exception":false,"start_time":"2023-06-18T11:17:40.751572","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T14:03:12.806915Z","iopub.execute_input":"2023-06-18T14:03:12.807310Z","iopub.status.idle":"2023-06-18T14:03:12.827753Z","shell.execute_reply.started":"2023-06-18T14:03:12.807268Z","shell.execute_reply":"2023-06-18T14:03:12.826636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_quest(new_train, train, q):\n    train_q = new_train.copy()\n    texts = {\n        1: [\"Yes! This cool old slip from 1916.\", \n             \"Go ahead, take a peek at the shirt!\", \n             \"I'll be at the Capitol. Let me know if you find anything!\", \n             \"We need to talk about that missing paperwork.\", \n             \"The slip is from 1916 but the team didn't start until 1974!\"], \n         2: [\"It's already all done!\", \n             \"Gramps is the best historian ever!\"], \n         3: [\"I suppose historians are boring, too?\" \n             \"Why don't you head to the Basketball Center and rustle up some clues?\", \n             \"We need to talk about that missing paperwork.\"],    \n        \n         4: ['I need to find the owner of this slip.',\n             'She led marches and helped women get the right to vote!', \n             \"Here's a call number to find more info in the Stacks.\", \n             \"What was Wells doing here?\"],\n\n         5: [\"Your gramps is awesome! Always full of stories.\",\n             \"Here's a call number to find more info in the Stacks.\", \n             \"Where did you get that coffee?\"],         \n        \n         6: [\"Oh, that's from Bean Town.\", \n             \"Wells? I knew it!\"], \n           \n         7: [\"Try not to panic, Jo.\",\n             \"I've got a stack of business cards from my favorite cleaners.\",\n             \"Check out our microfiche. It's right through that door.\", \n             \"I'm afraid my papers have gone missing in this mess.\", \n             \"Nope. But Youmans and other suffragists worked hard to change that.\"], \n            \n         8: [\"What should I do first?\",\n             \"Thanks to them, Wisconsin was the first state to approve votes for women!\"], \n\n         9: [ \"Can you help me? I need to find the owner of this slip.\",\n             'Looks like a dry cleaning receipt.',\n             \"I knew I could count on you, Jo!\", \n             \"Nope, that's from Bean Town. I only drink Holdgers!\"], \n\n         10:[\"I love these photos of me and Teddy.\"\n             'Your gramps is awesome! Always full of stories.',\n             \"Nope. But Youmans and other suffragists worked hard to change that.\", \n             \"Right outside the door.\", \n             \"Do you have any info on Theodora Youmans?\"], \n                   \n         11:[\"I ran into Wells there this morning\",\n             'Your gramps is awesome! Always full of stories.',\n             \"Wait a sec. Women couldn't vote?!\", \n             \"I've got a stack of business cards from my favorite cleaners.\",\n             \"An old shirt? Try the university.\"],  \n         12:[],\n         13:[],        \n         14:[],\n         15:[],\n         16:[],\n         17:[],\n         18:[]\n        }\n    i = 0\n    for text in texts[q]:\n        i += 1\n        train_q['text' + str(i)] = train[train['text'] == text].groupby(['session_id'])['delt_time'].sum()\n    \n    fqids = {\n         1: ['directory'], \n         2: ['notebook','chap1_finale_c'], \n         3: ['tostacks','doorblock'], \n         4: ['journals.pic_1.next', 'businesscards.card_1.next', 'block'], \n         5: ['janitor', 'journals.pic_2.next'], \n         6: ['businesscards', 'journals.pic_0.next','tobasement', 'logbook.page.bingo', 'tohallway'],  \n         7: ['journals.pic_1.next','reader.paper2.bingo','businesscards.card_bingo.next', \n             'logbook.page.bingo', 'tunic.kohlcenter'],  \n         8: ['reader.paper2.bingo'],  \n         9: ['journals.pic_1.next','businesscards.card_bingo.bingo', 'reader'],  \n         10:['tunic.kohlcenter','magnify','block','journals.pic_1.next', 'journals'], \n         11:['tostacks','block_magnify','block','businesscards.card_bingo.next'], \n         12:['businesscards.card_1.next','tofrontdesk'],  \n         13:['tocloset_dirty','reader.paper1.next'], \n         14:['tracks'], \n         15:['groupconvo_flag'], \n         16:['savedteddy'], \n         17:['journals_flag.pic_0.next'], \n         18:['chap4_finale_c'], \n        }\n    for fqid in fqids[q]:\n        train_q['t_fqid_' + fqid] = train[train['fqid'] == fqid].groupby(['session_id'])['delt_time'].sum()\n\n    text_fqids = {\n        1:[],\n        2:['tunic.historicalsociety.collection.gramps.found'],\n        3:[],\n        4: ['tunic.humanecology.frontdesk.worker.intro',\n            'tunic.library.frontdesk.worker.wells', \n            'tunic.library.frontdesk.worker.hello'], \n        5: ['tunic.humanecology.frontdesk.worker.intro',\n            'tunic.historicalsociety.closet_dirty.gramps.helpclean',\n            'tunic.historicalsociety.closet_dirty.gramps.news'],     \n        6: ['tunic.humanecology.frontdesk.worker.intro',\n            'tunic.historicalsociety.frontdesk.archivist.foundtheodora',\n            'tunic.historicalsociety.closet_dirty.trigger_coffee', \n            'tunic.historicalsociety.closet_dirty.gramps.archivist'], \n        7: ['tunic.historicalsociety.closet_dirty.door_block_talk',\n            'tunic.drycleaner.frontdesk.worker.hub',\n            'tunic.historicalsociety.closet_dirty.trigger_coffee'], \n        8: ['tunic.humanecology.frontdesk.worker.intro',\n            'tunic.historicalsociety.frontdesk.magnify', \n            'tunic.historicalsociety.closet_dirty.trigger_coffee'], \n        9: ['tunic.historicalsociety.frontdesk.archivist.hello',\n            'tunic.library.frontdesk.worker.wells', \n            'tunic.historicalsociety.frontdesk.archivist.foundtheodora'], \n        10: ['tunic.library.frontdesk.worker.wells',\n            'tunic.historicalsociety.frontdesk.archivist.have_glass_recap',\n             'tunic.historicalsociety.closet_dirty.gramps.news'], \n        11: ['tunic.historicalsociety.frontdesk.archivist.newspaper_recap',\n             'tunic.historicalsociety.closet_dirty.gramps.archivist'], \n        12:[],\n        13:['tunic.drycleaner.frontdesk.logbook.page.bingo'],\n        14: ['tunic.flaghouse.entry.flag_girl.symbol_recap', \n             'tunic.historicalsociety.frontdesk.archivist_glasses.confrontation_recap'],\n        15:['tunic.flaghouse.entry.colorbook'], \n        16:['tunic.library.frontdesk.worker.nelson'], \n        17:['tunic.historicalsociety.entry.wells.flag'], \n        18:['tunic.flaghouse.entry.flag_girl.symbol_recap'], \n    }\n    for text_fqid in text_fqids[q]:\n        maska = train['text_fqid'] == text_fqid\n        train_q['t_text_fqid_' + text_fqid] = train[maska].groupby(['session_id'])['delt_time'].sum()       \n        train_q['l_text_fqid_' + text_fqid] = train[train['text_fqid'] == text_fqid].groupby(['session_id'])['index'].count()\n\n\n    room_lvls = {\n         1: [['tunic.capitol_0.hall',4],['tunic.historicalsociety.collection',3],\n            ['tunic.historicalsociety.entry',1],['tunic.historicalsociety.collection', 2]], \n         2: [],\n         3: [['tunic.capitol_0.hall',4]], \n         4: [['tunic.historicalsociety.frontdesk',12], \n             ['tunic.historicalsociety.stacks',7]], \n         5: [['tunic.historicalsociety.stacks',12]],  \n         6: [['tunic.drycleaner.frontdesk',8],  \n             ['tunic.library.microfiche',9]], \n         7: [['tunic.library.frontdesk',10]], \n         8: [['tunic.kohlcenter.halloffame', 11], \n             ['tunic.kohlcenter.halloffame',6]], \n         9: [['tunic.capitol_1.hall', 12], \n             ['tunic.historicalsociety.collection',12]],\n         10:[['tunic.humanecology.frontdesk',7]], \n         11:[['tunic.drycleaner.frontdesk',9], \n             ['tunic.historicalsociety.collection',6]], \n         12:[['tunic.historicalsociety.stacks',6],\n             ['tunic.historicalsociety.frontdesk', 7],\n             ['tunic.historicalsociety.closet_dirty',11], \n             ['tunic.historicalsociety.frontdesk', 12]], \n         13:[['tunic.library.microfiche', 9], \n             ['tunic.historicalsociety.stacks', 11],\n             ['tunic.library.frontdesk', 10], \n             ['tunic.historicalsociety.entry', 5]], \n         14:[['tunic.historicalsociety.closet_dirty',17],\n             ['tunic.historicalsociety.entry',15]], \n         15:[['tunic.historicalsociety.entry',15],\n             ['tunic.library.frontdesk',20]], \n         16:[['tunic.library.frontdesk', 20],\n             ['tunic.wildlife.center',19]], \n         17:[['tunic.wildlife.center', 19],\n             ['tunic.historicalsociety.stacks', 21]], \n         18:[['tunic.wildlife.center', 22]], \n        }\n    for rl in room_lvls[q]:\n        nam = rl[0]+str(rl[1])\n        maska = (train['room_fqid'] == rl[0])&(train['level'] == rl[1])\n        train_q['t_' + nam] = train[maska].groupby(['session_id'])['delt_time'].sum()\n        train_q['l_' + nam] = train[maska].groupby(['session_id'])['index'].count()\n\n    return train_q","metadata":{"papermill":{"duration":0.048867,"end_time":"2023-06-18T11:17:40.838507","exception":false,"start_time":"2023-06-18T11:17:40.789640","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T14:03:12.829256Z","iopub.execute_input":"2023-06-18T14:03:12.829898Z","iopub.status.idle":"2023-06-18T14:03:12.870075Z","shell.execute_reply.started":"2023-06-18T14:03:12.829860Z","shell.execute_reply":"2023-06-18T14:03:12.868911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = {}\nbest_threshold = 0.63\nquests_0_4 =  [1, 3]\nquests_5_12 = [4, 5, 6, 7, 8, 9, 10, 11]\nquests_13_22 = [14, 15, 16, 17]","metadata":{"papermill":{"duration":0.016763,"end_time":"2023-06-18T11:17:40.862844","exception":false,"start_time":"2023-06-18T11:17:40.846081","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T14:03:12.871912Z","iopub.execute_input":"2023-06-18T14:03:12.872289Z","iopub.status.idle":"2023-06-18T14:03:12.891071Z","shell.execute_reply.started":"2023-06-18T14:03:12.872256Z","shell.execute_reply":"2023-06-18T14:03:12.889752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Model Reading\ndir = '/kaggle/input/catbust/'\nfor q in quests_0_4 + quests_5_12 + quests_13_22:\n    models[q] = CatBoostClassifier().load_model(dir+f'cat_model_{q}.bin')","metadata":{"papermill":{"duration":0.224032,"end_time":"2023-06-18T11:17:41.094653","exception":false,"start_time":"2023-06-18T11:17:40.870621","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T14:03:12.892590Z","iopub.execute_input":"2023-06-18T14:03:12.893305Z","iopub.status.idle":"2023-06-18T14:03:12.930111Z","shell.execute_reply.started":"2023-06-18T14:03:12.893261Z","shell.execute_reply":"2023-06-18T14:03:12.928121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jo_wilder\ntry:\n    jo_wilder.make_env.__called__ = False\n    env.__called__ = False\n    type(env)._state = type(type(env)._state).__dict__['INIT']\nexcept:\n    pass\n\nenv = jo_wilder.make_env()\niter_test = env.iter_test() ","metadata":{"papermill":{"duration":0.038295,"end_time":"2023-06-18T11:17:41.140573","exception":false,"start_time":"2023-06-18T11:17:41.102278","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T14:03:12.931774Z","iopub.execute_input":"2023-06-18T14:03:12.932888Z","iopub.status.idle":"2023-06-18T14:03:12.943115Z","shell.execute_reply.started":"2023-06-18T14:03:12.932841Z","shell.execute_reply":"2023-06-18T14:03:12.941529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('/kaggle/input/catboost-predict/importance_dict.pkl', 'rb') as f_read:\n    importance_dict = pickle.load(f_read)","metadata":{"papermill":{"duration":0.027211,"end_time":"2023-06-18T11:17:41.175443","exception":false,"start_time":"2023-06-18T11:17:41.148232","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T14:03:12.945205Z","iopub.execute_input":"2023-06-18T14:03:12.946019Z","iopub.status.idle":"2023-06-18T14:03:12.960350Z","shell.execute_reply.started":"2023-06-18T14:03:12.945978Z","shell.execute_reply":"2023-06-18T14:03:12.958841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nf_read = open('/kaggle/input/catboost-predict/importance_dict.pkl', 'rb')\nimportance_dict = pickle.load(f_read)\nf_read.close()","metadata":{"papermill":{"duration":0.017345,"end_time":"2023-06-18T11:17:41.200354","exception":false,"start_time":"2023-06-18T11:17:41.183009","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T14:03:12.962452Z","iopub.execute_input":"2023-06-18T14:03:12.962877Z","iopub.status.idle":"2023-06-18T14:03:12.972397Z","shell.execute_reply.started":"2023-06-18T14:03:12.962840Z","shell.execute_reply":"2023-06-18T14:03:12.970923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = []","metadata":{"execution":{"iopub.status.busy":"2023-06-18T14:03:12.974331Z","iopub.execute_input":"2023-06-18T14:03:12.974730Z","iopub.status.idle":"2023-06-18T14:03:12.982476Z","shell.execute_reply.started":"2023-06-18T14:03:12.974672Z","shell.execute_reply":"2023-06-18T14:03:12.981183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_q = {'0-4':[1,2,3], '5-12':[4,5,6,7,8,9,10,11,12,13], '13-22':[14,15,16,17,18]}\n\n# Prepare how to mix two models\nlist1m = [15,16,18] \nlist2m = [8,17]\nlist12m = [1,3,4,5,6,7,9,10,11,14] \nlimits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n# Extract test data\nfor (test, sam_sub) in iter_test:\n    sam_sub['question'] = [int(label.split('_')[1][1:]) for label in sam_sub['session_id']]      \n    grp = test.level_group.values[0]  \n    # 1 model\n    # Prepare features for model 1\n    df = (pl.from_pandas(test)\n      .drop([\"fullscreen\", \"hq\", \"music\"])\n      .with_columns(columns))\n    df = feature_engineer(df, grp, use_extra=True, feature_suffix='')\n    df = time_feature(df)\n    \n    # 2 model\n    # Prepare features for model 2\n    old_train = delt_time_def(test[test.level_group == grp])\n    train = feature_engineer_2(old_train)  \n\n    sam_sub['correct'] = 1\n    sam_sub.loc[sam_sub.question.isin([5, 8, 10, 13, 15]), 'correct'] = 0   \n    \n    #grp = test.level_group.values[0]\n    session_id = test.session_id.values[0]\n\n    df = (pl.from_pandas(test)\n          .drop([\"fullscreen\", \"hq\", \"music\"])\n          .with_columns(columns))\n    df = feature_engineer(df, grp, use_extra=True, feature_suffix='')\n    df = time_feature(df)\n\n    fold = 0\n    a,b = limits[grp]\n    for q in range(a, b):\n        if q in list12m:    \n       \n            FEATURES = importance_dict[str(q)]\n            model = models_list[q-1][fold]\n\n            pred = model.predict_proba(df[FEATURES].astype(np.float32).values)[0,1]\n            train_q = feature_quest(train, old_train, q)\n            clf = models[q]\n            p = clf.predict_proba(train_q.astype('float32'))[:,1]\n            # ensemble model, the value can be tuned\n            kmix = 0.6\n            pred = pred*kmix + p*(1-kmix)\n        elif q in list1m:\n            if q == 18:\n                FEATURES = importance_dict[str(q)]\n                model = models_list[q-1][0]\n                pred = model.predict_proba(df[FEATURES].astype(np.float32).values)[0,1]\n            else:\n                FEATURES = importance_dict[str(q)]\n                model = models_list[q-1][0]\n                pred = model.predict_proba(df[FEATURES].astype(np.float32).values)[0,1]         \n\n                # model 2\n                train_q = feature_quest(train, old_train, q)\n                clf = models[q]\n                p = clf.predict_proba(train_q.astype('float32'))[:,1]\n\n                # ensemble model, the value can be tuned\n                kmix = 0.8\n                pred = pred*kmix + p*(1-kmix)\n                \n        elif q in list2m:\n\n            FEATURES = importance_dict[str(q)]\n            model = models_list[q-1][0]\n            pred = model.predict_proba(df[FEATURES].astype(np.float32).values)[0,1]         \n\n            # model 2\n            train_q = feature_quest(train, old_train, q)\n            clf = models[q]\n            p = clf.predict_proba(train_q.astype('float32'))[:,1]\n\n            # ensemble model, the value can be tuned\n            kmix = 0.2\n            pred = pred*kmix + p*(1-kmix)\n        \n        else:\n            continue               \n        mask = sam_sub.question == q\n        x = int(pred > 0.63) \n        #x = np.mean([x,preds],axis=0) \n       # x1 = x*0.6 + preds*0.4\n        sam_sub.loc[mask,'correct'] =x\n\n    sam_sub = sam_sub[['session_id', 'correct']]\n    env.predict(sam_sub)","metadata":{"papermill":{"duration":0.037749,"end_time":"2023-06-18T11:17:43.128395","exception":false,"start_time":"2023-06-18T11:17:43.090646","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-18T14:03:12.984619Z","iopub.execute_input":"2023-06-18T14:03:12.985022Z","iopub.status.idle":"2023-06-18T14:03:17.388326Z","shell.execute_reply.started":"2023-06-18T14:03:12.984982Z","shell.execute_reply":"2023-06-18T14:03:17.386577Z"},"trusted":true},"execution_count":null,"outputs":[]}]}