{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Hi there!👋💥 Thank you for visiting my notebook!🙏\n\n## Summary\n📌 In this [notebook]( https://www.kaggle.com/code/ivanisaev/model-based-on-previous-predictions) I wrote about how to train model on correct labels for other questions in current session. \n\n📌 For this notebook I similarly trained a meta-XGBoost model but instead of correct labels for training I used predictions (probabilities not labels after converting by threshold) of this model on trained dataset. And I use this meta-model to mix it with main model to obtain final predictions.\n\n## My approach\n📌 As a main model I used model from [this notebook](https://www.kaggle.com/code/vadimkamaev/catboost-mix) and trained the meta-model using predictions of this main model on train data.\n\n📌☝️ In the iter test I save predictions for the previous questions for current and previous level groups. This previous predictions are used in final mix by meta-model to make final predictions more robust.\n \n📌 To mix predictions of main Catboost-mix model and meta-model I used similar schema as in [my previous notebook](https://www.kaggle.com/code/ivanisaev/0-697-ensamble-nn-catboost). \n\n📌 Again I didn't receive score improvement in public LB but performance didn't decrease. I hope this is a good sign that model become tore robust for private LB data.\n \n ","metadata":{"papermill":{"duration":0.010803,"end_time":"2023-05-20T14:51:34.920001","exception":false,"start_time":"2023-05-20T14:51:34.909198","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from tqdm.auto import tqdm\nimport os\nimport gc\nimport sys\nimport datetime\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nfrom catboost import CatBoostClassifier, Pool\nfrom xgboost import XGBClassifier\nCATS = ['event_name', 'name', 'fqid', 'room_fqid', 'text_fqid']\nNUMS = ['page', 'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y',\n        'hover_duration', 'elapsed_time_diff']\nDIALOGS = ['that', 'this', 'it', 'you','find','found','Found','notebook','Wells','wells','help','need', 'Oh','Ooh','Jo', 'flag', 'can','and','is','the','to']\n\nname_feature = ['basic', 'undefined', 'close', 'open', 'prev', 'next']\nevent_name_feature = ['cutscene_click', 'person_click', 'navigate_click',\n       'observation_click', 'notification_click', 'object_click',\n       'object_hover', 'map_hover', 'map_click', 'checkpoint',\n       'notebook_click']\n\n\nsub_fqid_lists = {'0-4': ['gramps',\n 'wells',\n 'toentry',\n 'groupconvo',\n 'tomap',\n 'tostacks',\n 'tobasement',\n 'boss',\n 'cs',\n 'teddy',\n 'tunic.historicalsociety',\n 'plaque',\n 'directory',\n 'tunic',\n 'tunic.kohlcenter',\n 'plaque.face.date',\n 'notebook',\n 'tunic.hub.slip',\n 'tocollection',\n 'tunic.capitol_0',\n 'photo',\n 'intro',\n 'retirement_letter',\n 'togrampa',\n 'janitor',\n 'chap1_finale',\n 'report',\n 'outtolunch',\n 'chap1_finale_c',\n 'block_0',\n 'doorblock',\n 'tocloset',\n 'block_tomap2',\n 'block_tocollection',\n 'block_tomap1'],\n                  '5-12': ['worker',\n 'archivist',\n 'gramps',\n 'toentry',\n 'tomap',\n 'tostacks',\n 'tobasement',\n 'boss',\n 'journals',\n 'businesscards',\n 'tunic.historicalsociety',\n 'tofrontdesk',\n 'plaque',\n 'tunic.drycleaner',\n 'tunic.library',\n 'trigger_scarf',\n 'reader',\n 'directory',\n 'tunic.capitol_1',\n 'journals.pic_0.next',\n 'tunic',\n 'what_happened',\n 'tunic.kohlcenter',\n 'tunic.humanecology',\n 'logbook',\n 'businesscards.card_0.next',\n 'journals.hub.topics',\n 'logbook.page.bingo',\n 'journals.pic_1.next',\n 'reader.paper0.next',\n 'trigger_coffee',\n 'wellsbadge',\n 'journals.pic_2.next',\n 'tomicrofiche',\n 'tocloset_dirty',\n 'businesscards.card_bingo.bingo',\n 'businesscards.card_1.next',\n 'tunic.hub.slip',\n 'journals.pic_2.bingo',\n 'tocollection',\n 'chap2_finale_c',\n 'tunic.capitol_0',\n 'photo',\n 'reader.paper1.next',\n 'businesscards.card_bingo.next',\n 'reader.paper2.bingo',\n 'magnify',\n 'janitor',\n 'tohallway',\n 'outtolunch',\n 'reader.paper2.next',\n 'door_block_talk',\n 'block_magnify',\n 'reader.paper0.prev',\n 'block',\n 'block_0',\n 'door_block_clean',\n 'reader.paper2.prev',\n 'reader.paper1.prev',\n 'block_badge',\n 'block_badge_2',\n 'block_1'],\n                  '13-22': ['worker',\n 'gramps',\n 'wells',\n 'toentry',\n 'confrontation',\n 'crane_ranger',\n 'flag_girl',\n 'tomap',\n 'tostacks',\n 'tobasement',\n 'archivist_glasses',\n 'boss',\n 'journals',\n 'seescratches',\n 'groupconvo_flag',\n 'teddy',\n 'expert',\n 'businesscards',\n 'ch3start',\n 'tunic.historicalsociety',\n 'tofrontdesk',\n 'savedteddy',\n 'plaque',\n 'glasses',\n 'tunic.drycleaner',\n 'reader_flag',\n 'tunic.library',\n 'tracks',\n 'tunic.capitol_2',\n 'reader',\n 'directory',\n 'tunic.capitol_1',\n 'journals.pic_0.next',\n 'unlockdoor',\n 'tunic',\n 'tunic.kohlcenter',\n 'tunic.humanecology',\n 'colorbook',\n 'logbook',\n 'businesscards.card_0.next',\n 'journals.hub.topics',\n 'journals.pic_1.next',\n 'journals_flag',\n 'reader.paper0.next',\n 'tracks.hub.deer',\n 'reader_flag.paper0.next',\n 'journals.pic_2.next',\n 'tomicrofiche',\n 'journals_flag.pic_0.bingo',\n 'tocloset_dirty',\n 'businesscards.card_1.next',\n 'tunic.wildlife',\n 'tunic.hub.slip',\n 'tocage',\n 'journals.pic_2.bingo',\n 'tocollectionflag',\n 'tocollection',\n 'chap4_finale_c',\n 'lockeddoor',\n 'journals_flag.hub.topics',\n 'reader_flag.paper2.bingo',\n 'photo',\n 'tunic.flaghouse',\n 'reader.paper1.next',\n 'directory.closeup.archivist',\n 'businesscards.card_bingo.next',\n 'remove_cup',\n 'journals_flag.pic_0.next',\n 'coffee',\n 'key',\n 'reader_flag.paper1.next',\n 'tohallway',\n 'outtolunch',\n 'journals_flag.hub.topics_old',\n 'journals_flag.pic_1.next',\n 'reader.paper2.next',\n 'reader_flag.paper2.next',\n 'journals_flag.pic_1.bingo',\n 'journals_flag.pic_2.next',\n 'journals_flag.pic_2.bingo',\n 'reader.paper0.prev',\n 'reader_flag.paper0.prev',\n 'reader.paper2.prev',\n 'reader.paper1.prev',\n 'reader_flag.paper2.prev',\n 'reader_flag.paper1.prev',\n 'journals_flag.pic_0_old.next',\n 'journals_flag.pic_1_old.next',\n 'block_nelson',\n 'journals_flag.pic_2_old.next',\n 'need_glasses',\n 'fox'],\n                 }\n\nsub_room_lists = {'0-4': ['tunic.historicalsociety.entry',\n 'tunic.historicalsociety.stacks',\n 'tunic.historicalsociety.basement',\n 'tunic.kohlcenter.halloffame',\n 'tunic.historicalsociety.collection',\n 'tunic.historicalsociety.closet',\n 'tunic.capitol_0.hall'],\n                  '5-12': ['tunic.historicalsociety.entry',\n 'tunic.library.frontdesk',\n 'tunic.historicalsociety.frontdesk',\n 'tunic.historicalsociety.stacks',\n 'tunic.historicalsociety.closet_dirty',\n 'tunic.humanecology.frontdesk',\n 'tunic.historicalsociety.basement',\n 'tunic.kohlcenter.halloffame',\n 'tunic.library.microfiche',\n 'tunic.drycleaner.frontdesk',\n 'tunic.historicalsociety.collection',\n 'tunic.capitol_1.hall',\n 'tunic.capitol_0.hall'],\n                  '13-22': ['tunic.historicalsociety.entry',\n 'tunic.wildlife.center',\n 'tunic.historicalsociety.cage',\n 'tunic.library.frontdesk',\n 'tunic.historicalsociety.frontdesk',\n 'tunic.historicalsociety.stacks',\n 'tunic.historicalsociety.closet_dirty',\n 'tunic.humanecology.frontdesk',\n 'tunic.historicalsociety.basement',\n 'tunic.kohlcenter.halloffame',\n 'tunic.library.microfiche',\n 'tunic.drycleaner.frontdesk',\n 'tunic.historicalsociety.collection',\n 'tunic.flaghouse.entry',\n 'tunic.historicalsociety.collection_flag',\n 'tunic.capitol_1.hall',\n 'tunic.capitol_2.hall'],\n                 }\n\n\nsub_text_lists = {'0-4': ['tunic.historicalsociety.entry.groupconvo',\n 'tunic.historicalsociety.collection.cs',\n 'tunic.historicalsociety.collection.gramps.found',\n 'tunic.historicalsociety.closet.gramps.intro_0_cs_0',\n 'tunic.historicalsociety.closet.teddy.intro_0_cs_0',\n 'tunic.historicalsociety.closet.intro',\n 'tunic.historicalsociety.closet.retirement_letter.hub',\n 'tunic.historicalsociety.collection.tunic.slip',\n 'tunic.kohlcenter.halloffame.plaque.face.date',\n 'tunic.kohlcenter.halloffame.togrampa',\n 'tunic.historicalsociety.collection.gramps.lost',\n 'tunic.historicalsociety.closet.notebook',\n 'tunic.historicalsociety.basement.janitor',\n 'tunic.historicalsociety.stacks.outtolunch',\n 'tunic.historicalsociety.closet.photo',\n 'tunic.historicalsociety.collection.tunic',\n 'tunic.historicalsociety.closet.teddy.intro_0_cs_5',\n 'tunic.historicalsociety.entry.wells.talktogramps',\n 'tunic.historicalsociety.entry.boss.talktogramps',\n 'tunic.historicalsociety.closet.doorblock',\n 'tunic.historicalsociety.entry.block_tomap2',\n 'tunic.historicalsociety.entry.block_tocollection',\n 'tunic.historicalsociety.entry.block_tomap1',\n 'tunic.historicalsociety.collection.gramps.look_0',\n 'tunic.kohlcenter.halloffame.block_0',\n 'tunic.capitol_0.hall.chap1_finale_c',\n 'tunic.historicalsociety.entry.gramps.hub'],\n               '5-12': ['tunic.historicalsociety.frontdesk.archivist.newspaper',\n 'tunic.historicalsociety.frontdesk.archivist.have_glass',\n 'tunic.drycleaner.frontdesk.worker.hub',\n 'tunic.historicalsociety.closet_dirty.gramps.news',\n 'tunic.humanecology.frontdesk.worker.intro',\n 'tunic.library.frontdesk.worker.hello',\n 'tunic.library.frontdesk.worker.wells',\n 'tunic.historicalsociety.frontdesk.archivist.hello',\n 'tunic.historicalsociety.closet_dirty.trigger_scarf',\n 'tunic.drycleaner.frontdesk.worker.done',\n 'tunic.historicalsociety.closet_dirty.what_happened',\n 'tunic.historicalsociety.stacks.journals.pic_2.bingo',\n 'tunic.humanecology.frontdesk.worker.badger',\n 'tunic.historicalsociety.closet_dirty.trigger_coffee',\n 'tunic.drycleaner.frontdesk.logbook.page.bingo',\n 'tunic.library.microfiche.reader.paper2.bingo',\n 'tunic.historicalsociety.closet_dirty.gramps.helpclean',\n 'tunic.historicalsociety.frontdesk.archivist.have_glass_recap',\n 'tunic.historicalsociety.frontdesk.magnify',\n 'tunic.humanecology.frontdesk.businesscards.card_bingo.bingo',\n 'tunic.library.frontdesk.wellsbadge.hub',\n 'tunic.capitol_1.hall.boss.haveyougotit',\n 'tunic.historicalsociety.basement.janitor',\n 'tunic.historicalsociety.closet_dirty.photo',\n 'tunic.historicalsociety.stacks.outtolunch',\n 'tunic.library.frontdesk.worker.wells_recap',\n 'tunic.capitol_0.hall.boss.talktogramps',\n 'tunic.historicalsociety.closet_dirty.gramps.archivist',\n 'tunic.historicalsociety.closet_dirty.door_block_talk',\n 'tunic.historicalsociety.frontdesk.archivist.need_glass_0',\n 'tunic.historicalsociety.frontdesk.block_magnify',\n 'tunic.historicalsociety.frontdesk.archivist.foundtheodora',\n 'tunic.historicalsociety.closet_dirty.gramps.nothing',\n 'tunic.historicalsociety.closet_dirty.door_block_clean',\n 'tunic.library.frontdesk.worker.hello_short',\n 'tunic.historicalsociety.stacks.block',\n 'tunic.historicalsociety.frontdesk.archivist.need_glass_1',\n 'tunic.historicalsociety.frontdesk.archivist.newspaper_recap',\n 'tunic.drycleaner.frontdesk.worker.done2',\n 'tunic.humanecology.frontdesk.block_0',\n 'tunic.library.frontdesk.worker.preflag',\n 'tunic.drycleaner.frontdesk.worker.takealook',\n 'tunic.library.frontdesk.worker.droppedbadge',\n 'tunic.library.microfiche.block_0',\n 'tunic.library.frontdesk.block_badge',\n 'tunic.library.frontdesk.block_badge_2',\n 'tunic.capitol_1.hall.chap2_finale_c',\n 'tunic.drycleaner.frontdesk.block_0',\n 'tunic.humanecology.frontdesk.block_1',\n 'tunic.drycleaner.frontdesk.block_1'],\n               '13-22': ['tunic.historicalsociety.cage.confrontation',\n 'tunic.wildlife.center.crane_ranger.crane',\n 'tunic.wildlife.center.wells.nodeer',\n 'tunic.historicalsociety.frontdesk.archivist_glasses.confrontation',\n 'tunic.historicalsociety.basement.seescratches',\n 'tunic.flaghouse.entry.flag_girl.hello',\n 'tunic.historicalsociety.basement.ch3start',\n 'tunic.historicalsociety.entry.groupconvo_flag',\n 'tunic.historicalsociety.collection_flag.gramps.flag',\n 'tunic.historicalsociety.basement.savedteddy',\n 'tunic.library.frontdesk.worker.nelson',\n 'tunic.wildlife.center.expert.removed_cup',\n 'tunic.library.frontdesk.worker.flag',\n 'tunic.historicalsociety.entry.boss.flag',\n 'tunic.flaghouse.entry.flag_girl.symbol',\n 'tunic.wildlife.center.wells.animals',\n 'tunic.historicalsociety.cage.glasses.afterteddy',\n 'tunic.historicalsociety.cage.teddy.trapped',\n 'tunic.historicalsociety.cage.unlockdoor',\n 'tunic.historicalsociety.stacks.journals.pic_2.bingo',\n 'tunic.historicalsociety.entry.wells.flag',\n 'tunic.humanecology.frontdesk.worker.badger',\n 'tunic.historicalsociety.stacks.journals_flag.pic_0.bingo',\n 'tunic.historicalsociety.entry.directory.closeup.archivist',\n 'tunic.capitol_2.hall.boss.haveyougotit',\n 'tunic.wildlife.center.wells.nodeer_recap',\n 'tunic.historicalsociety.cage.glasses.beforeteddy',\n 'tunic.wildlife.center.expert.recap',\n 'tunic.historicalsociety.stacks.journals_flag.pic_1.bingo',\n 'tunic.historicalsociety.cage.lockeddoor',\n 'tunic.historicalsociety.stacks.journals_flag.pic_2.bingo',\n 'tunic.wildlife.center.remove_cup',\n 'tunic.wildlife.center.tracks.hub.deer',\n 'tunic.historicalsociety.frontdesk.key',\n 'tunic.library.microfiche.reader_flag.paper2.bingo',\n 'tunic.flaghouse.entry.colorbook',\n 'tunic.wildlife.center.coffee',\n 'tunic.historicalsociety.collection_flag.gramps.recap',\n 'tunic.wildlife.center.wells.animals2',\n 'tunic.flaghouse.entry.flag_girl.symbol_recap',\n 'tunic.historicalsociety.closet_dirty.photo',\n 'tunic.historicalsociety.stacks.outtolunch',\n 'tunic.historicalsociety.frontdesk.archivist_glasses.confrontation_recap',\n 'tunic.historicalsociety.entry.boss.flag_recap',\n 'tunic.capitol_1.hall.boss.writeitup',\n 'tunic.library.frontdesk.worker.nelson_recap',\n 'tunic.historicalsociety.entry.wells.flag_recap',\n 'tunic.drycleaner.frontdesk.worker.done2',\n 'tunic.library.frontdesk.worker.flag_recap',\n 'tunic.library.frontdesk.worker.preflag',\n 'tunic.historicalsociety.basement.gramps.seeyalater',\n 'tunic.flaghouse.entry.flag_girl.hello_recap',\n 'tunic.historicalsociety.basement.gramps.whatdo',\n 'tunic.library.frontdesk.block_nelson',\n 'tunic.historicalsociety.cage.need_glasses',\n 'tunic.capitol_2.hall.chap4_finale_c',\n 'tunic.wildlife.center.fox.concern']\n              }\n\n\nSUB_LEVELS = {'0-4': [1, 2, 3, 4],\n              '5-12': [5, 6, 7, 8, 9, 10, 11, 12],\n              '13-22': [13, 14, 15, 16, 17, 18, 19, 20, 21, 22]}\nlevel_groups = [\"0-4\", \"5-12\", \"13-22\"]","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":1.924503,"end_time":"2023-05-20T14:51:36.854444","exception":false,"start_time":"2023-05-20T14:51:34.929941","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-24T20:48:43.126551Z","iopub.execute_input":"2023-05-24T20:48:43.126990Z","iopub.status.idle":"2023-05-24T20:48:43.174604Z","shell.execute_reply.started":"2023-05-24T20:48:43.126952Z","shell.execute_reply":"2023-05-24T20:48:43.173374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = [\n    pl.col(\"page\").cast(pl.Float32),\n    (\n        (pl.col(\"elapsed_time\") - pl.col(\"elapsed_time\").shift(1))\n        .fill_null(0)\n        .clip(0, 1e9)\n        .over([\"session_id\", \"level\"])\n        .alias(\"elapsed_time_diff\")\n    ),\n    (\n        (pl.col(\"screen_coor_x\") - pl.col(\"screen_coor_x\").shift(1))\n        .abs()\n        .over([\"session_id\", \"level\"])\n    ),\n    (\n        (pl.col(\"screen_coor_y\") - pl.col(\"screen_coor_y\").shift(1))\n        .abs()\n        .over([\"session_id\", \"level\"])\n    ),\n    pl.col(\"fqid\").fill_null(\"fqid_None\"),\n    pl.col(\"text_fqid\").fill_null(\"text_fqid_None\")\n\n]\n\nmodels_list = [[CatBoostClassifier().load_model(\n    f\"/kaggle/input/catboost-predict/fold{fold}_q{q}.cbm\"\n) for fold in range(5)] for q in range(1, 19)]","metadata":{"papermill":{"duration":1.310853,"end_time":"2023-05-20T14:51:38.174979","exception":false,"start_time":"2023-05-20T14:51:36.864126","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-24T20:48:43.177320Z","iopub.execute_input":"2023-05-24T20:48:43.177934Z","iopub.status.idle":"2023-05-24T20:48:44.830015Z","shell.execute_reply.started":"2023-05-24T20:48:43.177873Z","shell.execute_reply":"2023-05-24T20:48:44.828859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(x, grp, use_extra, feature_suffix):\n    LEVELS = SUB_LEVELS[grp]\n    text_lists = sub_text_lists[grp]\n    room_lists = sub_room_lists[grp]\n    fqid_lists = sub_fqid_lists[grp]\n    aggs = [\n        pl.col(\"index\").count().alias(f\"session_number_{feature_suffix}\"),\n\n        *[pl.col('index').filter(pl.col('text').str.contains(c)).count().alias(f'word_{c}') for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).mean().alias(f'word_mean_{c}') for c in\n          DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).std().alias(f'word_std_{c}') for c in\n          DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).max().alias(f'word_max_{c}') for c in\n          DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).sum().alias(f'word_sum_{c}') for c in\n          DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col('text').str.contains(c))).median().alias(f'word_median_{c}') for c\n          in DIALOGS],\n\n        *[pl.col(c).drop_nulls().n_unique().alias(f\"{c}_unique_{feature_suffix}\") for c in CATS],\n\n        *[pl.col(c).mean().alias(f\"{c}_mean_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).std().alias(f\"{c}_std_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).min().alias(f\"{c}_min_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).max().alias(f\"{c}_max_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).median().alias(f\"{c}_median_{feature_suffix}\") for c in NUMS],\n\n        *[pl.col(\"fqid\").filter(pl.col(\"fqid\") == c).count().alias(f\"{c}_fqid_counts{feature_suffix}\")\n          for c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for\n          c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\") for\n          c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in fqid_lists],\n\n        *[pl.col(\"text_fqid\").filter(pl.col(\"text_fqid\") == c).count().alias(f\"{c}_text_fqid_counts{feature_suffix}\")\n          for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\")\n          for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in text_lists],\n\n        *[pl.col(\"room_fqid\").filter(pl.col(\"room_fqid\") == c).count().alias(f\"{c}_room_fqid_counts{feature_suffix}\")\n          for c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for\n          c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\")\n          for\n          c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in room_lists],\n\n        *[pl.col(\"event_name\").filter(pl.col(\"event_name\") == c).count().alias(f\"{c}_event_name_counts{feature_suffix}\")\n          for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\")\n          for\n          c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).median().alias(\n            f\"{c}_ET_median_{feature_suffix}\") for\n          c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in event_name_feature],\n\n        *[pl.col(\"name\").filter(pl.col(\"name\") == c).count().alias(f\"{c}_name_counts{feature_suffix}\") for c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\") for\n          c in\n          name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in\n          name_feature],\n\n        *[pl.col(\"level\").filter(pl.col(\"level\") == c).count().alias(f\"{c}_LEVEL_count{feature_suffix}\") for c in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c\n          in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for c in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).median().alias(f\"{c}_ET_median_{feature_suffix}\") for\n          c in\n          LEVELS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in\n          LEVELS],\n\n        *[pl.col(\"level_group\").filter(pl.col(\"level_group\") == c).count().alias(\n            f\"{c}_LEVEL_group_count{feature_suffix}\") for c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\")\n          for c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).median().alias(\n            f\"{c}_ET_median_{feature_suffix}\") for c in\n          level_groups],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level_group\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in\n          level_groups],\n\n    ]\n\n    df = x.groupby(['session_id'], maintain_order=True).agg(aggs).sort(\"session_id\")\n\n    if use_extra:\n        if grp == '5-12':\n            aggs = [\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Here's the log book.\")\n                                              | (pl.col(\"fqid\") == 'logbook.page.bingo'))\n                    .apply(lambda s: s.max() - s.min()).alias(\"logbook_bingo_duration\"),\n                pl.col(\"index\").filter(\n                    (pl.col(\"text\") == \"Here's the log book.\") | (pl.col(\"fqid\") == 'logbook.page.bingo')).apply(\n                    lambda s: s.max() - s.min()).alias(\"logbook_bingo_indexCount\"),\n                pl.col(\"elapsed_time\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader')) | (\n                            pl.col(\"fqid\") == \"reader.paper2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\n                    \"reader_bingo_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader')) | (\n                        pl.col(\"fqid\") == \"reader.paper2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\n                    \"reader_bingo_indexCount\"),\n                pl.col(\"elapsed_time\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals')) | (\n                            pl.col(\"fqid\") == \"journals.pic_2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\n                    \"journals_bingo_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals')) | (\n                        pl.col(\"fqid\") == \"journals.pic_2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\n                    \"journals_bingo_indexCount\"),\n            ]\n            tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n            df = df.join(tmp, on=\"session_id\", how='left')\n\n        if grp == '13-22':\n            aggs = [\n                pl.col(\"elapsed_time\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader_flag')) | (\n                            pl.col(\"fqid\") == \"tunic.library.microfiche.reader_flag.paper2.bingo\")).apply(\n                    lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"reader_flag_duration\"),\n                pl.col(\"index\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader_flag')) | (\n                            pl.col(\"fqid\") == \"tunic.library.microfiche.reader_flag.paper2.bingo\")).apply(\n                    lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"reader_flag_indexCount\"),\n                pl.col(\"elapsed_time\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals_flag')) | (\n                            pl.col(\"fqid\") == \"journals_flag.pic_0.bingo\")).apply(\n                    lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"journalsFlag_bingo_duration\"),\n                pl.col(\"index\").filter(\n                    ((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals_flag')) | (\n                            pl.col(\"fqid\") == \"journals_flag.pic_0.bingo\")).apply(\n                    lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"journalsFlag_bingo_indexCount\")\n            ]\n            tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n            df = df.join(tmp, on=\"session_id\", how='left')\n\n    return df.to_pandas()","metadata":{"papermill":{"duration":0.062784,"end_time":"2023-05-20T14:51:38.248014","exception":false,"start_time":"2023-05-20T14:51:38.185230","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-24T20:48:44.831936Z","iopub.execute_input":"2023-05-24T20:48:44.832656Z","iopub.status.idle":"2023-05-24T20:48:44.916439Z","shell.execute_reply.started":"2023-05-24T20:48:44.832615Z","shell.execute_reply":"2023-05-24T20:48:44.915098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def time_feature(train):\n    train[\"year\"] = train[\"session_id\"].apply(lambda x: int(str(x)[:2])).astype(np.uint8)\n    train[\"month\"] = train[\"session_id\"].apply(lambda x: int(str(x)[2:4])+1).astype(np.uint8)\n    train[\"day\"] = train[\"session_id\"].apply(lambda x: int(str(x)[4:6])).astype(np.uint8)\n    train[\"hour\"] = train[\"session_id\"].apply(lambda x: int(str(x)[6:8])).astype(np.uint8)\n    train[\"minute\"] = train[\"session_id\"].apply(lambda x: int(str(x)[8:10])).astype(np.uint8)\n    train[\"second\"] = train[\"session_id\"].apply(lambda x: int(str(x)[10:12])).astype(np.uint8)\n    return train","metadata":{"papermill":{"duration":0.020071,"end_time":"2023-05-20T14:51:38.277570","exception":false,"start_time":"2023-05-20T14:51:38.257499","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-24T20:48:44.919518Z","iopub.execute_input":"2023-05-24T20:48:44.920168Z","iopub.status.idle":"2023-05-24T20:48:44.930807Z","shell.execute_reply.started":"2023-05-24T20:48:44.920128Z","shell.execute_reply":"2023-05-24T20:48:44.929480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**ДРУГОЙ КАТБУСТ**","metadata":{"papermill":{"duration":0.009524,"end_time":"2023-05-20T14:51:38.297018","exception":false,"start_time":"2023-05-20T14:51:38.287494","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def delt_time_def(df):\n    df.sort_values(by=['session_id', 'elapsed_time'], inplace=True)\n    df['d_time'] = df['elapsed_time'].diff(1)\n    df['d_time'].fillna(0, inplace=True)\n    df['delt_time'] = df['d_time'].clip(0, 103000)  \n    return df","metadata":{"papermill":{"duration":0.018774,"end_time":"2023-05-20T14:51:38.325088","exception":false,"start_time":"2023-05-20T14:51:38.306314","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-24T20:48:44.932700Z","iopub.execute_input":"2023-05-24T20:48:44.933063Z","iopub.status.idle":"2023-05-24T20:48:44.943350Z","shell.execute_reply.started":"2023-05-24T20:48:44.933028Z","shell.execute_reply":"2023-05-24T20:48:44.942260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer_2(train):\n    CATS = ['event_name', 'fqid', 'room_fqid', 'text_fqid', 'page']\n    NUMS = ['delt_time', 'hover_duration']\n    EV_NAME = ['checkpoint','observation_click', 'cutscene_click', 'notification_click', 'person_click',\n               'object_click', 'map_click', 'object_hover']    \n    new_train = pd.DataFrame(index=train['session_id'].unique(), columns=[])\n    for c in EV_NAME:\n        new_train['l_ev_name_' + c] = train[train['event_name'] == c].groupby(['session_id'])['index'].count()\n        new_train['t_ev_name_' + c] = train[train['event_name'] == c].groupby(['session_id'])['delt_time'].sum()\n    maska = train['name'] == 'basic'\n    \n    # ДОБАВЛЯЕМ КВАНТИЛИ\n    qvant = train.groupby(['session_id'])['d_time'].quantile(q=0.3)\n    qvant.name = 'qvant1_0_3'\n    new_train = new_train.join(qvant)\n\n    qvant = train.groupby(['session_id'])['d_time'].quantile(q=0.8)\n    qvant.name = 'qvant2_0_8'\n    new_train = new_train.join(qvant)\n\n    qvant = train.groupby(['session_id'])['d_time'].quantile(q=0.5)\n    qvant.name = 'qvant3_0_5'\n    new_train = new_train.join(qvant)\n\n    qvant = train.groupby(['session_id'])['d_time'].quantile(q=0.65)\n    qvant.name = 'qvant4_0_65'\n    new_train = new_train.join(qvant)\n    \n    new_train['finish'] = train[maska].groupby(['session_id'])['elapsed_time'].last(1)  \n    new_train['len'] = train[maska].groupby(['session_id'])['index'].count()\n    for c in CATS:\n        tmp = train[maska].groupby(['session_id'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        new_train = new_train.join(tmp)\n    for c in NUMS:\n        tmp = train[maska].groupby(['session_id'])[c].agg('mean')\n        new_train = new_train.join(tmp)\n    for c in NUMS:\n        tmp = train[maska].groupby(['session_id'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        new_train = new_train.join(tmp)\n    new_train[\"session_id\"] = new_train.index\n    new_train = time_feature(new_train)\n    new_train.drop(columns=[\"session_id\"], inplace=True)\n    new_train = new_train.fillna(-1)\n    return new_train","metadata":{"papermill":{"duration":0.026032,"end_time":"2023-05-20T14:51:38.361321","exception":false,"start_time":"2023-05-20T14:51:38.335289","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-24T20:48:44.945164Z","iopub.execute_input":"2023-05-24T20:48:44.945549Z","iopub.status.idle":"2023-05-24T20:48:44.965104Z","shell.execute_reply.started":"2023-05-24T20:48:44.945510Z","shell.execute_reply":"2023-05-24T20:48:44.964157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_quest(new_train, train, q):\n    train_q = new_train.copy()\n    texts = {\n        1: [\"Yes! This cool old slip from 1916.\", \n             \"Go ahead, take a peek at the shirt!\", \n             \"I'll be at the Capitol. Let me know if you find anything!\", \n             \"We need to talk about that missing paperwork.\", \n             \"The slip is from 1916 but the team didn't start until 1974!\"], \n         2: [\"It's already all done!\", \n             \"Gramps is the best historian ever!\"], \n         3: [\"I suppose historians are boring, too?\" \n             \"Why don't you head to the Basketball Center and rustle up some clues?\", \n             \"We need to talk about that missing paperwork.\"],    \n        \n         4: ['I need to find the owner of this slip.',\n             'She led marches and helped women get the right to vote!', \n             \"Here's a call number to find more info in the Stacks.\", \n             \"What was Wells doing here?\"],\n\n         5: [\"Your gramps is awesome! Always full of stories.\",\n             \"Here's a call number to find more info in the Stacks.\", \n             \"Where did you get that coffee?\"],         \n        \n         6: [\"Oh, that's from Bean Town.\", \n             \"Wells? I knew it!\"], \n           \n         7: [\"Try not to panic, Jo.\",\n             \"I've got a stack of business cards from my favorite cleaners.\",\n             \"Check out our microfiche. It's right through that door.\", \n             \"I'm afraid my papers have gone missing in this mess.\", \n             \"Nope. But Youmans and other suffragists worked hard to change that.\"], \n            \n         8: [\"What should I do first?\",\n             \"Thanks to them, Wisconsin was the first state to approve votes for women!\"], \n\n         9: [ \"Can you help me? I need to find the owner of this slip.\",\n             'Looks like a dry cleaning receipt.',\n             \"I knew I could count on you, Jo!\", \n             \"Nope, that's from Bean Town. I only drink Holdgers!\"], \n\n         10:[\"I love these photos of me and Teddy.\"\n             'Your gramps is awesome! Always full of stories.',\n             \"Nope. But Youmans and other suffragists worked hard to change that.\", \n             \"Right outside the door.\", \n             \"Do you have any info on Theodora Youmans?\"], \n                   \n         11:[\"I ran into Wells there this morning\",\n             'Your gramps is awesome! Always full of stories.',\n             \"Wait a sec. Women couldn't vote?!\", \n             \"I've got a stack of business cards from my favorite cleaners.\",\n             \"An old shirt? Try the university.\"],  \n         12:[],\n         13:[],        \n         14:[],\n         15:[],\n         16:[],\n         17:[],\n         18:[]\n        }\n    i = 0\n    for text in texts[q]:\n        i += 1\n        train_q['text' + str(i)] = train[train['text'] == text].groupby(['session_id'])['delt_time'].sum()\n    \n    fqids = {\n         1: ['directory'], \n         2: ['notebook','chap1_finale_c'], \n         3: ['tostacks','doorblock'], \n         4: ['journals.pic_1.next', 'businesscards.card_1.next', 'block'], \n         5: ['janitor', 'journals.pic_2.next'], \n         6: ['businesscards', 'journals.pic_0.next','tobasement', 'logbook.page.bingo', 'tohallway'],  \n         7: ['journals.pic_1.next','reader.paper2.bingo','businesscards.card_bingo.next', \n             'logbook.page.bingo', 'tunic.kohlcenter'],  \n         8: ['reader.paper2.bingo'],  \n         9: ['journals.pic_1.next','businesscards.card_bingo.bingo', 'reader'],  \n         10:['tunic.kohlcenter','magnify','block','journals.pic_1.next', 'journals'], \n         11:['tostacks','block_magnify','block','businesscards.card_bingo.next'], \n         12:['businesscards.card_1.next','tofrontdesk'],  \n         13:['tocloset_dirty','reader.paper1.next'], \n         14:['tracks'], \n         15:['groupconvo_flag'], \n         16:['savedteddy'], \n         17:['journals_flag.pic_0.next'], \n         18:['chap4_finale_c'], \n        }\n    for fqid in fqids[q]:\n        train_q['t_fqid_' + fqid] = train[train['fqid'] == fqid].groupby(['session_id'])['delt_time'].sum()\n\n    text_fqids = {\n        1:[],\n        2:['tunic.historicalsociety.collection.gramps.found'],\n        3:[],\n        4: ['tunic.humanecology.frontdesk.worker.intro',\n            'tunic.library.frontdesk.worker.wells', \n            'tunic.library.frontdesk.worker.hello'], \n        5: ['tunic.humanecology.frontdesk.worker.intro',\n            'tunic.historicalsociety.closet_dirty.gramps.helpclean',\n            'tunic.historicalsociety.closet_dirty.gramps.news'],     \n        6: ['tunic.humanecology.frontdesk.worker.intro',\n            'tunic.historicalsociety.frontdesk.archivist.foundtheodora',\n            'tunic.historicalsociety.closet_dirty.trigger_coffee', \n            'tunic.historicalsociety.closet_dirty.gramps.archivist'], \n        7: ['tunic.historicalsociety.closet_dirty.door_block_talk',\n            'tunic.drycleaner.frontdesk.worker.hub',\n            'tunic.historicalsociety.closet_dirty.trigger_coffee'], \n        8: ['tunic.humanecology.frontdesk.worker.intro',\n            'tunic.historicalsociety.frontdesk.magnify', \n            'tunic.historicalsociety.closet_dirty.trigger_coffee'], \n        9: ['tunic.historicalsociety.frontdesk.archivist.hello',\n            'tunic.library.frontdesk.worker.wells', \n            'tunic.historicalsociety.frontdesk.archivist.foundtheodora'], \n        10: ['tunic.library.frontdesk.worker.wells',\n            'tunic.historicalsociety.frontdesk.archivist.have_glass_recap',\n             'tunic.historicalsociety.closet_dirty.gramps.news'], \n        11: ['tunic.historicalsociety.frontdesk.archivist.newspaper_recap',\n             'tunic.historicalsociety.closet_dirty.gramps.archivist'], \n        12:[],\n        13:['tunic.drycleaner.frontdesk.logbook.page.bingo'],\n        14: ['tunic.flaghouse.entry.flag_girl.symbol_recap', \n             'tunic.historicalsociety.frontdesk.archivist_glasses.confrontation_recap'],\n        15:['tunic.flaghouse.entry.colorbook'], \n        16:['tunic.library.frontdesk.worker.nelson'], \n        17:['tunic.historicalsociety.entry.wells.flag'], \n        18:['tunic.flaghouse.entry.flag_girl.symbol_recap'], \n    }\n    for text_fqid in text_fqids[q]:\n        maska = train['text_fqid'] == text_fqid\n        train_q['t_text_fqid_' + text_fqid] = train[maska].groupby(['session_id'])['delt_time'].sum()       \n        train_q['l_text_fqid_' + text_fqid] = train[train['text_fqid'] == text_fqid].groupby(['session_id'])['index'].count()\n\n\n    room_lvls = {\n         1: [['tunic.capitol_0.hall',4],['tunic.historicalsociety.collection',3],\n            ['tunic.historicalsociety.entry',1],['tunic.historicalsociety.collection', 2]], \n         2: [],\n         3: [['tunic.capitol_0.hall',4]], \n         4: [['tunic.historicalsociety.frontdesk',12], \n             ['tunic.historicalsociety.stacks',7]], \n         5: [['tunic.historicalsociety.stacks',12]],  \n         6: [['tunic.drycleaner.frontdesk',8],  \n             ['tunic.library.microfiche',9]], \n         7: [['tunic.library.frontdesk',10]], \n         8: [['tunic.kohlcenter.halloffame', 11], \n             ['tunic.kohlcenter.halloffame',6]], \n         9: [['tunic.capitol_1.hall', 12], \n             ['tunic.historicalsociety.collection',12]],\n         10:[['tunic.humanecology.frontdesk',7]], \n         11:[['tunic.drycleaner.frontdesk',9], \n             ['tunic.historicalsociety.collection',6]], \n         12:[['tunic.historicalsociety.stacks',6],\n             ['tunic.historicalsociety.frontdesk', 7],\n             ['tunic.historicalsociety.closet_dirty',11], \n             ['tunic.historicalsociety.frontdesk', 12]], \n         13:[['tunic.library.microfiche', 9], \n             ['tunic.historicalsociety.stacks', 11],\n             ['tunic.library.frontdesk', 10], \n             ['tunic.historicalsociety.entry', 5]], \n         14:[['tunic.historicalsociety.closet_dirty',17],\n             ['tunic.historicalsociety.entry',15]], \n         15:[['tunic.historicalsociety.entry',15],\n             ['tunic.library.frontdesk',20]], \n         16:[['tunic.library.frontdesk', 20],\n             ['tunic.wildlife.center',19]], \n         17:[['tunic.wildlife.center', 19],\n             ['tunic.historicalsociety.stacks', 21]], \n         18:[['tunic.wildlife.center', 22]], \n        }\n    for rl in room_lvls[q]:\n        nam = rl[0]+str(rl[1])\n        maska = (train['room_fqid'] == rl[0])&(train['level'] == rl[1])\n        train_q['t_' + nam] = train[maska].groupby(['session_id'])['delt_time'].sum()\n        train_q['l_' + nam] = train[maska].groupby(['session_id'])['index'].count()\n\n    return train_q","metadata":{"papermill":{"duration":0.038716,"end_time":"2023-05-20T14:51:38.409486","exception":false,"start_time":"2023-05-20T14:51:38.370770","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-24T20:48:44.966570Z","iopub.execute_input":"2023-05-24T20:48:44.967138Z","iopub.status.idle":"2023-05-24T20:48:45.003882Z","shell.execute_reply.started":"2023-05-24T20:48:44.967102Z","shell.execute_reply":"2023-05-24T20:48:45.002898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = {}\nbest_threshold = 0.63\nquests_0_4 =  [1, 3]\nquests_5_12 = [4, 5, 6, 7, 8, 9, 10, 11]\nquests_13_22 = [14, 15, 16, 17]","metadata":{"papermill":{"duration":0.018691,"end_time":"2023-05-20T14:51:38.437961","exception":false,"start_time":"2023-05-20T14:51:38.419270","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-24T20:48:45.005283Z","iopub.execute_input":"2023-05-24T20:48:45.005873Z","iopub.status.idle":"2023-05-24T20:48:45.023145Z","shell.execute_reply.started":"2023-05-24T20:48:45.005835Z","shell.execute_reply":"2023-05-24T20:48:45.021724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Model Reading\ndir = '/kaggle/input/catbust/'\nfor q in quests_0_4 + quests_5_12 + quests_13_22:\n    models[q] = CatBoostClassifier().load_model(dir+f'cat_model_{q}.bin')","metadata":{"papermill":{"duration":0.187971,"end_time":"2023-05-20T14:51:38.635868","exception":false,"start_time":"2023-05-20T14:51:38.447897","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-24T20:48:45.024790Z","iopub.execute_input":"2023-05-24T20:48:45.025174Z","iopub.status.idle":"2023-05-24T20:48:45.220456Z","shell.execute_reply.started":"2023-05-24T20:48:45.025138Z","shell.execute_reply":"2023-05-24T20:48:45.219427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nimport os\nfrom typing import Any, Dict, List, Optional, Tuple\nimport pickle\nimport warnings\nwarnings.simplefilter(\"ignore\")\n\nimport numpy as np\nimport pandas as pd\nimport yaml\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch import Tensor\n\nexp_dump_path = \"/kaggle/input/0330-16-53-13/0330-16_53_13\"\ncat2code_path = \"/kaggle/input/cat2code\"\n\nCAT_FEATS = [\"event_comb_code\", \"room_fqid_code\"]\nCAT_FEAT_SIZE = {\n    \"event_comb_code\": 19,\n    \"room_fqid_code\": 19,\n}\nDEVICVE = torch.device(\"cpu\")\n\n# Load model configuration\nwith open(os.path.join(exp_dump_path, \"config/cfg.yaml\"), \"r\") as f:\n    cfg = yaml.full_load(f)\nmodel_cfg = cfg[\"model\"]\n\nbest_thres = 0.63\nt_window = 1000\n\nclass TConvLayer(nn.Module):\n    \"\"\"Dilated temporal convolution layer.\"\"\"\n    \n    def __init__(\n        self,\n        in_dim: int,\n        out_dim: int,\n        kernel_size: int,\n        dilation: int,\n        bias: bool = True,\n        act: str = \"relu\",\n        dropout: float = 0.1,\n    ):\n        super(TConvLayer, self).__init__()\n        \n        # Network parameters\n        self.in_dim = in_dim\n        self.out_dim = out_dim\n        self.kernel_size = kernel_size\n        self.dilation = dilation\n        self.bias = bias\n        \n        # Model blocks\n        self.conv = nn.utils.weight_norm(\n            nn.Conv1d(in_dim, out_dim, kernel_size, dilation=dilation, bias=bias),\n            dim=None\n        )\n        self.bn = nn.BatchNorm1d(out_dim)\n        if act == \"relu\":\n            self.act = nn.ReLU()\n        if dropout == 0:\n            self.dropout = None\n        else:\n            self.dropout = nn.Dropout(dropout)\n            \n    def forward(self, x: Tensor) -> Tensor:\n        \"\"\"Forward pass.\n        \n        Shape:\n            x: (B, C, P)\n        \"\"\"\n        x = self.conv(x)\n        x = self.bn(x)\n        x = self.act(x)\n        if self.dropout is not None:\n            x = self.dropout(x)\n            \n        return x\n    \n\nclass EventAwareEncoder(nn.Module):\n    \n    def __init__(\n        self,\n        h_dim: int = 128,\n        out_dim: int = 128,\n        readout: bool = True,\n        cat_feats: List[str] = [\"event_comb_code\", \"room_fqid_code\"]\n    ):\n        super(EventAwareEncoder, self).__init__()\n\n        # Network parameters\n        self.h_dim = h_dim\n        self.out_dim = out_dim\n        self.cat_feats = cat_feats\n\n        # Model blocks\n        # Categorical embeddings\n        self.embs = nn.ModuleList()\n        for cat_feat in cat_feats:\n            self.embs.append(nn.Embedding(CAT_FEAT_SIZE[cat_feat] + 1, 32, padding_idx=0))\n        self.emb_enc = nn.Sequential(\n            nn.Linear(64, 128),\n            nn.ReLU(),\n            nn.Linear(128, 64),\n            nn.ReLU(),\n        )\n        self.dropout = nn.Dropout(0.2)\n        # Feature extractor\n        self.convs = nn.ModuleList()\n        for l, (dilation, kernel_size) in enumerate(zip([2**i for i in range(3)], [7, 7, 5])):\n            self.convs.append(TConvLayer(64, h_dim, kernel_size, dilation=1))   # No dilation\n        # Readout layer\n        if readout:\n            self.readout = nn.Sequential(\n                nn.Linear(2 * (h_dim // 2), out_dim),\n                nn.ReLU(),\n                nn.Dropout(0.2),\n            )\n        else:\n            self.readout = None\n\n    def forward(self, x: Tensor, x_cat: Tensor) -> Tensor:\n        \"\"\"Forward pass.\n\n        Shape:\n            x: (B, P, C)\n            x_cat: (B, P, #cat_feats)\n        \"\"\"\n\n        # Categorical embeddings\n        x_cat = x_cat + 1\n        x_emb = []\n        for i in range(len(self.cat_feats)):\n            x_emb.append(self.embs[i](x_cat[..., i]))  # (B, P, emb_dim)\n        x_emb = torch.cat(x_emb, dim=-1)  # (B, P, emb_dim_cat)\n        x_emb = self.emb_enc(x_emb) + x_emb  # (B, P, emb_dim_cat)\n        x = x * x_emb  # (B, P, emb_dim_cat)\n        x = self.dropout(x)\n        \n        # Feature extractor\n        x = x.transpose(1, 2)  # (B, C, P)\n        x_skip = []\n        for l in range(3):\n            x_conv = self.convs[l](x)   # (B, C * 2, P')\n            x_filter, x_gate = torch.split(x_conv, x_conv.size(1) // 2, dim=1)\n            x_conv = F.tanh(x_filter) * F.sigmoid(x_gate)   # (B, C, P')\n            \n            x_conv = self.dropout(x_conv)\n            \n            # Skip connection\n            x_skip.append(x_conv.unsqueeze(dim=1))  # (B, L (1), C, P')\n            \n            # Residual\n            x = x_conv\n            \n        # Process skipped latent representation\n        for l in range(3-1):\n            x_skip[l] = x_skip[l][..., -x_skip[-1].size(3) : ]\n        x_skip = torch.cat(x_skip, dim=1)   # (B, L, C, P)\n        x_skip = torch.sum(x_skip, dim=1)   # (B, C, P')\n\n        # Readout layer\n        if self.readout is not None:\n            x_std = torch.std(x_skip, dim=-1)  # Std pooling\n            x_mean = torch.mean(x_skip, dim=-1)  # Mean pooling\n            x = torch.cat([x_std, x_mean], dim=1)\n            x = self.readout(x)  # (B, out_dim)\n\n        return x\n\n\nclass EventConvSimple(nn.Module):\n\n    def __init__(self, n_lvs: int, out_dim: int, **model_cfg: Any):\n        self.name = self.__class__.__name__\n        super(EventConvSimple, self).__init__()\n\n        enc_out_dim = 128\n\n        # Network parameters\n        self.n_lvs = n_lvs\n        self.out_dim = out_dim\n        self.cat_feats = model_cfg[\"cat_feats\"]\n        \n        self.encoder = EventAwareEncoder(h_dim=128, out_dim=enc_out_dim, cat_feats=self.cat_feats)\n        self.clf = nn.Sequential(\n            nn.Linear(enc_out_dim, enc_out_dim // 2),\n            nn.ReLU(),\n            nn.Dropout(0.2),\n            nn.Linear(enc_out_dim // 2, out_dim),\n        )\n\n    def forward(self, x: Tensor, x_cat: Tensor) -> Tensor:\n        \"\"\"Forward pass.\n\n        Shape:\n            x: (B, P, C)\n            x_cat: (B, P, 3)\n        \"\"\"\n        x = self.encoder(x, x_cat)\n        x = self.clf(x)\n        \n        x = F.sigmoid(x)\n\n        return x\n    \n\nmodel_path = os.path.join(exp_dump_path, \"models\")\n\nmodels_tconv = {\"0-4\": [], \"5-12\": [], \"13-22\": []}\nfor model_file in os.listdir(model_path):\n    if \"encoder\" in model_file: continue\n    if \"0-4\" in model_file:\n        model = EventConvSimple(5, 3, **model_cfg)\n        model.load_state_dict(\n            torch.load(\n                os.path.join(model_path, model_file),\n                map_location=torch.device('cpu')\n            )\n        )\n        models_tconv[\"0-4\"].append(model)\n    elif \"5-12\" in model_file:\n        model = EventConvSimple(8, 10, **model_cfg)\n        model.load_state_dict(\n            torch.load(\n                os.path.join(model_path, model_file),\n                map_location=torch.device('cpu')\n            )\n        )\n        models_tconv[\"5-12\"].append(model)\n    elif \"13-22\" in model_file:\n        model = EventConvSimple(10, 5, **model_cfg)\n        model.load_state_dict(\n            torch.load(\n                os.path.join(model_path, model_file),\n                map_location=torch.device('cpu')\n            )\n        )\n        models_tconv[\"13-22\"].append(model)","metadata":{"papermill":{"duration":2.478242,"end_time":"2023-05-20T14:51:41.123713","exception":false,"start_time":"2023-05-20T14:51:38.645471","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-24T20:48:45.227092Z","iopub.execute_input":"2023-05-24T20:48:45.227534Z","iopub.status.idle":"2023-05-24T20:48:48.407830Z","shell.execute_reply.started":"2023-05-24T20:48:45.227491Z","shell.execute_reply":"2023-05-24T20:48:48.406770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**ЗАВЕРШЕНИЕ МОДЕЛИ - ДРУГОЙ КАТБУСТ**","metadata":{"papermill":{"duration":0.00924,"end_time":"2023-05-20T14:51:41.142595","exception":false,"start_time":"2023-05-20T14:51:41.133355","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import jo_wilder\ntry:\n    jo_wilder.make_env.__called__ = False\n    env.__called__ = False\n    type(env)._state = type(type(env)._state).__dict__['INIT']\nexcept:\n    pass\n\nenv = jo_wilder.make_env()\niter_test = env.iter_test() ","metadata":{"papermill":{"duration":0.041993,"end_time":"2023-05-20T14:51:41.194334","exception":false,"start_time":"2023-05-20T14:51:41.152341","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-24T20:48:48.412524Z","iopub.execute_input":"2023-05-24T20:48:48.415638Z","iopub.status.idle":"2023-05-24T20:48:48.444163Z","shell.execute_reply.started":"2023-05-24T20:48:48.415589Z","shell.execute_reply":"2023-05-24T20:48:48.443197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nf_read = open('/kaggle/input/catboost-predict/importance_dict.pkl', 'rb')\nimportance_dict = pickle.load(f_read)\nf_read.close()","metadata":{"papermill":{"duration":0.023484,"end_time":"2023-05-20T14:51:41.227416","exception":false,"start_time":"2023-05-20T14:51:41.203932","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-24T20:48:48.447705Z","iopub.execute_input":"2023-05-24T20:48:48.448089Z","iopub.status.idle":"2023-05-24T20:48:48.458132Z","shell.execute_reply.started":"2023-05-24T20:48:48.448053Z","shell.execute_reply":"2023-05-24T20:48:48.457086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat2code_ = {}\nfor cat_feat in CAT_FEATS:\n    with open(os.path.join(cat2code_path, f\"{cat_feat[:-5]}2code.pkl\"), \"rb\") as f:\n        cat2code_[cat_feat] = pickle.load(f)\n        \net_diff_upper_bound = 3.6e6\n\ndef drop_multi_game_naive(df: pd.DataFrame, local: bool = True) -> pd.DataFrame:\n    \"\"\"Drop events not occurring at the first game play.\n    \n    Parameters:\n        df: input DataFrame\n    \n    Return:\n        df: DataFrame with events occurring at the first game play only\n    \"\"\"\n    if local:\n        df[\"lv_diff\"] = df.groupby(\"session_id\").apply(lambda x: x[\"level\"].diff().fillna(0)).values\n    else:\n        df[\"lv_diff\"] = df[\"level\"].diff().fillna(0)\n    reversed_lv_pts = df[\"lv_diff\"] < 0\n    df.loc[~reversed_lv_pts, \"lv_diff\"] = 0\n    if local:\n        df[\"multi_game_flag\"] = df.groupby(\"session_id\")[\"lv_diff\"].cumsum()\n    else:\n        df[\"multi_game_flag\"] = df[\"lv_diff\"].cumsum()\n    multi_game_mask = df[\"multi_game_flag\"] < 0\n    multi_game_rows = df[multi_game_mask].index\n    df = df.drop(multi_game_rows).reset_index(drop=True)\n    \n    return df\n\n@torch.no_grad()\ndef quick_infer(X: Tuple[Tensor, Optional[Tensor]], models: List[nn.Module]) -> Tensor:\n    \n    x, x_cat = X\n    n_models = len(models)\n    \n    for i, model in enumerate(models):\n        model.eval()\n        if i == 0:\n            y_pred = model(x, x_cat) / n_models   # (1, n_qns (out_dim))\n        else:\n            y_pred += model(x, x_cat) / n_models   # (1, n_qns (out_dim))\n    \n    y_pred = y_pred.reshape((-1, 1)).detach().numpy()\n    \n    \n    return y_pred","metadata":{"papermill":{"duration":0.041764,"end_time":"2023-05-20T14:51:41.279099","exception":false,"start_time":"2023-05-20T14:51:41.237335","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-24T20:48:48.461755Z","iopub.execute_input":"2023-05-24T20:48:48.462155Z","iopub.status.idle":"2023-05-24T20:48:48.491821Z","shell.execute_reply.started":"2023-05-24T20:48:48.462117Z","shell.execute_reply":"2023-05-24T20:48:48.490520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from joblib import dump, load\nmeta_models = {}\ndir_meta = '/kaggle/input/meta-xgb/meta_xgb/'\nfor q in range(1, 19):\n    meta_models[q] = load(dir_meta + f'meta_XGB_question{q}.pkl')","metadata":{"papermill":{"duration":0.357074,"end_time":"2023-05-20T14:51:41.645827","exception":false,"start_time":"2023-05-20T14:51:41.288753","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-24T20:48:48.493486Z","iopub.execute_input":"2023-05-24T20:48:48.494785Z","iopub.status.idle":"2023-05-24T20:48:48.805735Z","shell.execute_reply.started":"2023-05-24T20:48:48.494743Z","shell.execute_reply":"2023-05-24T20:48:48.804295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = {}\n\nlimits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nlist_q = {'0-4':[1,2,3], '5-12':[4,5,6,7,8,9,10,11,12,13], '13-22':[14,15,16,17,18]}\nlist1m = [15, 16,18] \nlist2m = [8,17]\nlist12m = [1,3,4,5,6,7,9,10,11,14] \n\nfor (test, sam_sub) in iter_test:\n    sam_sub['question'] = [int(label.split('_')[1][1:]) for label in sam_sub['session_id']]\n    sam_sub['session'] = [int(label.split('_')[0]) for label in sam_sub['session_id']]\n    grp = test.level_group.values[0]\n    \n    # CREATE TRUE LABLES FOR META MODEL\n    ALL_USERS = sam_sub.session.unique()\n    # PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\n    true = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS)\n    true.columns = [x for x in range(1,19)]\n    \n    # Initialise preds for this session_id\n    sid = sam_sub.iloc[0, :]['session_id'].split('_')[0]\n    if sid not in preds.keys():\n        preds[sid] = {}\n\n    \n    # 1 model\n    df = (pl.from_pandas(test)\n      .drop([\"fullscreen\", \"hq\", \"music\"])\n      .with_columns(columns))\n    df = feature_engineer(df, grp, use_extra=True, feature_suffix='')\n    df = time_feature(df) \n    \n    # 2 model\n    old_train = delt_time_def(test[test.level_group == grp])\n    train = feature_engineer_2(old_train)    \n\n    sam_sub['correct'] = 1\n    sam_sub.loc[sam_sub.question.isin([5, 8, 10, 13, 15]), 'correct'] = 0   \n    \n    for q in list_q[grp]:\n        if q in list12m:         \n            # 1 model\n            FEATURES = importance_dict[str(q)]\n            model = models_list[q-1][0]\n            pred = model.predict_proba(df[FEATURES].astype(np.float32).values)[0,1]         \n        \n            # 2 model\n            train_q = feature_quest(train, old_train, q)\n            clf = models[q]\n            p = clf.predict_proba(train_q.astype('float32'))[:,1]\n            \n            # mix model\n            kmix = 0.7\n            pred = pred*kmix + p*(1-kmix)\n        elif q in list1m:\n            # 1 model\n            FEATURES = importance_dict[str(q)]\n            model = models_list[q-1][0]\n            pred = model.predict_proba(df[FEATURES].astype(np.float32).values)[0,1] \n        elif q in list2m:\n            # 2 model\n            train_q = feature_quest(train, old_train, q)\n            clf = models[q]\n            pred = clf.predict_proba(train_q.astype('float32'))[:,1]\n        elif q in [2, 12]:\n            pred = 1 \n        elif q == 13:\n            pred = 0\n        else:\n            continue               \n\n        mask = sam_sub.question == q\n        x = int(pred > 0.63)\n        preds[sid][q] = x\n        sam_sub.loc[mask,'correct'] = x\n\n    # ================================================ #\n    # ADD PREDS FROM PREVIOUS LEVEL GROUPS TO PREDICT! #\n    \n    if grp == '1-4':\n        for i in range(1,4):\n            true[i] = preds[sid][i] \n    elif grp == '5-12':\n        for i in range(1,14):\n            true[i] = preds[sid][i]\n    elif grp == '13-22':\n        for i in range(1,19):\n            true[i] = preds[sid][i]\n    \n    if grp == '0-4': \n        FEATURES = [x for x in list_q['0-4'] if x != q]\n    elif grp == '5-12': \n        FEATURES = [x for x in (list_q['0-4'] + list_q['5-12']) if x != q]\n    elif grp == '13-22':\n        FEATURES = [x for x in (list_q['0-4'] + list_q['5-12'] + list_q['13-22']) if x != q]  \n    \n    a,b = limits[grp]\n    for q in range(a,b):\n        clf = meta_models[q]\n        test_meta = true[FEATURES]\n        p = clf.predict_proba(test_meta.astype('float32'))[:,1]\n        p = 0.7*preds[sid][q] + 0.3*p\n        mask = sam_sub.session_id.str.contains(f'q{q}')\n        sam_sub.loc[mask,'correct'] = int((p.item())>0.63)\n        preds[sid][q]\n            \n    sam_sub = sam_sub[['session_id', 'correct']]        \n    env.predict(sam_sub)","metadata":{"papermill":{"duration":2.812053,"end_time":"2023-05-20T14:51:44.468528","exception":false,"start_time":"2023-05-20T14:51:41.656475","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-24T20:48:48.817434Z","iopub.execute_input":"2023-05-24T20:48:48.818155Z","iopub.status.idle":"2023-05-24T20:48:52.696234Z","shell.execute_reply.started":"2023-05-24T20:48:48.818113Z","shell.execute_reply":"2023-05-24T20:48:52.695129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('submission.csv')\nprint(sub.shape, sub.correct.mean())","metadata":{"papermill":{"duration":0.017917,"end_time":"2023-05-20T14:51:44.496164","exception":false,"start_time":"2023-05-20T14:51:44.478247","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.011469,"end_time":"2023-05-20T14:51:45.144922","exception":false,"start_time":"2023-05-20T14:51:45.133453","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}