{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars as pl\nfrom tqdm import tqdm\nfrom pathlib import Path\nimport warnings,pickle\nfrom itertools import combinations\nimport math,gc,os,joblib\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import GroupKFold, KFold\nfrom catboost import CatBoostClassifier, Pool\nfrom xgboost import XGBClassifier\nimport matplotlib.pyplot as plt\nfrom datetime import date,datetime\nfrom collections import defaultdict\ntoday = str(date.today()) \nwarnings.filterwarnings('ignore')\npd.set_option(\"display.max_columns\", None)\npd.set_option(\"display.max_rows\", 200)\n\nclass CFG:\n    train=\"/kaggle/input/how-to-get-32gb-ram/train.parquet\" \n    seed=222222\n    model=\"xgb\" \n    n_folds=5\n    tune=False\n    GEN_FEAT=True\n    INFER=False\n    time_feat = False\n    MODE=\"median\"\n    version=f\"{model}_{str(date.today())+'-'+ str(datetime.now().hour)}\"  \n# Create Versioned directory \nCFG.version = 'xgb_2023-06-23-1'\nprint(f\"The version of training is [{CFG.version}]\")\noutput_dir = Path(f\"{CFG.version}\")\noutput_dir.mkdir(exist_ok=True, parents=True)\ntargets = pd.read_parquet('/kaggle/input/how-to-get-32gb-ram/train_labels.parquet')\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]))\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]))   \ntargets.shape","metadata":{"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets.shape","metadata":{"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile fe_v2.py\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom tqdm import tqdm\nimport warnings\nimport gc,joblib\nfrom itertools import combinations\n\nCATS = ['event_name', 'name', 'fqid', 'room_fqid', 'text_fqid']\nNUMS = ['page', 'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y','hover_duration', 'elapsed_time_diff']\nfqid_lists = ['worker', 'archivist', 'gramps', 'wells', 'toentry', 'confrontation', 'crane_ranger', 'groupconvo', 'flag_girl', 'tomap', 'tostacks', 'tobasement', 'archivist_glasses', 'boss', 'journals', 'seescratches', 'groupconvo_flag', 'cs', 'teddy', 'expert', 'businesscards', 'ch3start', 'tunic.historicalsociety', 'tofrontdesk', 'savedteddy', 'plaque', 'glasses', 'tunic.drycleaner', 'reader_flag', 'tunic.library', 'tracks', 'tunic.capitol_2', 'trigger_scarf', 'reader', 'directory', 'tunic.capitol_1', 'journals.pic_0.next', 'unlockdoor', 'tunic', 'what_happened', 'tunic.kohlcenter', 'tunic.humanecology', 'colorbook', 'logbook', 'businesscards.card_0.next', 'journals.hub.topics', 'logbook.page.bingo', 'journals.pic_1.next', 'journals_flag', 'reader.paper0.next', 'tracks.hub.deer', 'reader_flag.paper0.next', 'trigger_coffee', 'wellsbadge', 'journals.pic_2.next', 'tomicrofiche', 'journals_flag.pic_0.bingo', 'plaque.face.date', 'notebook', 'tocloset_dirty', 'businesscards.card_bingo.bingo', 'businesscards.card_1.next', 'tunic.wildlife', 'tunic.hub.slip', 'tocage', 'journals.pic_2.bingo', 'tocollectionflag', 'tocollection', 'chap4_finale_c', 'chap2_finale_c', 'lockeddoor', 'journals_flag.hub.topics', 'tunic.capitol_0', 'reader_flag.paper2.bingo', 'photo', 'tunic.flaghouse', 'reader.paper1.next', 'directory.closeup.archivist', 'intro', 'businesscards.card_bingo.next', 'reader.paper2.bingo', 'retirement_letter', 'remove_cup', 'journals_flag.pic_0.next', 'magnify', 'coffee', 'key', 'togrampa', 'reader_flag.paper1.next', 'janitor', 'tohallway', 'chap1_finale', 'report', 'outtolunch', 'journals_flag.hub.topics_old', 'journals_flag.pic_1.next', 'reader.paper2.next', 'chap1_finale_c', 'reader_flag.paper2.next', 'door_block_talk', 'journals_flag.pic_1.bingo', 'journals_flag.pic_2.next', 'journals_flag.pic_2.bingo', 'block_magnify', 'reader.paper0.prev', 'block', 'reader_flag.paper0.prev', 'block_0', 'door_block_clean', 'reader.paper2.prev', 'reader.paper1.prev', 'doorblock', 'tocloset', 'reader_flag.paper2.prev', 'reader_flag.paper1.prev', 'block_tomap2', 'journals_flag.pic_0_old.next', 'journals_flag.pic_1_old.next', 'block_tocollection', 'block_nelson', 'journals_flag.pic_2_old.next', 'block_tomap1', 'block_badge', 'need_glasses', 'block_badge_2', 'fox', 'block_1']\nFQID=fqid_lists\nDIALOGS = ['that', 'this', 'it', 'you', 'flag', 'can','and','is','the','to']\nname_feature = ['basic', 'undefined', 'close', 'open', 'prev', 'next']\nevent_name_feature = ['cutscene_click', 'person_click', 'navigate_click', 'observation_click', 'notification_click', 'object_click',\n       'object_hover', 'map_hover', 'map_click', 'checkpoint', 'notebook_click']\nTEXTS1 = ['undefined', 'Whatcha doing over there, Jo?', 'Just talking to Teddy.', 'I gotta run to my meeting!', 'Can I come, Gramps?', 'Sure thing, Jo. Grab your notebook and come upstairs!', 'See you later, Teddy.', \"I get to go to Gramps's meeting!\", 'Now where did I put my notebook?', '\\\\u00f0\\\\u0178\\\\u02dc\\\\u00b4', 'I love these photos of me and Teddy!', 'Found it!', 'Gramps is in trouble for losing papers?', \"This can't be right!\", 'Gramps is a great historian!', \"Hmm. Button's still not working.\", \"Let's get started. The Wisconsin Wonders exhibit opens tomorrow!\", 'Who wants to investigate the shirt artifact?', \"Not Leopold here. He's been losing papers lately.\", 'Hey!', \"It's true, they do keep going missing lately.\", 'See?', 'Besides, I already figured out the shirt.', \"It's a women's basketball jersey!\", 'That settles it.', 'Wells, finish up your report.', \"Leopold, why don't you help me set up in the Capitol?\", 'We need to talk about that missing paperwork.', 'Will do, Boss.', \"Hey Jo, let's take a look at the shirt!\", 'Your grampa is waiting for you in the collection room.', \"Why don't you go catch up with your grampa?\", 'What a fascinating artifact!', \"Wow, that's so cool, Gramps!\", 'Can I take a closer look?', \"Hmmm. Shouldn't you be doing your homework?\", \"It's already all done!\", 'Plus, my teacher said I could help you out for extra credit!', \"Well, that's good enough for me.\", 'Go ahead, take a peek at the shirt!', 'This looks like a clue!', \"I'll record this in my notebook.\", 'Find anything?', 'Yes! This old slip from 1916.', 'I knew it!', \"I'm not so sure that this is a basketball jersey.\", 'Wait, you mean Wells is wrong?!', 'Could be. But we need evidence!', \"Why don't you head to the Basketball Center and rustle up some clues?\", 'Sure!', \"I'll be at the Capitol. Let me know if you find anything!\", 'Better check back later.', \"That's it!\", \"The slip is from 1916 but the team didn't start until 1974!\", 'Our shirt is too old to be a basketball jersey!', 'I need to get to the Capitol and tell Gramps!', 'I should see what Grampa is up to!', 'Ugh. Meetings are so boring.', 'Grab your notebook and come upstairs!', 'Hang tight, Teddy.', \"I'll hurry back and then we can go exploring!\", 'Well, Leopold here is always losing papers...', 'Ha. Told you so!', 'Can we hurry up, Gramps?', 'Teddy and I were gonna go climb that huge tree out back!', \"Hmmm. Don't forget about your homework.\", 'Your teacher said you missed 7 assignments in a row!', 'So? History is boring!', 'I suppose historians are boring, too?', \"No way, Gramps. You're the best!\", 'Then do it for me!', 'Your teacher said you could help me for extra credit.', 'A boring old shirt.', 'Just this old slip from 1916.', 'Do I have to?', 'Um... what did you want me to do again?', 'Head over to the Basketball Center.', 'Hopefully you can find some clues!', 'Meetings are BORING!', \"I feel like I'm forgetting something.\", 'Gramps is the best historian ever!', 'This button never works!', \"Why don't you go play with your grampa?\", \"Look at that! It's the bee's knees!\", \"Well, I did SOME of those. I just couldn't find them!\", 'Did you do all of them?', 'No... because history is boring!', 'Hooray, a boring old shirt.', 'Hot Dog! I knew it!', 'Ooh, I like clues!', 'Hopefully you can rustle up some clues!', 'I should go talk to Gramps!', 'Yes! This cool old slip from 1916.', 'I should see what Gramps is up to!', 'Gramps said to look for clues. Better look around.', 'Have a look at the artifact!', 'Come on, Jo!', \"Meet me back in my office and we'll get started!\"] \nTEXTS2 = ['undefined', \"Hmm. Button's still not working.\", 'Better check back later.', 'What are you still doing here,  Jolie?', 'Go find your grampa and get to work!', 'Oh no!', 'What happened here?!', \"I don't know!\", 'I got here and the whole place was a mess!', 'Can you help me tidy up?', \"Teddy's scarf! Somebody must've taken him!\", 'Try not to panic, Jo.', 'Maybe he just got scared and ran off.', 'But he never goes anywhere without his scarf!', \"I think he's in trouble!\", 'Is this your coffee, Gramps?', \"Nope, that's from Bean Town. I only drink Holdgers!\", \"Who could've done this?\", \"It must've been Wells.\", \"He's always trying to get you in trouble, and he doesn't like animals!\", 'Slow down, Jo.', 'But what if Wells kidnapped Teddy?', 'Then we need evidence.', \"You're right, Gramps. Let's investigate!\", \"I'm afraid my papers have gone missing in this mess.\", \"You'll have to get started without me.\", \"Okay. I'll find Teddy!\", \"And I'll figure out the shirt, too.\", 'I knew I could count on you, Jo!', \"Why don't you go upstairs and see the archivist?\", \"He's our expert record keeper.\", 'I need your help!', 'Who are you?', \"I'm Leopold's grandkid!\", \"Sorry, I'm too busy for kids right now.\", 'Now if only I could read this thing.', \"Can't believe I lost my reading glasses.\", 'I bet the archivist could use this!', \"Ah, that's better!\", 'Did you have a question?', 'Yes! I was wondering-', 'Wait a minute!', 'Where did you get that coffee?', \"Oh, that's from Bean Town.\", 'I ran into Wells there this morning.', 'Wells? I knew it!', 'Do you know anything about this slip?', 'I found it on an old shirt.', 'An old shirt? Try the university.', 'You can talk to a textile expert there.', \"What's a textile expert?\", 'They study clothes and fabric.', 'Great! Thanks for the help!', 'Head over to the university.', 'Hello there!', 'Wow! What is all this stuff?', \"It's our Norwegian Craft exhibit!\", 'Can I give you the tour?', \"Sorry, I'm in a hurry.\", 'Do you know what this slip is?', 'Looks like a dry cleaning receipt.', 'Thanks.', 'Now I Just need to find all the cleaners from way back in 1916.', 'Maybe I can help!', \"I've got a stack of business cards from my favorite cleaners.\", \"Why don't you take a look?\", 'This place was around in 1916! I can start there!', \"You haven't seen any badgers around here, have you?\", 'Badgers? No.', 'Okay. Thanks anyway.', 'Hi! How can I help you?', 'I need to find the owner of this slip.', \"Well, I can't show our log books to just anybody.\", 'Please?', \"It's for Grampa Leo. He's a historian!\", 'Leo... you mean Leopold?', 'Your gramps is awesome! Always full of stories.', \"Guess it couldn't hurt to let you take a look.\", \"Here's the log book.\", \"It's a match!\", 'Theodora Youmans must be the owner!', 'Do you know who Theodora Youmans is?', \"Hmmm... not sure. Why don't you try the library?\", 'Thanks for the help!', 'Oh, hello there!', 'How can I help you?', 'Have you seen a badger around here?', \"I'm afraid not.\", 'Please let me know if you do.', \"I'm also looking for Theodora Youmans. Have you heard of her?\", 'Theodora Youmans? Of course!', \"Check out our microfiche. It's right through that door.\", 'Youmans was a suffragist!', 'She helped get votes for women!', 'Wells! What was he doing here? I should ask the librarian.', 'What was Wells doing here?', 'He was looking for a taxidermist.', \"What's a taxidermist?\", 'Not sure. Here, let me look it up.', '\\\\Taxidermy: the art of preparing, stuffing, and mounting the skins of animals.\\\\', 'Oh no... Teddy!', 'Can you help me find Wells?', 'You could ask the archivist. He knows everybody!', \"Jolie! I was hoping you'd stop by. Any news on the shirt artifact?\", \"I haven't quite figured it out just yet...\", \"Well, get on it. I'm counting on you and your gramps to figure this out!\", 'Can you help me? I need to find Wells!', \"I haven't seen him.\", 'Please? This is really important.', \"Sorry, can't help you.\", 'Do you have any info on Theodora Youmans?', 'Theodora Youmans? Is that who owned the shirt?', 'Yep.', \"Why didn't you say so?\", 'Youmans was a suffragist here in Wisconsin.', 'She led marches and helped women get the right to vote!', \"Wait a sec. Women couldn't vote?!\", 'Nope. But Youmans and other suffragists worked hard to change that.', 'Thanks to them, Wisconsin was the first state to approve votes for women!', 'Wow!', \"Here's a call number to find more info in the Stacks.\", 'Where are the Stacks?', 'Right outside the door.', 'Hey, this is Youmans!', \"And look! She's wearing the shirt!\", 'I should go to the Capitol and tell everyone!', 'Ugh. Fine.', 'What should I do first?', 'Head upstairs and talk to the archivist. He might be able to help!', 'Nice seeing you, Jolie!', \"It's such a nice fall day.\", 'I love these photos of me and Teddy.', 'What the-', 'I have an idea.', \"He's wrong about old shirts and his name rhymes with \\\\smells\\\\...\", 'BUT WELLS STOLE TEDDY!', 'Could be. But we need evidence.', \"Fine. Let's investigate!\", \"Don't worry, Gramps. I'll find Teddy!\", \"Please let me know if you do. It's important!\", 'I need to find Wells right away! Do you know where he is?', 'I need to find Wells!!!', \"I can't calm down. This is important!\", 'I should stay and look for clues!', 'Where should I go again?', 'You could try the archivist. Maybe he can help you find Wells!', 'Hi, Mrs. M.', \"I don't need that right now.\", 'This button never works!', 'I got here and the whole place was ransacked!', 'Hold your horses, Jo.', '*grumble grumble*', 'And you are?', \"I don't have time for kids.\", 'Now if only I could read this thing. Blasted tiny letters...', 'Knew what?', 'Did you have a question or not?', 'Yes!', \"You're still here? I'm trying to work!\", 'Run along to the university.', 'Ooh, thanks!', 'Now I just need to find all the cleaners from wayyyy back in 1916.', 'Yikes... this could take a while.', 'Hi! *cough*', 'Can you help-', '*cough cough*', 'Can you help me-', '*COUGH COUGH COUGH*', 'Um, are you okay?', \"Oh, I'm fine! Just a little hoarse.\", 'Ha! What do you call a pony with a sore throat?', 'Huh?', 'A little horse!', \"Ha! You're funny.\", 'I got that one from my Gramps!', 'Can you help me? I need to find the owner of this slip.', \"Yup, that's him!\", \"Unless you're too busy horsing around.\", 'Ha! Good one.', \"You look like you're on a mission.\", 'Two missions, actually!', 'Oh my!', 'I need to find Wells right away!! Do you know where he is?', \"Calm down, kid. I haven't seen him.\", 'I should help Gramps clean.', \"Maybe there's a clue in this mess!\", \"Poor Gramps! I should make sure he's okay.\", 'The archivist said I should look in the stacks.', 'Yeah. Thanks anyway.', 'What are you waiting for? The Stacks are right outside the door.', 'Are you okay?', \"Weren't you going to check out our microfiche?\", \"I'm sure you'll find Theodora in there somewhere!\", \"But I hear the museum's got one on the loose!\", 'Well? What are you still doing here?', 'So much cleaning to do...', 'I used to have a magnifying glass around here\\\\u00e2\\\\u20ac\\\\u00a6', \"Did you drop something, Dear? There's a card on the floor.\", 'Take a look!', 'I found it!', 'Theodora wearing the shirt!', 'You better get to the capitol!', 'Nice decorations.', 'Did you drop something, Dear?', 'I should find out if she can help me!', 'Ooh, nice decorations!', 'The libarian said I could find some information on Youmans in here...', 'I should ask the librarian why Wells was here.', \"I wonder if there's a clue in those business cards...\", 'Thanks. Did you figure out the shirt?', 'Welcome back, Jolie. Did you figure out the shirt?', 'I should check that logbook to see who owned this slip...', 'AND I know who took Teddy!', 'Who is Teddy?', \"And where's your grampa?\", 'Sorry for the delay, Boss.', 'I had some cleaning up to do in my office.', 'Mrs. M, I think Wells kidnapped Teddy.', \"And he messed up Gramps's office, too!\", 'One step at a time, Jo.', 'Did you figure out the shirt?', 'I knew you could do it, Jo!', 'Now can I tell you what happened to Teddy?', 'He needs our help!', \"Sorry I'm late.\", \"Wells! Where's Teddy? Is he okay?\", 'I figured out that you kidnapped him!', 'Easy, Jo.', \"Why don't you prove your case?\", \"It'll be okay, Jo. We'll find Teddy!\", 'Nice work on the shirt, Jolie!', 'Leopold, can you run back to the museum?', 'Sounds good, Boss.', 'Jo, meet me back at my office.', 'I hope you find your badger, kid.', 'Here I am!', 'Wells sabotaged Gramps!', 'AND he stole Teddy!'] \nTEXTS3 = ['undefined', 'Hey!', 'Your grampa is waiting for you in the collection room.', 'Better check back later.', \"You haven't seen any badgers around here, have you?\", 'Badgers? No.', 'Okay. Thanks anyway.', \"I haven't quite figured it out just yet...\", 'Hey, this is Youmans!', \"And look! She's wearing the shirt!\", 'I should go to the Capitol and tell everyone!', 'Jo!', 'Check out the next artifact!', 'What is it?', \"I think it's a flag! Pretty interesting, huh?\", \"It's really cool, Gramps. But I'm worried about Teddy.\", \"He's still missing!\", \"We'll find him, Jo.\", 'Want to look for more clues?', \"We'll find Teddy.\", 'We just have to keep our eyes open!', 'Hey, look at those scratches!', 'The kidnapper probably took Teddy on the elevator!', \"You're right, Jo!\", \"Why isn't the button working?\", \"We'll need a key card.\", 'I had one, but Teddy chewed it up.', \"I've got Wells's ID!\", 'What should we do next?', \"I need to take the artifact upstairs. Why don't you investigate those scratch marks?\", \"Okay. I'll try.\", 'Teddy, here I come!', 'I wonder whose glasses these are.', 'Teddy!!!', \"Hang on. I'll get you out of there!\", 'Whoever lost these glasses probably took Teddy!', 'How can I find out whose glasses these are?', 'Oh! There was a staff directory in the entryway!', \"I'll go look at everyone's pictures!\", 'Those are the same glasses!', \"The archivist must've taken Teddy!\", \"Yes! It's the key for Teddy's cage!\", 'I found the key!', \"Come on, let's get out of here!\", \"Here's your scarf back!\", 'What are you doing down here?', 'And how did that badger get free?', \"I'm here to rescue my friend!\", \"What's going on here?\", 'Thanks for coming, Boss.', 'I told you!', 'I captured a badger in our museum!', \"He's been eating my lunch every day this week!\", 'He has??', \"I've seen him eating homework and important papers, too.\", \"Jolie- keep your badger under control, or he'll have to go.\", 'And you, Frank-', \"You can't just steal Jolie's pet.\", 'Ugh. Fine.', 'Alright, Jolie. Back to work.', \"Come on, Teddy. Let's go help Gramps!\", \"Let's go help Gramps!\", 'Gramps must be up in the collection room.', \"Let's go find him!\", \"Teddy! I'm glad to see you.\", 'The archivist had him locked up!', 'Poor badger.', \"You're becoming quite the detective, Jo.\", 'Notice any clues about this flag?', 'Well... it looks hand-stitched.', 'Good catch!', 'Go on, tell the boss what you found!', \"I'm telling you, Boss. Taxidermy is the way to go!\", 'Nonsense. I want live animals at the exhibit, not stuffed ones.', \"Ah, Jolie! I'm glad you're here.\", \"I'm putting you in charge of the flag case.\", 'Make sure to get some old photos for the exhibit, like last time!', \"Wait! Can't I do it?\", 'The symbol on the flag looks sort of like a deer hoof.', 'It could be an early design for the Wisconsin state flag!', 'Wells, you already have a job to do.', 'What now, kid?', 'Do you really think that symbol is a deer hoof?', 'Not sure.', 'Do you know where I can find a deer expert?', 'Hmm. You could try the Aldo Leopold Wildlife Center.', 'I have to head over there and check out the animals.', \"I'll ride with you!\", \"Come on, kid. Let's go.\", 'Head over to the Wildlife Center!', \"I'm sure they'll be able to help.\", 'People sure drink a lot of coffee around here.', \"I can't believe this.\", 'Ugh...', 'Oh no! What happened to that crane?', 'Her beak is stuck in a coffee cup.', \"It's lucky we found her.\", 'Ugh! Those cups are all over the place.', \"I need to get her free. She won't hold still!\", 'Can Teddy and I help?', 'Sure! Give it a try.', 'Careful. That beak is sharp!', 'We need to calm her down, Teddy.', 'Any ideas?', '\\\\u00f0\\\\u0178\\\\u00a6\\\\u2014', 'Oh yeah, cranes eat insects!', 'Luckily there are tons of insects around here...', 'Got one!', \"Maybe she'll let me take off the cup!\", \"It's OK, girl! Look, I found you a cricket!\", 'You did it! Thanks, kid.', 'Can I help you with anything?', \"I'm investigating this symbol.\", 'Does it look like a deer hoof?', \"There's a diagram of animal tracks over there.\", 'Go take a look!', \"That hoofprint doesn't match the flag!\", 'Thanks for your help, kid!', \"So? What'd you find out?\", \"Looks like it's not a deer hoof.\", \"Oh no. If I don't impress the boss soon,  I'm gonna get fired!\", 'Hey, Wells...', 'I think I might be able to help you.', \"No thanks. I don't need help from kids.\", 'Are you sure? I know where you can find a real, live badger for the exhibit!', 'Wait! What?! Really?', 'Wells, meet Teddy.', '\\\\u00f0\\\\u0178\\\\u02dc\\\\u0160', \"He says he'd be willing to help out.\", 'Yes!!!', 'We still need to figure out that flag. Do you know anyone who could help?', \"Hmm. Let's see...\", 'Actually, I went to school with somebody who LOVES old flags.', \"Why don't you go talk to her? I'll let her know you're coming.\", 'Hey, nice dog! What breed is he?', \"Actually, he's a badger.\", \"Oh, cool! I've never seen a badger in real life.\", \"You've got a million flags here!\", \"Yep. I'm a vexillophile!\", \"What's a vexillophile? \", 'It just means flag expert. How can I help?', \"I'm investigating this flag.\", 'Can you take a look?', \"Hey, I've seen that symbol before! Check it out!\", '\\\\Ecology flag, by Ron Cobb.\\\\', \"It's an ecology flag!\", 'Do you know what this flag was used for?', \"I'm not sure.\", \"If I were you, I'd go to the library and do some digging.\", 'Good idea. Thanks!', 'Welcome back, Dear! How can I help you?', 'I need to learn more about this flag!', 'It has something to do with ecology.', 'Hmm... those stripes remind me of the American flag.', 'Your flag must have been part of a national movement!', \"Go check the microfiche. Maybe you'll find something!\", \"Hey! That's Governor Nelson in front of our flag!\", 'I found the flag! Governor Nelson used it on the first Earth Day!', 'Wow! You figured it out!', 'Now I just need some old photos, like last time.', 'The boss is gonna love it!', 'You could try the archives.', 'Though the archivist might be too busy to help...', 'Okay. Thanks!', 'What are you doing here?', \"We're looking for some photos.\", \"It's for the flag display!\", 'Wait a minute...', \"YOU'RE the new history detective everybody's talking about?\", \"Teddy's helping too.\", 'What kind of photos do you need?', 'Something to do with ecology and Wisconsin.', \"Here's a call number for the Stacks. Go find some photos.\", 'Look at all those activists!', 'This is perfect for the exhibit.', 'I should go to the Capitol and tell Mrs. M!', \"It's locked!\", \"Jolie! I was hoping you'd stop by. Any news on the flag artifact?\", \"Well, get on it. I'm counting on you to figure this out!\", 'Nice seeing you, Jolie!', \"It's such a nice fall day.\", 'I love these photos of me and Teddy.', \"Why don't you go talk to the boss?\", \"She's right outside.\", 'My friend is a flag expert.', 'She should be able to help you out.', 'There are some old newspapers loaded up in the microfiche.', 'The Stacks are right outside the door. Go find some photos!', \"I don't have time for this, Gramps.\", 'Teddy is still missing!', \"Let's follow those scratch marks!\", \"I can't go with you. I need to take the artifact upstairs.\", \"It's okay, Gramps. I'll go by myself.\", 'You stole Teddy! How could you?!', \"No he hasn't!\", \"Yes, he has. I've seen him eating homework and important papers, too.\", 'Come on, Teddy.', \"Let's go find Gramps!\", 'I think I can help with your animal problem.', \"Ha! I don't need your help.\", \"Fine. Then I guess you don't want a real, live badger for the exhibit.\", \"Oh, trust me. He'll make time.\", 'Head back to the museum. Your gramps is waiting for you.', 'Huh?', \"I think it's a flag! Pretty spiffy, eh?\", \"Great Scott, you're right!\", \"Jo! I can't go with you. I need to take the artifact upstairs.\", '\\\\u00f0\\\\u0178\\\\u02dc\\\\u00ad', '\\\\u00e2\\\\u009d\\\\u00a4\\\\u00ef\\\\u00b8\\\\u008f', 'GRRRRRRR', 'GAH! And what is THAT doing out of its cage?!', '\\\\u00f0\\\\u0178\\\\u02dc\\\\u0090', 'Teddy! Did you really eat his lunch?', \"Did you steal Gramps's paperwork too?!\", 'And my homework?!?!', 'See?!', \"That thing's a monster!\", \"I don't have time for this.\", 'YEAH!', 'Wait- me?', \"You can't just steal Jolie's pet. Don't you know badgers are protected animals?\", 'Besides, he looks friendly to me.', 'Wha?!', '\\\\u00f0\\\\u0178\\\\u02dc\\\\u009d', \"Teddy! I'm sure glad to see you.\", 'Gadzooks! Poor critter.', 'Aha! Good catch, Jo.', 'Not sure. Do I look like a deer expert to you?', 'Ugh. I have to head over there and check out the animals.', 'FINE. That possum better not scratch my leather seats...', \"He's a badger!\", '\\\\u00f0\\\\u0178\\\\u00a7\\\\u02dc', 'Yoga does sound nice.', \"But cranes can't do yoga, Teddy!\", '\\\\u00f0\\\\u0178\\\\u008d\\\\u00a9', \"Cranes don't eat donuts!\", 'Besides, you just ate my last snack.', \"Gah. I can't believe this.\", \"I'm a historian, not a zookeeper!\", 'And this place is dirty, and itchy, and-', 'I love it!', \"Of course you do. You've got a rodent following you around.\", \"Actually, badgers aren't rodents-\", 'Whatever.', 'Great. Just great. Could this day get any worse?!', \"Yes!!! I'm saved!\", 'A real, live ferret!', \"He's. A. Badger.\", 'And we still need to figure out that flag!', \"Fine, fine. Let's see...\", 'A vexy-wha?', 'Ooh... \\\\Ecology flag, by Ron Cobb.\\\\', 'The boss is gonna love it!!!', \"You again! Don't let him hurt me!\", '\\\\u00f0\\\\u0178\\\\u2122\\\\u201e', \"Actually, we're just here for some photos.\", 'Guess so!', 'YOU?!', \"Just please, don't let your badger eat them!\", 'There should be some info about that symbol in my book.', 'Yeah. Thanks anyway.', \"I'll be in the collection room. Come find me when you're ready to check out the artifact.\", 'Good luck!', 'What?!', 'Can I ride with you?', \"Don't worry, he won't! (And he's a badger, by the way.)\", 'Ugh... I think that lynx is looking at me funny.', \"Don't worry, Teddy won't eat your lunch anymore!\", \"We're just looking for photos for the flag display.\", \"But I hear the museum's got one on the loose!\", 'I should check out that pair of glasses.', 'I should ask the librarian where to go next.', \"Check out the archives. They've got tons of old photos!\", \"Come on, kid. You're slowing me down.\", 'What is it, Teddy?', 'Oh no... they got sick from polluted water?', 'Poor foxes!', 'Jolie! Where have you been?', 'The exhibit opens tomorrow.', 'Wells got in trouble for littering at the Wildlife Center.', 'Thanks!', \"Are you going home now? Tomorrow's the big day!\", 'He got a park named after him? Cool!']\nTEXTS_grp = { '0-4':TEXTS1, '5-12':TEXTS2, '13-22':TEXTS3 }\ntext_lists = ['tunic.historicalsociety.cage.confrontation', 'tunic.wildlife.center.crane_ranger.crane', 'tunic.historicalsociety.frontdesk.archivist.newspaper', 'tunic.historicalsociety.entry.groupconvo', 'tunic.wildlife.center.wells.nodeer', 'tunic.historicalsociety.frontdesk.archivist.have_glass', 'tunic.drycleaner.frontdesk.worker.hub', 'tunic.historicalsociety.closet_dirty.gramps.news', 'tunic.humanecology.frontdesk.worker.intro', 'tunic.historicalsociety.frontdesk.archivist_glasses.confrontation', 'tunic.historicalsociety.basement.seescratches', 'tunic.historicalsociety.collection.cs', 'tunic.flaghouse.entry.flag_girl.hello', 'tunic.historicalsociety.collection.gramps.found', 'tunic.historicalsociety.basement.ch3start', 'tunic.historicalsociety.entry.groupconvo_flag', 'tunic.library.frontdesk.worker.hello', 'tunic.library.frontdesk.worker.wells', 'tunic.historicalsociety.collection_flag.gramps.flag', 'tunic.historicalsociety.basement.savedteddy', 'tunic.library.frontdesk.worker.nelson', 'tunic.wildlife.center.expert.removed_cup', 'tunic.library.frontdesk.worker.flag', 'tunic.historicalsociety.frontdesk.archivist.hello', 'tunic.historicalsociety.closet.gramps.intro_0_cs_0', 'tunic.historicalsociety.entry.boss.flag', 'tunic.flaghouse.entry.flag_girl.symbol', 'tunic.historicalsociety.closet_dirty.trigger_scarf', 'tunic.drycleaner.frontdesk.worker.done', 'tunic.historicalsociety.closet_dirty.what_happened', 'tunic.wildlife.center.wells.animals', 'tunic.historicalsociety.closet.teddy.intro_0_cs_0', 'tunic.historicalsociety.cage.glasses.afterteddy', 'tunic.historicalsociety.cage.teddy.trapped', 'tunic.historicalsociety.cage.unlockdoor', 'tunic.historicalsociety.stacks.journals.pic_2.bingo', 'tunic.historicalsociety.entry.wells.flag', 'tunic.humanecology.frontdesk.worker.badger', 'tunic.historicalsociety.stacks.journals_flag.pic_0.bingo', 'tunic.historicalsociety.closet.intro', 'tunic.historicalsociety.closet.retirement_letter.hub', 'tunic.historicalsociety.entry.directory.closeup.archivist', 'tunic.historicalsociety.collection.tunic.slip', 'tunic.kohlcenter.halloffame.plaque.face.date', 'tunic.historicalsociety.closet_dirty.trigger_coffee', 'tunic.drycleaner.frontdesk.logbook.page.bingo', 'tunic.library.microfiche.reader.paper2.bingo', 'tunic.kohlcenter.halloffame.togrampa', 'tunic.capitol_2.hall.boss.haveyougotit', 'tunic.wildlife.center.wells.nodeer_recap', 'tunic.historicalsociety.cage.glasses.beforeteddy', 'tunic.historicalsociety.closet_dirty.gramps.helpclean', 'tunic.wildlife.center.expert.recap', 'tunic.historicalsociety.frontdesk.archivist.have_glass_recap', 'tunic.historicalsociety.stacks.journals_flag.pic_1.bingo', 'tunic.historicalsociety.cage.lockeddoor', 'tunic.historicalsociety.stacks.journals_flag.pic_2.bingo', 'tunic.historicalsociety.collection.gramps.lost', 'tunic.historicalsociety.closet.notebook', 'tunic.historicalsociety.frontdesk.magnify', 'tunic.humanecology.frontdesk.businesscards.card_bingo.bingo', 'tunic.wildlife.center.remove_cup', 'tunic.library.frontdesk.wellsbadge.hub', 'tunic.wildlife.center.tracks.hub.deer', 'tunic.historicalsociety.frontdesk.key', 'tunic.library.microfiche.reader_flag.paper2.bingo', 'tunic.flaghouse.entry.colorbook', 'tunic.wildlife.center.coffee', 'tunic.capitol_1.hall.boss.haveyougotit', 'tunic.historicalsociety.basement.janitor', 'tunic.historicalsociety.collection_flag.gramps.recap', 'tunic.wildlife.center.wells.animals2', 'tunic.flaghouse.entry.flag_girl.symbol_recap', 'tunic.historicalsociety.closet_dirty.photo', 'tunic.historicalsociety.stacks.outtolunch', 'tunic.library.frontdesk.worker.wells_recap', 'tunic.historicalsociety.frontdesk.archivist_glasses.confrontation_recap', 'tunic.capitol_0.hall.boss.talktogramps', 'tunic.historicalsociety.closet.photo', 'tunic.historicalsociety.collection.tunic', 'tunic.historicalsociety.closet.teddy.intro_0_cs_5', 'tunic.historicalsociety.closet_dirty.gramps.archivist', 'tunic.historicalsociety.closet_dirty.door_block_talk', 'tunic.historicalsociety.entry.boss.flag_recap', 'tunic.historicalsociety.frontdesk.archivist.need_glass_0', 'tunic.historicalsociety.entry.wells.talktogramps', 'tunic.historicalsociety.frontdesk.block_magnify', 'tunic.historicalsociety.frontdesk.archivist.foundtheodora', 'tunic.historicalsociety.closet_dirty.gramps.nothing', 'tunic.historicalsociety.closet_dirty.door_block_clean', 'tunic.capitol_1.hall.boss.writeitup', 'tunic.library.frontdesk.worker.nelson_recap', 'tunic.library.frontdesk.worker.hello_short', 'tunic.historicalsociety.stacks.block', 'tunic.historicalsociety.frontdesk.archivist.need_glass_1', 'tunic.historicalsociety.entry.boss.talktogramps', 'tunic.historicalsociety.frontdesk.archivist.newspaper_recap', 'tunic.historicalsociety.entry.wells.flag_recap', 'tunic.drycleaner.frontdesk.worker.done2', 'tunic.library.frontdesk.worker.flag_recap', 'tunic.humanecology.frontdesk.block_0', 'tunic.library.frontdesk.worker.preflag', 'tunic.historicalsociety.basement.gramps.seeyalater', 'tunic.flaghouse.entry.flag_girl.hello_recap', 'tunic.historicalsociety.closet.doorblock', 'tunic.drycleaner.frontdesk.worker.takealook', 'tunic.historicalsociety.basement.gramps.whatdo', 'tunic.library.frontdesk.worker.droppedbadge', 'tunic.historicalsociety.entry.block_tomap2', 'tunic.library.frontdesk.block_nelson', 'tunic.library.microfiche.block_0', 'tunic.historicalsociety.entry.block_tocollection', 'tunic.historicalsociety.entry.block_tomap1', 'tunic.historicalsociety.collection.gramps.look_0', 'tunic.library.frontdesk.block_badge', 'tunic.historicalsociety.cage.need_glasses', 'tunic.library.frontdesk.block_badge_2', 'tunic.kohlcenter.halloffame.block_0', 'tunic.capitol_0.hall.chap1_finale_c', 'tunic.capitol_1.hall.chap2_finale_c', 'tunic.capitol_2.hall.chap4_finale_c', 'tunic.wildlife.center.fox.concern', 'tunic.drycleaner.frontdesk.block_0', 'tunic.historicalsociety.entry.gramps.hub', 'tunic.humanecology.frontdesk.block_1', 'tunic.drycleaner.frontdesk.block_1']\nroom_lists = ['tunic.historicalsociety.entry', 'tunic.wildlife.center', 'tunic.historicalsociety.cage', 'tunic.library.frontdesk', 'tunic.historicalsociety.frontdesk', 'tunic.historicalsociety.stacks', 'tunic.historicalsociety.closet_dirty', 'tunic.humanecology.frontdesk', 'tunic.historicalsociety.basement', 'tunic.kohlcenter.halloffame', 'tunic.library.microfiche', 'tunic.drycleaner.frontdesk', 'tunic.historicalsociety.collection', 'tunic.historicalsociety.closet', 'tunic.flaghouse.entry', 'tunic.historicalsociety.collection_flag', 'tunic.capitol_1.hall', 'tunic.capitol_0.hall', 'tunic.capitol_2.hall']\ncluster_dict={'cutscene_click': {'0-4': 7, '5-12': 7, '13-22': 6},\n 'person_click': {'0-4': 7, '5-12': 7, '13-22': 6},\n 'navigate_click': {'0-4': 7, '5-12': 7, '13-22': 6},\n 'observation_click': {'0-4': 7, '5-12': 7, '13-22': 6},\n 'object_click': {'0-4': 7, '5-12': 7, '13-22': 8},}\ntext_lists1 = ['tunic.historicalsociety.entry.groupconvo', 'tunic.historicalsociety.collection.cs', 'tunic.historicalsociety.collection.gramps.found', 'tunic.historicalsociety.closet.gramps.intro_0_cs_0', 'tunic.historicalsociety.closet.teddy.intro_0_cs_0', 'tunic.historicalsociety.closet.intro', 'tunic.historicalsociety.closet.retirement_letter.hub', 'tunic.historicalsociety.collection.tunic.slip', 'tunic.kohlcenter.halloffame.plaque.face.date', 'tunic.kohlcenter.halloffame.togrampa', 'tunic.historicalsociety.collection.gramps.lost', 'tunic.historicalsociety.closet.notebook', 'tunic.historicalsociety.basement.janitor', 'tunic.historicalsociety.stacks.outtolunch', 'tunic.historicalsociety.closet.photo', 'tunic.historicalsociety.collection.tunic', 'tunic.historicalsociety.closet.teddy.intro_0_cs_5', 'tunic.historicalsociety.entry.wells.talktogramps', 'tunic.historicalsociety.entry.boss.talktogramps', 'tunic.historicalsociety.closet.doorblock', 'tunic.historicalsociety.entry.block_tomap2', 'tunic.historicalsociety.entry.block_tocollection', 'tunic.historicalsociety.entry.block_tomap1', 'tunic.historicalsociety.collection.gramps.look_0', 'tunic.kohlcenter.halloffame.block_0', 'tunic.capitol_0.hall.chap1_finale_c', 'tunic.historicalsociety.entry.gramps.hub']\ntext_lists2 = ['tunic.historicalsociety.frontdesk.archivist.newspaper', 'tunic.historicalsociety.frontdesk.archivist.have_glass', 'tunic.drycleaner.frontdesk.worker.hub', 'tunic.historicalsociety.closet_dirty.gramps.news', 'tunic.humanecology.frontdesk.worker.intro', 'tunic.library.frontdesk.worker.hello', 'tunic.library.frontdesk.worker.wells', 'tunic.historicalsociety.frontdesk.archivist.hello', 'tunic.historicalsociety.closet_dirty.trigger_scarf', 'tunic.drycleaner.frontdesk.worker.done', 'tunic.historicalsociety.closet_dirty.what_happened', 'tunic.historicalsociety.stacks.journals.pic_2.bingo', 'tunic.humanecology.frontdesk.worker.badger', 'tunic.historicalsociety.closet_dirty.trigger_coffee', 'tunic.drycleaner.frontdesk.logbook.page.bingo', 'tunic.library.microfiche.reader.paper2.bingo', 'tunic.historicalsociety.closet_dirty.gramps.helpclean', 'tunic.historicalsociety.frontdesk.archivist.have_glass_recap', 'tunic.historicalsociety.frontdesk.magnify', 'tunic.humanecology.frontdesk.businesscards.card_bingo.bingo', 'tunic.library.frontdesk.wellsbadge.hub', 'tunic.capitol_1.hall.boss.haveyougotit', 'tunic.historicalsociety.basement.janitor', 'tunic.historicalsociety.closet_dirty.photo', 'tunic.historicalsociety.stacks.outtolunch', 'tunic.library.frontdesk.worker.wells_recap', 'tunic.capitol_0.hall.boss.talktogramps', 'tunic.historicalsociety.closet_dirty.gramps.archivist', 'tunic.historicalsociety.closet_dirty.door_block_talk', 'tunic.historicalsociety.frontdesk.archivist.need_glass_0', 'tunic.historicalsociety.frontdesk.block_magnify', 'tunic.historicalsociety.frontdesk.archivist.foundtheodora', 'tunic.historicalsociety.closet_dirty.gramps.nothing', 'tunic.historicalsociety.closet_dirty.door_block_clean', 'tunic.library.frontdesk.worker.hello_short', 'tunic.historicalsociety.stacks.block', 'tunic.historicalsociety.frontdesk.archivist.need_glass_1', 'tunic.historicalsociety.frontdesk.archivist.newspaper_recap', 'tunic.drycleaner.frontdesk.worker.done2', 'tunic.humanecology.frontdesk.block_0', 'tunic.library.frontdesk.worker.preflag', 'tunic.drycleaner.frontdesk.worker.takealook', 'tunic.library.frontdesk.worker.droppedbadge', 'tunic.library.microfiche.block_0', 'tunic.library.frontdesk.block_badge', 'tunic.library.frontdesk.block_badge_2', 'tunic.capitol_1.hall.chap2_finale_c', 'tunic.drycleaner.frontdesk.block_0', 'tunic.humanecology.frontdesk.block_1', 'tunic.drycleaner.frontdesk.block_1']\ntext_lists3 = ['tunic.historicalsociety.cage.confrontation', 'tunic.wildlife.center.crane_ranger.crane', 'tunic.wildlife.center.wells.nodeer', 'tunic.historicalsociety.frontdesk.archivist_glasses.confrontation', 'tunic.historicalsociety.basement.seescratches', 'tunic.flaghouse.entry.flag_girl.hello', 'tunic.historicalsociety.basement.ch3start', 'tunic.historicalsociety.entry.groupconvo_flag', 'tunic.historicalsociety.collection_flag.gramps.flag', 'tunic.historicalsociety.basement.savedteddy', 'tunic.library.frontdesk.worker.nelson', 'tunic.wildlife.center.expert.removed_cup', 'tunic.library.frontdesk.worker.flag', 'tunic.historicalsociety.entry.boss.flag', 'tunic.flaghouse.entry.flag_girl.symbol', 'tunic.wildlife.center.wells.animals', 'tunic.historicalsociety.cage.glasses.afterteddy', 'tunic.historicalsociety.cage.teddy.trapped', 'tunic.historicalsociety.cage.unlockdoor', 'tunic.historicalsociety.stacks.journals.pic_2.bingo', 'tunic.historicalsociety.entry.wells.flag', 'tunic.humanecology.frontdesk.worker.badger', 'tunic.historicalsociety.stacks.journals_flag.pic_0.bingo', 'tunic.historicalsociety.entry.directory.closeup.archivist', 'tunic.capitol_2.hall.boss.haveyougotit', 'tunic.wildlife.center.wells.nodeer_recap', 'tunic.historicalsociety.cage.glasses.beforeteddy', 'tunic.wildlife.center.expert.recap', 'tunic.historicalsociety.stacks.journals_flag.pic_1.bingo', 'tunic.historicalsociety.cage.lockeddoor', 'tunic.historicalsociety.stacks.journals_flag.pic_2.bingo', 'tunic.wildlife.center.remove_cup', 'tunic.wildlife.center.tracks.hub.deer', 'tunic.historicalsociety.frontdesk.key', 'tunic.library.microfiche.reader_flag.paper2.bingo', 'tunic.flaghouse.entry.colorbook', 'tunic.wildlife.center.coffee', 'tunic.historicalsociety.collection_flag.gramps.recap', 'tunic.wildlife.center.wells.animals2', 'tunic.flaghouse.entry.flag_girl.symbol_recap', 'tunic.historicalsociety.closet_dirty.photo', 'tunic.historicalsociety.stacks.outtolunch', 'tunic.historicalsociety.frontdesk.archivist_glasses.confrontation_recap', 'tunic.historicalsociety.entry.boss.flag_recap', 'tunic.capitol_1.hall.boss.writeitup', 'tunic.library.frontdesk.worker.nelson_recap', 'tunic.historicalsociety.entry.wells.flag_recap', 'tunic.drycleaner.frontdesk.worker.done2', 'tunic.library.frontdesk.worker.flag_recap', 'tunic.library.frontdesk.worker.preflag', 'tunic.historicalsociety.basement.gramps.seeyalater', 'tunic.flaghouse.entry.flag_girl.hello_recap', 'tunic.historicalsociety.basement.gramps.whatdo', 'tunic.library.frontdesk.block_nelson', 'tunic.historicalsociety.cage.need_glasses', 'tunic.capitol_2.hall.chap4_finale_c', 'tunic.wildlife.center.fox.concern']\ntext_lists_grp = {\n    '0-4':text_lists1,\n    '5-12':text_lists2,\n    '13-22':text_lists3\n}    \ncluster_level_dict={'0-4': {'0-4': 6}, '5-12': {'5-12': 6}, '13-22': {'13-22': 6}}\nFQID1 = ['intro', 'gramps', 'teddy', 'photo', 'notebook', 'retirement_letter', 'tobasement', 'janitor', 'toentry', 'groupconvo', 'report', 'boss', 'wells', 'directory', 'tocollection', 'cs', 'tunic', 'tunic.hub.slip', 'tostacks', 'outtolunch', 'tocloset', 'tomap', 'tunic.historicalsociety', 'tunic.kohlcenter', 'plaque', 'plaque.face.date', 'togrampa', 'tunic.capitol_0', 'chap1_finale', 'chap1_finale_c', 'block_tocollection', 'block_0', 'doorblock', 'block_tomap1', 'block_tomap2']\nFQID2 = ['gramps', 'photo', 'tobasement', 'janitor', 'toentry', 'boss', 'directory', 'tocollection', 'tunic', 'tunic.hub.slip', 'tostacks', 'outtolunch', 'tomap', 'tunic.historicalsociety', 'tunic.kohlcenter', 'plaque', 'tunic.capitol_0', 'tocloset_dirty', 'what_happened', 'trigger_scarf', 'trigger_coffee', 'tunic.capitol_1', 'tofrontdesk', 'archivist', 'magnify', 'tunic.humanecology', 'worker', 'businesscards', 'businesscards.card_0.next', 'businesscards.card_1.next', 'businesscards.card_bingo.next', 'businesscards.card_bingo.bingo', 'tohallway', 'tunic.drycleaner', 'logbook', 'logbook.page.bingo', 'tunic.library', 'tomicrofiche', 'reader', 'reader.paper0.next', 'reader.paper1.next', 'reader.paper2.bingo', 'wellsbadge', 'journals', 'journals.hub.topics', 'journals.pic_0.next', 'journals.pic_1.next', 'journals.pic_2.bingo', 'chap2_finale_c', 'reader.paper2.next', 'journals.pic_2.next', 'reader.paper2.prev', 'reader.paper0.prev', 'door_block_clean', 'door_block_talk', 'block', 'reader.paper1.prev', 'block_magnify', 'block_0', 'block_badge', 'block_badge_2', 'block_1']\nFQID3 = ['gramps', 'teddy', 'photo', 'tobasement', 'toentry', 'boss', 'wells', 'directory', 'tocollection', 'tunic', 'tunic.hub.slip', 'tostacks', 'outtolunch', 'tomap', 'tunic.historicalsociety', 'tunic.kohlcenter', 'plaque', 'tocloset_dirty', 'tunic.capitol_1', 'tofrontdesk', 'tunic.humanecology', 'worker', 'businesscards', 'businesscards.card_0.next', 'businesscards.card_1.next', 'businesscards.card_bingo.next', 'tohallway', 'tunic.drycleaner', 'logbook', 'tunic.library', 'tomicrofiche', 'reader', 'reader.paper0.next', 'reader.paper1.next', 'journals', 'journals.hub.topics', 'journals.pic_0.next', 'journals.pic_1.next', 'journals.pic_2.bingo', 'ch3start', 'seescratches', 'tocage', 'glasses', 'directory.closeup.archivist', 'key', 'unlockdoor', 'confrontation', 'savedteddy', 'tocollectionflag', 'groupconvo_flag', 'tunic.capitol_2', 'tunic.wildlife', 'coffee', 'crane_ranger', 'remove_cup', 'expert', 'tracks', 'tracks.hub.deer', 'tunic.flaghouse', 'flag_girl', 'colorbook', 'reader_flag', 'reader_flag.paper0.next', 'reader_flag.paper1.next', 'reader_flag.paper2.bingo', 'archivist_glasses', 'journals_flag', 'journals_flag.hub.topics_old', 'journals_flag.hub.topics', 'journals_flag.pic_0.bingo', 'journals_flag.pic_0.next', 'chap4_finale_c', 'reader.paper2.next', 'journals.pic_2.next', 'lockeddoor', 'reader.paper2.prev', 'reader.paper0.prev', 'reader_flag.paper1.prev', 'journals_flag.pic_0_old.next', 'journals_flag.pic_1_old.next', 'reader_flag.paper2.next', 'journals_flag.pic_1.bingo', 'journals_flag.pic_1.next', 'journals_flag.pic_2.bingo', 'journals_flag.pic_2.next', 'reader_flag.paper0.prev', 'reader.paper1.prev', 'journals_flag.pic_2_old.next', 'reader_flag.paper2.prev', 'need_glasses', 'block_nelson', 'fox']\nFQID_grp = {\n    '0-4':FQID1,\n    '5-12':FQID2,\n    '13-22':FQID3\n}\nTEXTS=['undefined', 'Whatcha doing over there, Jo?', 'Just talking to Teddy.', 'I gotta run to my meeting!', 'Can I come, Gramps?', 'Sure thing, Jo. Grab your notebook and come upstairs!', 'See you later, Teddy.', \"I get to go to Gramps's meeting!\", 'Now where did I put my notebook?', '\\\\u00f0\\\\u0178\\\\u02dc\\\\u00b4', 'I love these photos of me and Teddy!', 'Found it!', 'Gramps is in trouble for losing papers?', \"This can't be right!\", 'Gramps is a great historian!', \"Hmm. Button's still not working.\", \"Let's get started. The Wisconsin Wonders exhibit opens tomorrow!\", 'Who wants to investigate the shirt artifact?', \"Not Leopold here. He's been losing papers lately.\", 'Hey!', \"It's true, they do keep going missing lately.\", 'See?', 'Besides, I already figured out the shirt.', \"It's a women's basketball jersey!\", 'That settles it.', 'Wells, finish up your report.', \"Leopold, why don't you help me set up in the Capitol?\", 'We need to talk about that missing paperwork.', 'Will do, Boss.', \"Hey Jo, let's take a look at the shirt!\", 'Your grampa is waiting for you in the collection room.', \"Why don't you go catch up with your grampa?\", 'What a fascinating artifact!', \"Wow, that's so cool, Gramps!\", 'Can I take a closer look?', \"Hmmm. Shouldn't you be doing your homework?\", \"It's already all done!\", 'Plus, my teacher said I could help you out for extra credit!', \"Well, that's good enough for me.\", 'Go ahead, take a peek at the shirt!', 'This looks like a clue!', \"I'll record this in my notebook.\", 'Find anything?', 'Yes! This old slip from 1916.', 'I knew it!', \"I'm not so sure that this is a basketball jersey.\", 'Wait, you mean Wells is wrong?!', 'Could be. But we need evidence!', \"Why don't you head to the Basketball Center and rustle up some clues?\", 'Sure!', \"I'll be at the Capitol. Let me know if you find anything!\", 'Better check back later.', \"That's it!\", \"The slip is from 1916 but the team didn't start until 1974!\", 'Our shirt is too old to be a basketball jersey!', 'I need to get to the Capitol and tell Gramps!', 'What are you still doing here,  Jolie?', 'Go find your grampa and get to work!', 'Oh no!', 'What happened here?!', \"I don't know!\", 'I got here and the whole place was a mess!', 'Can you help me tidy up?', \"Teddy's scarf! Somebody must've taken him!\", 'Try not to panic, Jo.', 'Maybe he just got scared and ran off.', 'But he never goes anywhere without his scarf!', \"I think he's in trouble!\", 'Is this your coffee, Gramps?', \"Nope, that's from Bean Town. I only drink Holdgers!\", \"Who could've done this?\", \"It must've been Wells.\", \"He's always trying to get you in trouble, and he doesn't like animals!\", 'Slow down, Jo.', 'But what if Wells kidnapped Teddy?', 'Then we need evidence.', \"You're right, Gramps. Let's investigate!\", \"I'm afraid my papers have gone missing in this mess.\", \"You'll have to get started without me.\", \"Okay. I'll find Teddy!\", \"And I'll figure out the shirt, too.\", 'I knew I could count on you, Jo!', \"Why don't you go upstairs and see the archivist?\", \"He's our expert record keeper.\", 'I need your help!', 'Who are you?', \"I'm Leopold's grandkid!\", \"Sorry, I'm too busy for kids right now.\", 'Now if only I could read this thing.', \"Can't believe I lost my reading glasses.\", 'I bet the archivist could use this!', \"Ah, that's better!\", 'Did you have a question?', 'Yes! I was wondering-', 'Wait a minute!', 'Where did you get that coffee?', \"Oh, that's from Bean Town.\", 'I ran into Wells there this morning.', 'Wells? I knew it!', 'Do you know anything about this slip?', 'I found it on an old shirt.', 'An old shirt? Try the university.', 'You can talk to a textile expert there.', \"What's a textile expert?\", 'They study clothes and fabric.', 'Great! Thanks for the help!', 'Head over to the university.', 'Hello there!', 'Wow! What is all this stuff?', \"It's our Norwegian Craft exhibit!\", 'Can I give you the tour?', \"Sorry, I'm in a hurry.\", 'Do you know what this slip is?', 'Looks like a dry cleaning receipt.', 'Thanks.', 'Now I Just need to find all the cleaners from way back in 1916.', 'Maybe I can help!', \"I've got a stack of business cards from my favorite cleaners.\", \"Why don't you take a look?\", 'This place was around in 1916! I can start there!', \"You haven't seen any badgers around here, have you?\", 'Badgers? No.', 'Okay. Thanks anyway.', 'Hi! How can I help you?', 'I need to find the owner of this slip.', \"Well, I can't show our log books to just anybody.\", 'Please?', \"It's for Grampa Leo. He's a historian!\", 'Leo... you mean Leopold?', 'Your gramps is awesome! Always full of stories.', \"Guess it couldn't hurt to let you take a look.\", \"Here's the log book.\", \"It's a match!\", 'Theodora Youmans must be the owner!', 'Do you know who Theodora Youmans is?', \"Hmmm... not sure. Why don't you try the library?\", 'Thanks for the help!', 'Oh, hello there!', 'How can I help you?', 'Have you seen a badger around here?', \"I'm afraid not.\", 'Please let me know if you do.', \"I'm also looking for Theodora Youmans. Have you heard of her?\", 'Theodora Youmans? Of course!', \"Check out our microfiche. It's right through that door.\", 'Youmans was a suffragist!', 'She helped get votes for women!', 'Wells! What was he doing here? I should ask the librarian.', 'What was Wells doing here?', 'He was looking for a taxidermist.', \"What's a taxidermist?\", 'Not sure. Here, let me look it up.', '\\\\Taxidermy: the art of preparing, stuffing, and mounting the skins of animals.\\\\', 'Oh no... Teddy!', 'Can you help me find Wells?', 'You could ask the archivist. He knows everybody!', \"Jolie! I was hoping you'd stop by. Any news on the shirt artifact?\", \"I haven't quite figured it out just yet...\", \"Well, get on it. I'm counting on you and your gramps to figure this out!\", 'Can you help me? I need to find Wells!', \"I haven't seen him.\", 'Please? This is really important.', \"Sorry, can't help you.\", 'Do you have any info on Theodora Youmans?', 'Theodora Youmans? Is that who owned the shirt?', 'Yep.', \"Why didn't you say so?\", 'Youmans was a suffragist here in Wisconsin.', 'She led marches and helped women get the right to vote!', \"Wait a sec. Women couldn't vote?!\", 'Nope. But Youmans and other suffragists worked hard to change that.', 'Thanks to them, Wisconsin was the first state to approve votes for women!', 'Wow!', \"Here's a call number to find more info in the Stacks.\", 'Where are the Stacks?', 'Right outside the door.', 'Hey, this is Youmans!', \"And look! She's wearing the shirt!\", 'I should go to the Capitol and tell everyone!', 'Jo!', 'Check out the next artifact!', 'What is it?', \"I think it's a flag! Pretty interesting, huh?\", \"It's really cool, Gramps. But I'm worried about Teddy.\", \"He's still missing!\", \"We'll find him, Jo.\", 'Want to look for more clues?', \"We'll find Teddy.\", 'We just have to keep our eyes open!', 'Hey, look at those scratches!', 'The kidnapper probably took Teddy on the elevator!', \"You're right, Jo!\", \"Why isn't the button working?\", \"We'll need a key card.\", 'I had one, but Teddy chewed it up.', \"I've got Wells's ID!\", 'What should we do next?', \"I need to take the artifact upstairs. Why don't you investigate those scratch marks?\", \"Okay. I'll try.\", 'Teddy, here I come!', 'I wonder whose glasses these are.', 'Teddy!!!', \"Hang on. I'll get you out of there!\", 'Whoever lost these glasses probably took Teddy!', 'How can I find out whose glasses these are?', 'Oh! There was a staff directory in the entryway!', \"I'll go look at everyone's pictures!\", 'Those are the same glasses!', \"The archivist must've taken Teddy!\", \"Yes! It's the key for Teddy's cage!\", 'I found the key!', \"Come on, let's get out of here!\", \"Here's your scarf back!\", 'What are you doing down here?', 'And how did that badger get free?', \"I'm here to rescue my friend!\", \"What's going on here?\", 'Thanks for coming, Boss.', 'I told you!', 'I captured a badger in our museum!', \"He's been eating my lunch every day this week!\", 'He has??', \"I've seen him eating homework and important papers, too.\", \"Jolie- keep your badger under control, or he'll have to go.\", 'And you, Frank-', \"You can't just steal Jolie's pet.\", 'Ugh. Fine.', 'Alright, Jolie. Back to work.', \"Come on, Teddy. Let's go help Gramps!\", \"Let's go help Gramps!\", 'Gramps must be up in the collection room.', \"Let's go find him!\", \"Teddy! I'm glad to see you.\", 'The archivist had him locked up!', 'Poor badger.', \"You're becoming quite the detective, Jo.\", 'Notice any clues about this flag?', 'Well... it looks hand-stitched.', 'Good catch!', 'Go on, tell the boss what you found!', \"I'm telling you, Boss. Taxidermy is the way to go!\", 'Nonsense. I want live animals at the exhibit, not stuffed ones.', \"Ah, Jolie! I'm glad you're here.\", \"I'm putting you in charge of the flag case.\", 'Make sure to get some old photos for the exhibit, like last time!', \"Wait! Can't I do it?\", 'The symbol on the flag looks sort of like a deer hoof.', 'It could be an early design for the Wisconsin state flag!', 'Wells, you already have a job to do.', 'What now, kid?', 'Do you really think that symbol is a deer hoof?', 'Not sure.', 'Do you know where I can find a deer expert?', 'Hmm. You could try the Aldo Leopold Wildlife Center.', 'I have to head over there and check out the animals.', \"I'll ride with you!\", \"Come on, kid. Let's go.\", 'Head over to the Wildlife Center!', \"I'm sure they'll be able to help.\", 'People sure drink a lot of coffee around here.', \"I can't believe this.\", 'Ugh...', 'Oh no! What happened to that crane?', 'Her beak is stuck in a coffee cup.', \"It's lucky we found her.\", 'Ugh! Those cups are all over the place.', \"I need to get her free. She won't hold still!\", 'Can Teddy and I help?', 'Sure! Give it a try.', 'Careful. That beak is sharp!', 'We need to calm her down, Teddy.', 'Any ideas?', '\\\\u00f0\\\\u0178\\\\u00a6\\\\u2014', 'Oh yeah, cranes eat insects!', 'Luckily there are tons of insects around here...', 'Got one!', \"Maybe she'll let me take off the cup!\", \"It's OK, girl! Look, I found you a cricket!\", 'You did it! Thanks, kid.', 'Can I help you with anything?', \"I'm investigating this symbol.\", 'Does it look like a deer hoof?', \"There's a diagram of animal tracks over there.\", 'Go take a look!', \"That hoofprint doesn't match the flag!\", 'Thanks for your help, kid!', \"So? What'd you find out?\", \"Looks like it's not a deer hoof.\", \"Oh no. If I don't impress the boss soon,  I'm gonna get fired!\", 'Hey, Wells...', 'I think I might be able to help you.', \"No thanks. I don't need help from kids.\", 'Are you sure? I know where you can find a real, live badger for the exhibit!', 'Wait! What?! Really?', 'Wells, meet Teddy.', '\\\\u00f0\\\\u0178\\\\u02dc\\\\u0160', \"He says he'd be willing to help out.\", 'Yes!!!', 'We still need to figure out that flag. Do you know anyone who could help?', \"Hmm. Let's see...\", 'Actually, I went to school with somebody who LOVES old flags.', \"Why don't you go talk to her? I'll let her know you're coming.\", 'Hey, nice dog! What breed is he?', \"Actually, he's a badger.\", \"Oh, cool! I've never seen a badger in real life.\", \"You've got a million flags here!\", \"Yep. I'm a vexillophile!\", \"What's a vexillophile? \", 'It just means flag expert. How can I help?', \"I'm investigating this flag.\", 'Can you take a look?', \"Hey, I've seen that symbol before! Check it out!\", '\\\\Ecology flag, by Ron Cobb.\\\\', \"It's an ecology flag!\", 'Do you know what this flag was used for?', \"I'm not sure.\", \"If I were you, I'd go to the library and do some digging.\", 'Good idea. Thanks!', 'Welcome back, Dear! How can I help you?', 'I need to learn more about this flag!', 'It has something to do with ecology.', 'Hmm... those stripes remind me of the American flag.', 'Your flag must have been part of a national movement!', \"Go check the microfiche. Maybe you'll find something!\", \"Hey! That's Governor Nelson in front of our flag!\", 'I found the flag! Governor Nelson used it on the first Earth Day!', 'Wow! You figured it out!', 'Now I just need some old photos, like last time.', 'The boss is gonna love it!', 'You could try the archives.', 'Though the archivist might be too busy to help...', 'Okay. Thanks!', 'What are you doing here?', \"We're looking for some photos.\", \"It's for the flag display!\", 'Wait a minute...', \"YOU'RE the new history detective everybody's talking about?\", \"Teddy's helping too.\", 'What kind of photos do you need?', 'Something to do with ecology and Wisconsin.', \"Here's a call number for the Stacks. Go find some photos.\", 'Look at all those activists!', 'This is perfect for the exhibit.', 'I should go to the Capitol and tell Mrs. M!', 'I should see what Grampa is up to!', 'What should I do first?', 'Head upstairs and talk to the archivist. He might be able to help!', \"It's locked!\", \"Jolie! I was hoping you'd stop by. Any news on the flag artifact?\", \"Well, get on it. I'm counting on you to figure this out!\", 'Nice seeing you, Jolie!', \"It's such a nice fall day.\", 'I love these photos of me and Teddy.', \"Why don't you go talk to the boss?\", \"She's right outside.\", 'My friend is a flag expert.', 'She should be able to help you out.', 'There are some old newspapers loaded up in the microfiche.', 'The Stacks are right outside the door. Go find some photos!', 'Ugh. Meetings are so boring.', 'Grab your notebook and come upstairs!', 'Hang tight, Teddy.', \"I'll hurry back and then we can go exploring!\", 'Well, Leopold here is always losing papers...', 'Ha. Told you so!', 'Can we hurry up, Gramps?', 'Teddy and I were gonna go climb that huge tree out back!', \"Hmmm. Don't forget about your homework.\", 'Your teacher said you missed 7 assignments in a row!', 'So? History is boring!', 'I suppose historians are boring, too?', \"No way, Gramps. You're the best!\", 'Then do it for me!', 'Your teacher said you could help me for extra credit.', 'A boring old shirt.', 'Just this old slip from 1916.', 'Do I have to?', 'What the-', 'I have an idea.', \"He's wrong about old shirts and his name rhymes with \\\\smells\\\\...\", 'BUT WELLS STOLE TEDDY!', 'Could be. But we need evidence.', \"Fine. Let's investigate!\", \"Don't worry, Gramps. I'll find Teddy!\", \"Please let me know if you do. It's important!\", 'I need to find Wells right away! Do you know where he is?', 'I need to find Wells!!!', \"I can't calm down. This is important!\", \"I don't have time for this, Gramps.\", 'Teddy is still missing!', \"Let's follow those scratch marks!\", \"I can't go with you. I need to take the artifact upstairs.\", \"It's okay, Gramps. I'll go by myself.\", 'You stole Teddy! How could you?!', \"No he hasn't!\", \"Yes, he has. I've seen him eating homework and important papers, too.\", 'Come on, Teddy.', \"Let's go find Gramps!\", 'I think I can help with your animal problem.', \"Ha! I don't need your help.\", \"Fine. Then I guess you don't want a real, live badger for the exhibit.\", \"Oh, trust me. He'll make time.\", 'Um... what did you want me to do again?', 'Head over to the Basketball Center.', 'Hopefully you can find some clues!', 'I should stay and look for clues!', 'Where should I go again?', 'You could try the archivist. Maybe he can help you find Wells!', 'Hi, Mrs. M.', 'Head back to the museum. Your gramps is waiting for you.', \"I don't need that right now.\", 'Meetings are BORING!', \"I feel like I'm forgetting something.\", 'Gramps is the best historian ever!', 'This button never works!', \"Why don't you go play with your grampa?\", \"Look at that! It's the bee's knees!\", \"Well, I did SOME of those. I just couldn't find them!\", 'Did you do all of them?', 'No... because history is boring!', 'Hooray, a boring old shirt.', 'Hot Dog! I knew it!', 'Ooh, I like clues!', 'Hopefully you can rustle up some clues!', 'I got here and the whole place was ransacked!', 'Hold your horses, Jo.', '*grumble grumble*', 'And you are?', \"I don't have time for kids.\", 'Now if only I could read this thing. Blasted tiny letters...', 'Knew what?', 'Did you have a question or not?', 'Yes!', \"You're still here? I'm trying to work!\", 'Run along to the university.', 'Ooh, thanks!', 'Now I just need to find all the cleaners from wayyyy back in 1916.', 'Yikes... this could take a while.', 'Hi! *cough*', 'Can you help-', '*cough cough*', 'Can you help me-', '*COUGH COUGH COUGH*', 'Um, are you okay?', \"Oh, I'm fine! Just a little hoarse.\", 'Ha! What do you call a pony with a sore throat?', 'Huh?', 'A little horse!', \"Ha! You're funny.\", 'I got that one from my Gramps!', 'Can you help me? I need to find the owner of this slip.', \"Yup, that's him!\", \"Unless you're too busy horsing around.\", 'Ha! Good one.', \"You look like you're on a mission.\", 'Two missions, actually!', 'Oh my!', 'I need to find Wells right away!! Do you know where he is?', \"Calm down, kid. I haven't seen him.\", \"I think it's a flag! Pretty spiffy, eh?\", \"Great Scott, you're right!\", \"Jo! I can't go with you. I need to take the artifact upstairs.\", '\\\\u00f0\\\\u0178\\\\u02dc\\\\u00ad', '\\\\u00e2\\\\u009d\\\\u00a4\\\\u00ef\\\\u00b8\\\\u008f', 'GRRRRRRR', 'GAH! And what is THAT doing out of its cage?!', '\\\\u00f0\\\\u0178\\\\u02dc\\\\u0090', 'Teddy! Did you really eat his lunch?', \"Did you steal Gramps's paperwork too?!\", 'And my homework?!?!', 'See?!', \"That thing's a monster!\", \"I don't have time for this.\", 'YEAH!', 'Wait- me?', \"You can't just steal Jolie's pet. Don't you know badgers are protected animals?\", 'Besides, he looks friendly to me.', 'Wha?!', '\\\\u00f0\\\\u0178\\\\u02dc\\\\u009d', \"Teddy! I'm sure glad to see you.\", 'Gadzooks! Poor critter.', 'Aha! Good catch, Jo.', 'Not sure. Do I look like a deer expert to you?', 'Ugh. I have to head over there and check out the animals.', 'FINE. That possum better not scratch my leather seats...', \"He's a badger!\", '\\\\u00f0\\\\u0178\\\\u00a7\\\\u02dc', 'Yoga does sound nice.', \"But cranes can't do yoga, Teddy!\", '\\\\u00f0\\\\u0178\\\\u008d\\\\u00a9', \"Cranes don't eat donuts!\", 'Besides, you just ate my last snack.', \"Gah. I can't believe this.\", \"I'm a historian, not a zookeeper!\", 'And this place is dirty, and itchy, and-', 'I love it!', \"Of course you do. You've got a rodent following you around.\", \"Actually, badgers aren't rodents-\", 'Whatever.', 'Great. Just great. Could this day get any worse?!', \"Yes!!! I'm saved!\", 'A real, live ferret!', \"He's. A. Badger.\", 'And we still need to figure out that flag!', \"Fine, fine. Let's see...\", 'A vexy-wha?', 'Ooh... \\\\Ecology flag, by Ron Cobb.\\\\', 'The boss is gonna love it!!!', \"You again! Don't let him hurt me!\", '\\\\u00f0\\\\u0178\\\\u2122\\\\u201e', \"Actually, we're just here for some photos.\", 'Guess so!', 'YOU?!', \"Just please, don't let your badger eat them!\", 'I should help Gramps clean.', \"Maybe there's a clue in this mess!\", \"Poor Gramps! I should make sure he's okay.\", 'The archivist said I should look in the stacks.', 'There should be some info about that symbol in my book.', 'I should go talk to Gramps!', 'Yeah. Thanks anyway.', 'What are you waiting for? The Stacks are right outside the door.', 'Yes! This cool old slip from 1916.', 'Are you okay?', \"I'll be in the collection room. Come find me when you're ready to check out the artifact.\", 'Good luck!', 'What?!', 'Can I ride with you?', \"Don't worry, he won't! (And he's a badger, by the way.)\", 'Ugh... I think that lynx is looking at me funny.', \"Don't worry, Teddy won't eat your lunch anymore!\", \"We're just looking for photos for the flag display.\", \"Weren't you going to check out our microfiche?\", \"I'm sure you'll find Theodora in there somewhere!\", \"But I hear the museum's got one on the loose!\", 'Well? What are you still doing here?', 'So much cleaning to do...', 'I should check out that pair of glasses.', 'I should ask the librarian where to go next.', \"Check out the archives. They've got tons of old photos!\", 'I used to have a magnifying glass around here\\\\u00e2\\\\u20ac\\\\u00a6', \"Come on, kid. You're slowing me down.\", \"Did you drop something, Dear? There's a card on the floor.\", 'Take a look!', 'I should see what Gramps is up to!', 'I found it!', 'Theodora wearing the shirt!', 'You better get to the capitol!', 'Nice decorations.', 'Did you drop something, Dear?', 'Gramps said to look for clues. Better look around.', 'I should find out if she can help me!', 'Ooh, nice decorations!', 'The libarian said I could find some information on Youmans in here...', 'Have a look at the artifact!', 'What is it, Teddy?', 'Oh no... they got sick from polluted water?', 'Poor foxes!', 'I should ask the librarian why Wells was here.', \"I wonder if there's a clue in those business cards...\", 'Thanks. Did you figure out the shirt?', 'Jolie! Where have you been?', 'The exhibit opens tomorrow.', 'Welcome back, Jolie. Did you figure out the shirt?', 'Wells got in trouble for littering at the Wildlife Center.', 'I should check that logbook to see who owned this slip...', 'AND I know who took Teddy!', 'Who is Teddy?', \"And where's your grampa?\", 'Sorry for the delay, Boss.', 'I had some cleaning up to do in my office.', 'Mrs. M, I think Wells kidnapped Teddy.', \"And he messed up Gramps's office, too!\", 'One step at a time, Jo.', 'Did you figure out the shirt?', 'I knew you could do it, Jo!', 'Now can I tell you what happened to Teddy?', 'He needs our help!', \"Sorry I'm late.\", \"Wells! Where's Teddy? Is he okay?\", 'I figured out that you kidnapped him!', 'Easy, Jo.', \"Why don't you prove your case?\", \"It'll be okay, Jo. We'll find Teddy!\", 'Nice work on the shirt, Jolie!', 'Leopold, can you run back to the museum?', 'Sounds good, Boss.', 'Jo, meet me back at my office.', 'I hope you find your badger, kid.', 'Thanks!', \"Are you going home now? Tomorrow's the big day!\", 'He got a park named after him? Cool!', 'Come on, Jo!', \"Meet me back in my office and we'll get started!\", 'Here I am!', 'Wells sabotaged Gramps!', 'AND he stole Teddy!']\nLEVELS = [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22]\nlevel_groups = [\"0-4\", \"5-12\", \"13-22\"]\nev_feature = ['cutscene_click', 'person_click', 'navigate_click', 'observation_click', 'notification_click', 'object_click','map_click',  'notebook_click']# 'checkpoint',\nroom_lists1 = ['tunic.historicalsociety.entry', 'tunic.historicalsociety.stacks', 'tunic.historicalsociety.basement', 'tunic.kohlcenter.halloffame', 'tunic.historicalsociety.collection', 'tunic.historicalsociety.closet', 'tunic.capitol_0.hall']\nroom_lists2 = ['tunic.historicalsociety.entry', 'tunic.library.frontdesk', 'tunic.historicalsociety.frontdesk', 'tunic.historicalsociety.stacks', 'tunic.historicalsociety.closet_dirty', 'tunic.humanecology.frontdesk', 'tunic.historicalsociety.basement', 'tunic.kohlcenter.halloffame', 'tunic.library.microfiche', 'tunic.drycleaner.frontdesk', 'tunic.historicalsociety.collection', 'tunic.capitol_1.hall', 'tunic.capitol_0.hall']\nroom_lists3 = ['tunic.historicalsociety.entry', 'tunic.wildlife.center', 'tunic.historicalsociety.cage', 'tunic.library.frontdesk', 'tunic.historicalsociety.frontdesk', 'tunic.historicalsociety.stacks', 'tunic.historicalsociety.closet_dirty', 'tunic.humanecology.frontdesk', 'tunic.historicalsociety.basement', 'tunic.kohlcenter.halloffame', 'tunic.library.microfiche', 'tunic.drycleaner.frontdesk', 'tunic.historicalsociety.collection', 'tunic.flaghouse.entry', 'tunic.historicalsociety.collection_flag', 'tunic.capitol_1.hall', 'tunic.capitol_2.hall']\nfqid_lists1 = ['gramps', 'wells', 'toentry', 'groupconvo', 'tomap', 'tostacks', 'tobasement', 'boss', 'cs', 'teddy', 'tunic.historicalsociety', 'plaque', 'directory', 'tunic', 'tunic.kohlcenter', 'plaque.face.date', 'notebook', 'tunic.hub.slip', 'tocollection', 'tunic.capitol_0', 'photo', 'intro', 'retirement_letter', 'togrampa', 'janitor', 'chap1_finale', 'report', 'outtolunch', 'chap1_finale_c', 'block_0', 'doorblock', 'tocloset', 'block_tomap2', 'block_tocollection', 'block_tomap1']\nfqid_lists2 = ['worker', 'archivist', 'gramps', 'toentry', 'tomap', 'tostacks', 'tobasement', 'boss', 'journals', 'businesscards', 'tunic.historicalsociety', 'tofrontdesk', 'plaque', 'tunic.drycleaner', 'tunic.library', 'trigger_scarf', 'reader', 'directory', 'tunic.capitol_1', 'journals.pic_0.next', 'tunic', 'what_happened', 'tunic.kohlcenter', 'tunic.humanecology', 'logbook', 'businesscards.card_0.next', 'journals.hub.topics', 'logbook.page.bingo', 'journals.pic_1.next', 'reader.paper0.next', 'trigger_coffee', 'wellsbadge', 'journals.pic_2.next', 'tomicrofiche', 'tocloset_dirty', 'businesscards.card_bingo.bingo', 'businesscards.card_1.next', 'tunic.hub.slip', 'journals.pic_2.bingo', 'tocollection', 'chap2_finale_c', 'tunic.capitol_0', 'photo', 'reader.paper1.next', 'businesscards.card_bingo.next', 'reader.paper2.bingo', 'magnify', 'janitor', 'tohallway', 'outtolunch', 'reader.paper2.next', 'door_block_talk', 'block_magnify', 'reader.paper0.prev', 'block', 'block_0', 'door_block_clean', 'reader.paper2.prev', 'reader.paper1.prev', 'block_badge', 'block_badge_2', 'block_1']\nfqid_lists3 = ['worker', 'gramps', 'wells', 'toentry', 'confrontation', 'crane_ranger', 'flag_girl', 'tomap', 'tostacks', 'tobasement', 'archivist_glasses', 'boss', 'journals', 'seescratches', 'groupconvo_flag', 'teddy', 'expert', 'businesscards', 'ch3start', 'tunic.historicalsociety', 'tofrontdesk', 'savedteddy', 'plaque', 'glasses', 'tunic.drycleaner', 'reader_flag', 'tunic.library', 'tracks', 'tunic.capitol_2', 'reader', 'directory', 'tunic.capitol_1', 'journals.pic_0.next', 'unlockdoor', 'tunic', 'tunic.kohlcenter', 'tunic.humanecology', 'colorbook', 'logbook', 'businesscards.card_0.next', 'journals.hub.topics', 'journals.pic_1.next', 'journals_flag', 'reader.paper0.next', 'tracks.hub.deer', 'reader_flag.paper0.next', 'journals.pic_2.next', 'tomicrofiche', 'journals_flag.pic_0.bingo', 'tocloset_dirty', 'businesscards.card_1.next', 'tunic.wildlife', 'tunic.hub.slip', 'tocage', 'journals.pic_2.bingo', 'tocollectionflag', 'tocollection', 'chap4_finale_c', 'lockeddoor', 'journals_flag.hub.topics', 'reader_flag.paper2.bingo', 'photo', 'tunic.flaghouse', 'reader.paper1.next', 'directory.closeup.archivist', 'businesscards.card_bingo.next', 'remove_cup', 'journals_flag.pic_0.next', 'coffee', 'key', 'reader_flag.paper1.next', 'tohallway', 'outtolunch', 'journals_flag.hub.topics_old', 'journals_flag.pic_1.next', 'reader.paper2.next', 'reader_flag.paper2.next', 'journals_flag.pic_1.bingo', 'journals_flag.pic_2.next', 'journals_flag.pic_2.bingo', 'reader.paper0.prev', 'reader_flag.paper0.prev', 'reader.paper2.prev', 'reader.paper1.prev', 'reader_flag.paper2.prev', 'reader_flag.paper1.prev', 'journals_flag.pic_0_old.next', 'journals_flag.pic_1_old.next', 'block_nelson', 'journals_flag.pic_2_old.next', 'need_glasses', 'fox']\n\ndef applyNewClickAggs(group,field,alias,feature_suffix): \n        #print(f\"Creating click agg features for {field} grp {grp}\")\n        aggs = []\n        for level in group:\n            clickagggs=[\n                *[pl.col(\"event_name\").filter((pl.col(\"event_name\") == c) & (pl.col(field)==level)).count().alias(f\"{c}_EN_by_{alias}{level}_counts_{feature_suffix}\") for c in ev_feature],\n                *[pl.col(\"elapsed_time_diff\").filter((pl.col(\"event_name\") == c) & (pl.col(field)==level)).sum().alias(f\"{c}_ET_EN_by_{alias}{level}_sum_{feature_suffix}\") for c in ev_feature],\n                *[pl.col(\"event_name\").filter((pl.col(\"name\") == c) & (pl.col(field)==level)).count().alias(f\"{c}_EN_by_{alias}{level}_counts_{feature_suffix}\") for c in name_feature],\n                *[pl.col(\"elapsed_time_diff\").filter((pl.col(\"name\") == c) & (pl.col(field)==level)).sum().alias(f\"{c}_ET_EN_by_{alias}{level}_sum_{feature_suffix}\") for c in name_feature],\n             ]\n            aggs = aggs +clickagggs  \n        return aggs\nrcp_feats = ['observation_click', 'person_click']    \ndef recapAggs(group,field,alias,feature_suffix):   \n    aggs = []\n    for level in group:\n        clickagggs= [\n                    *[pl.col(\"event_name\").filter((pl.col(\"event_name\") == c) & (pl.col(field)==level)).count().alias(f\"{c}_EN_by_{alias}{level}_counts_{feature_suffix}\") for c in rcp_feats],\n                    *[pl.col(\"elapsed_time_diff\").filter((pl.col(\"event_name\") == c) & (pl.col(field)==level)).sum().alias(f\"{c}_ET_EN_by_{alias}{level}_sum_{feature_suffix}\") for c in rcp_feats],\n                 ]  \n        aggs = aggs +clickagggs  \n    #print(f\"Adding aggs recap {len(aggs)}\")    \n    return aggs\nrecaps_25 = ['My friend is a flag expert.',\n       'She should be able to help you out.',\n       \"If I were you, I'd go to the library and do some digging.\",\n       'The Stacks are right outside the door. Go find some photos!',\n       'Head over to the university.',\n       'You can talk to a textile expert there.',\n       \"You're still here? I'm trying to work!\",\n       'Run along to the university.',\n       \"Why don't you go talk to the boss?\", \"She's right outside.\",\n       'Um... what did you want me to do again?',\n       'Head over to the Basketball Center.',\n       'Hopefully you can find some clues!',\n       'Hopefully you can rustle up some clues!',\n       'Where should I go again?',\n       'You could try the archivist. Maybe he can help you find Wells!',\n       \"Check out the archives. They've got tons of old photos!\",\n       'The archivist said I should look in the stacks.',\n       'What are you waiting for? The Stacks are right outside the door.',\n       'There are some old newspapers loaded up in the microfiche.',\n       'Head over to the Wildlife Center!',\n       \"I'm sure they'll be able to help.\",\n       'I should stay and look for clues!',\n       'There should be some info about that symbol in my book.',\n       \"I feel like I'm forgetting something.\",\n       'I should go talk to Gramps!',\n       'I should ask the librarian where to go next.',\n       'The libarian said I could find some information on Youmans in here...',\n       'I should see what Grampa is up to!',\n       'I should see what Gramps is up to!',\n       'I should ask the librarian why Wells was here.',\n       'Gramps said to look for clues. Better look around.',\n       'I should find out if she can help me!',\n       \"I wonder if there's a clue in those business cards...\",\n       'I should check that logbook to see who owned this slip...']  \nrecaps = ['My friend is a flag expert.',\n       'She should be able to help you out.',\n       \"If I were you, I'd go to the library and do some digging.\",\n       'The Stacks are right outside the door. Go find some photos!',\n       'Head over to the university.',\n       'You can talk to a textile expert there.',\n       \"You're still here? I'm trying to work!\",\n       'Run along to the university.',\n       \"Why don't you go talk to the boss?\", \"She's right outside.\",\n       'Um... what did you want me to do again?',\n       'Head over to the Basketball Center.',\n       'Hopefully you can find some clues!',\n       'Hopefully you can rustle up some clues!',\n       'The archivist said I should look in the stacks.',\n       'I should stay and look for clues!',\n       \"I feel like I'm forgetting something.\",\n       'I should go talk to Gramps!',\n       'I should ask the librarian where to go next.',\n       'The libarian said I could find some information on Youmans in here...']\naggs_04 =  applyNewClickAggs([i for i in range(0,5)],\"level\",\"LVL\",'04') \naggs_04 = aggs_04+applyNewClickAggs(['0-4'],\"level_group\",\"LG\",'04')   \naggs_04 = aggs_04 +applyNewClickAggs(room_lists1,\"room_fqid\",\"RL\",'04')   \naggs_04 = aggs_04 + applyNewClickAggs(fqid_lists1,\"fqid\",\"FQ\",'04')\naggs_04 = aggs_04 + recapAggs(recaps,\"text\",\"RCP\",'04')\n\naggs_512 =  applyNewClickAggs([i for i in range(5,13)],\"level\",\"LVL\",'512') \naggs_512 = aggs_512+applyNewClickAggs(['5-12'],\"level_group\",\"LG\",'512')   \naggs_512 = aggs_512+applyNewClickAggs(room_lists2,\"room_fqid\",\"RL\",'512')   \naggs_512 = aggs_512 + applyNewClickAggs(fqid_lists2,\"fqid\",\"FQ\",'512')\n#aggs_512 = aggs_512 + recapAggs(recaps,\"text\",\"RCP\",'512')\n\naggs_1322 =  applyNewClickAggs([i for i in range(13,23)],\"level\",\"LVL\",'1322') \naggs_1322 = aggs_1322+applyNewClickAggs(['13-22'],\"level_group\",\"LG\",'1322')   \naggs_1322 = aggs_1322 + applyNewClickAggs(room_lists3,\"room_fqid\",\"RL\",'1322')   \naggs_1322 = aggs_1322 + applyNewClickAggs(fqid_lists3,\"fqid\",\"FQ\",'1322') \n#aggs_1322 = aggs_1322 + recapAggs(recaps,\"text\",\"RCP\",'1322')\n    \ndef feature_engineer(x, grp, use_extra, feature_suffix): \n    FQID = FQID_grp[grp]\n    fqid_lists=FQID\n    text_lists = text_lists_grp[grp]\n    TEXTS = TEXTS_grp[grp]\n    aggs = [\n        pl.col(\"index\").count().alias(f\"session_number_{feature_suffix}\"), \n        \n        *[pl.col(c).drop_nulls().n_unique().alias(f\"{c}_unique_{feature_suffix}\") for c in CATS],\n        *[pl.col(c).mean().alias(f\"{c}_mean_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).min().alias(f\"{c}_min_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).max().alias(f\"{c}_max_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).std().alias(f\"{c}_std_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).last().alias(f\"{c}_last_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).first().alias(f\"{c}_first_{feature_suffix}\") for c in NUMS],\n        *[(pl.col(\"text\").str.contains(c)).sum().alias(f\"{c}_sum_{feature_suffix}\") for c in DIALOGS],\n\n        *[pl.col(c).quantile(0.1, \"nearest\").alias(f\"{c}_quantile1_{feature_suffix}\") for c in NUMS],\n        #*[pl.col(c).quantile(0.3, \"nearest\").alias(f\"{c}_quantile3_{feature_suffix}\") for c in NUMS],#new\n        *[pl.col(c).quantile(0.2, \"nearest\").alias(f\"{c}_quantile2_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).quantile(0.4, \"nearest\").alias(f\"{c}_quantile4_{feature_suffix}\") for c in NUMS],\n        #*[pl.col(c).quantile(0.5, \"nearest\").alias(f\"{c}_quantile5_{feature_suffix}\") for c in NUMS],#new\n        *[pl.col(c).quantile(0.6, \"nearest\").alias(f\"{c}_quantile6_{feature_suffix}\") for c in NUMS],\n        #*[pl.col(c).quantile(0.65, \"nearest\").alias(f\"{c}_quantile65_{feature_suffix}\") for c in NUMS],#new\n        *[pl.col(c).quantile(0.8, \"nearest\").alias(f\"{c}_quantile8_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).quantile(0.9, \"nearest\").alias(f\"{c}_quantile9_{feature_suffix}\") for c in NUMS],\n        ####################################################\n\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).min().alias(f\"{c}_ET_min_{feature_suffix}\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).last().alias(f\"{c}_ET_last_{feature_suffix}\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).first().alias(f\"{c}_ET_first_{feature_suffix}\") for c in event_name_feature],\n        #*[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).median().alias(f\"{c}_ET_median_{feature_suffix}\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\")==c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c in name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\")==c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\")==c).min().alias(f\"{c}_ET_min_{feature_suffix}\") for c in name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\")==c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\")==c).last().alias(f\"{c}_ET_last_{feature_suffix}\") for c in name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\")==c).first().alias(f\"{c}_ET_first_{feature_suffix}\") for c in name_feature],\n        #*[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\")==c).median().alias(f\"{c}_ET_median_{feature_suffix}\") for c in name_feature],\n\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col(\"text\").str.contains(c))).mean().alias(f\"{c}_DIA_mean_{feature_suffix}\") for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col(\"text\").str.contains(c))).max().alias(f\"{c}_DIA_max_{feature_suffix}\") for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col(\"text\").str.contains(c))).min().alias(f\"{c}_DIA_min_{feature_suffix}\") for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col(\"text\").str.contains(c))).std().alias(f\"{c}_DIA_std_{feature_suffix}\") for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col(\"text\").str.contains(c))).last().alias(f\"{c}_DIA_last_{feature_suffix}\") for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff\").filter((pl.col(\"text\").str.contains(c))).first().alias(f\"{c}_DIA_first_{feature_suffix}\") for c in DIALOGS],\n       \n        *[pl.col(\"text_fqid\").filter(pl.col(\"text_fqid\") == c).count().alias(f\"{c}_text_fqid_counts{feature_suffix}\") for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).min().alias(f\"{c}_ET_min_{feature_suffix}\") for\n          c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for\n          c in text_lists],\n       \n        *[pl.col(\"elapsed_time\").filter(pl.col(\"fqid\")==c).apply(lambda s: s.max() - s.min() if s.len()>0 else 0 ).alias(f\"{c}_cha_{feature_suffix}f\") for c in FQID],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\")==c).mean().alias(f\"{c}_ET_mean_{feature_suffix}f\") for c in FQID],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\")==c).max().alias(f\"{c}_ET_max_{feature_suffix}f\") for c in FQID],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\")==c).min().alias(f\"{c}_ET_min_{feature_suffix}f\") for c in FQID],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\")==c).std().alias(f\"{c}_ET_std_{feature_suffix}f\") for c in FQID],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\")==c).last().alias(f\"{c}_ET_last_{feature_suffix}f\") for c in FQID],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\")==c).first().alias(f\"{c}_ET_first_{feature_suffix}f\") for c in FQID],\n\n        *[pl.col(\"text\").filter(pl.col(\"text\")==c).sum().alias(f\"{c}_sum_{feature_suffix}\") for c in TEXTS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text\")==c).mean().alias(f\"{c}_ET_mean_{feature_suffix}f\") for c in TEXTS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text\")==c).max().alias(f\"{c}_ET_max_{feature_suffix}f\") for c in TEXTS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text\")==c).min().alias(f\"{c}_ET_min_{feature_suffix}f\") for c in TEXTS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text\")==c).std().alias(f\"{c}_ET_std_{feature_suffix}f\") for c in TEXTS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text\")==c).last().alias(f\"{c}_ET_last_{feature_suffix}f\") for c in TEXTS],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text\")==c).first().alias(f\"{c}_ET_first_{feature_suffix}f\") for c in TEXTS],\n\n        *[pl.col(\"elapsed_time_diff2\").filter(pl.col(\"event_name\")==c).mean().alias(f\"{c}_ET_mean_{feature_suffix}2\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff2\").filter(pl.col(\"event_name\")==c).max().alias(f\"{c}_ET_max_{feature_suffix}2\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff2\").filter(pl.col(\"event_name\")==c).min().alias(f\"{c}_ET_min_{feature_suffix}2\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff2\").filter(pl.col(\"event_name\")==c).std().alias(f\"{c}_ET_std_{feature_suffix}2\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff2\").filter(pl.col(\"event_name\")==c).last().alias(f\"{c}_ET_last_{feature_suffix}2\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff2\").filter(pl.col(\"event_name\")==c).first().alias(f\"{c}_ET_first_{feature_suffix}2\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff2\").filter(pl.col(\"name\")==c).mean().alias(f\"{c}_ET_mean_{feature_suffix}2\") for c in name_feature],\n        *[pl.col(\"elapsed_time_diff2\").filter(pl.col(\"name\")==c).max().alias(f\"{c}_ET_max_{feature_suffix}2\") for c in name_feature],\n        *[pl.col(\"elapsed_time_diff2\").filter(pl.col(\"name\")==c).min().alias(f\"{c}_ET_min_{feature_suffix}2\") for c in name_feature],\n        *[pl.col(\"elapsed_time_diff2\").filter(pl.col(\"name\")==c).std().alias(f\"{c}_ET_std_{feature_suffix}2\") for c in name_feature],\n        *[pl.col(\"elapsed_time_diff2\").filter(pl.col(\"name\")==c).last().alias(f\"{c}_ET_last_{feature_suffix}2\") for c in name_feature],\n        *[pl.col(\"elapsed_time_diff2\").filter(pl.col(\"name\")==c).first().alias(f\"{c}_ET_first_{feature_suffix}2\") for c in name_feature],\n\n        *[pl.col(\"elapsed_time_diff2\").filter((pl.col(\"text\").str.contains(c))).mean().alias(f\"{c}_DIA_mean_{feature_suffix}2\") for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff2\").filter((pl.col(\"text\").str.contains(c))).max().alias(f\"{c}_DIA_max_{feature_suffix}2\") for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff2\").filter((pl.col(\"text\").str.contains(c))).min().alias(f\"{c}_DIA_min_{feature_suffix}2\") for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff2\").filter((pl.col(\"text\").str.contains(c))).std().alias(f\"{c}_DIA_std_{feature_suffix}2\") for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff2\").filter((pl.col(\"text\").str.contains(c))).last().alias(f\"{c}_DIA_last_{feature_suffix}2\") for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff2\").filter((pl.col(\"text\").str.contains(c))).first().alias(f\"{c}_DIA_first_{feature_suffix}2\") for c in DIALOGS],\n\n\n\n        *[pl.col(\"elapsed_time_diff3\").filter(pl.col(\"event_name\")==c).mean().alias(f\"{c}_ET_mean_{feature_suffix}3\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff3\").filter(pl.col(\"event_name\")==c).max().alias(f\"{c}_ET_max_{feature_suffix}3\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff3\").filter(pl.col(\"event_name\")==c).min().alias(f\"{c}_ET_min_{feature_suffix}3\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff3\").filter(pl.col(\"event_name\")==c).std().alias(f\"{c}_ET_std_{feature_suffix}3\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff3\").filter(pl.col(\"event_name\")==c).last().alias(f\"{c}_ET_last_{feature_suffix}3\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff3\").filter(pl.col(\"event_name\")==c).first().alias(f\"{c}_ET_first_{feature_suffix}3\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff3\").filter(pl.col(\"name\")==c).mean().alias(f\"{c}_ET_mean_{feature_suffix}3\") for c in name_feature],\n        *[pl.col(\"elapsed_time_diff3\").filter(pl.col(\"name\")==c).max().alias(f\"{c}_ET_max_{feature_suffix}3\") for c in name_feature],\n        *[pl.col(\"elapsed_time_diff3\").filter(pl.col(\"name\")==c).min().alias(f\"{c}_ET_min_{feature_suffix}3\") for c in name_feature],\n        *[pl.col(\"elapsed_time_diff3\").filter(pl.col(\"name\")==c).std().alias(f\"{c}_ET_std_{feature_suffix}3\") for c in name_feature],\n        *[pl.col(\"elapsed_time_diff3\").filter(pl.col(\"name\")==c).last().alias(f\"{c}_ET_last_{feature_suffix}3\") for c in name_feature],\n        *[pl.col(\"elapsed_time_diff3\").filter(pl.col(\"name\")==c).first().alias(f\"{c}_ET_first_{feature_suffix}3\") for c in name_feature],\n\n        *[pl.col(\"elapsed_time_diff3\").filter((pl.col(\"text\").str.contains(c))).mean().alias(f\"{c}_DIA_mean_{feature_suffix}3\") for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff3\").filter((pl.col(\"text\").str.contains(c))).max().alias(f\"{c}_DIA_max_{feature_suffix}3\") for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff3\").filter((pl.col(\"text\").str.contains(c))).min().alias(f\"{c}_DIA_min_{feature_suffix}3\") for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff3\").filter((pl.col(\"text\").str.contains(c))).std().alias(f\"{c}_DIA_std_{feature_suffix}3\") for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff3\").filter((pl.col(\"text\").str.contains(c))).last().alias(f\"{c}_DIA_last_{feature_suffix}3\") for c in DIALOGS],\n        *[pl.col(\"elapsed_time_diff3\").filter((pl.col(\"text\").str.contains(c))).first().alias(f\"{c}_DIA_first_{feature_suffix}3\") for c in DIALOGS],\n\n        *[pl.col(\"screen_coor_x_diff\").filter(pl.col(\"event_name\")==c).mean().alias(f\"{c}_ET_mean_{feature_suffix}x\") for c in event_name_feature],\n        *[pl.col(\"screen_coor_x_diff\").filter(pl.col(\"event_name\")==c).max().alias(f\"{c}_ET_max_{feature_suffix}x\") for c in event_name_feature],\n        *[pl.col(\"screen_coor_x_diff\").filter(pl.col(\"event_name\")==c).min().alias(f\"{c}_ET_min_{feature_suffix}x\") for c in event_name_feature],\n        *[pl.col(\"screen_coor_x_diff\").filter(pl.col(\"event_name\")==c).std().alias(f\"{c}_ET_std_{feature_suffix}x\") for c in event_name_feature],\n        *[pl.col(\"screen_coor_x_diff\").filter(pl.col(\"event_name\")==c).last().alias(f\"{c}_ET_last_{feature_suffix}x\") for c in event_name_feature],\n        *[pl.col(\"screen_coor_x_diff\").filter(pl.col(\"event_name\")==c).first().alias(f\"{c}_ET_first_{feature_suffix}x\") for c in event_name_feature],\n        *[pl.col(\"screen_coor_x_diff\").filter(pl.col(\"name\")==c).mean().alias(f\"{c}_ET_mean_{feature_suffix}x\") for c in name_feature],\n        *[pl.col(\"screen_coor_x_diff\").filter(pl.col(\"name\")==c).max().alias(f\"{c}_ET_max_{feature_suffix}x\") for c in name_feature],\n        *[pl.col(\"screen_coor_x_diff\").filter(pl.col(\"name\")==c).min().alias(f\"{c}_ET_min_{feature_suffix}x\") for c in name_feature],\n        *[pl.col(\"screen_coor_x_diff\").filter(pl.col(\"name\")==c).std().alias(f\"{c}_ET_std_{feature_suffix}x\") for c in name_feature],\n        *[pl.col(\"screen_coor_x_diff\").filter(pl.col(\"name\")==c).last().alias(f\"{c}_ET_last_{feature_suffix}x\") for c in name_feature],\n        *[pl.col(\"screen_coor_x_diff\").filter(pl.col(\"name\")==c).first().alias(f\"{c}_ET_first_{feature_suffix}x\") for c in name_feature],\n\n\n\n\n\n        *[pl.col(\"screen_coor_y_diff\").filter(pl.col(\"event_name\")==c).mean().alias(f\"{c}_ET_mean_{feature_suffix}y\") for c in event_name_feature],\n        *[pl.col(\"screen_coor_y_diff\").filter(pl.col(\"event_name\")==c).max().alias(f\"{c}_ET_max_{feature_suffix}y\") for c in event_name_feature],\n        *[pl.col(\"screen_coor_y_diff\").filter(pl.col(\"event_name\")==c).min().alias(f\"{c}_ET_min_{feature_suffix}y\") for c in event_name_feature],\n        *[pl.col(\"screen_coor_y_diff\").filter(pl.col(\"event_name\")==c).std().alias(f\"{c}_ET_std_{feature_suffix}y\") for c in event_name_feature],\n        *[pl.col(\"screen_coor_y_diff\").filter(pl.col(\"event_name\")==c).last().alias(f\"{c}_ET_last_{feature_suffix}y\") for c in event_name_feature],\n        *[pl.col(\"screen_coor_y_diff\").filter(pl.col(\"event_name\")==c).first().alias(f\"{c}_ET_first_{feature_suffix}y\") for c in event_name_feature],\n        *[pl.col(\"screen_coor_y_diff\").filter(pl.col(\"name\")==c).mean().alias(f\"{c}_ET_mean_{feature_suffix}y\") for c in name_feature],\n        *[pl.col(\"screen_coor_y_diff\").filter(pl.col(\"name\")==c).max().alias(f\"{c}_ET_max_{feature_suffix}y\") for c in name_feature],\n        *[pl.col(\"screen_coor_y_diff\").filter(pl.col(\"name\")==c).min().alias(f\"{c}_ET_min_{feature_suffix}y\") for c in name_feature],\n        *[pl.col(\"screen_coor_y_diff\").filter(pl.col(\"name\")==c).std().alias(f\"{c}_ET_std_{feature_suffix}y\") for c in name_feature],\n        *[pl.col(\"screen_coor_y_diff\").filter(pl.col(\"name\")==c).last().alias(f\"{c}_ET_last_{feature_suffix}y\") for c in name_feature],\n        *[pl.col(\"screen_coor_y_diff\").filter(pl.col(\"name\")==c).first().alias(f\"{c}_ET_first_{feature_suffix}y\") for c in name_feature],\n\n\n        *[pl.col(\"hover_duration\").filter(pl.col(\"event_name\")==c).mean().alias(f\"{c}_ET_mean_{feature_suffix}4\") for c in event_name_feature],\n        *[pl.col(\"hover_duration\").filter(pl.col(\"event_name\")==c).max().alias(f\"{c}_ET_max_{feature_suffix}4\") for c in event_name_feature],\n        *[pl.col(\"hover_duration\").filter(pl.col(\"event_name\")==c).min().alias(f\"{c}_ET_min_{feature_suffix}4\") for c in event_name_feature],\n        *[pl.col(\"hover_duration\").filter(pl.col(\"event_name\")==c).std().alias(f\"{c}_ET_std_{feature_suffix}4\") for c in event_name_feature],\n        *[pl.col(\"hover_duration\").filter(pl.col(\"event_name\")==c).last().alias(f\"{c}_ET_last_{feature_suffix}4\") for c in event_name_feature],\n        *[pl.col(\"hover_duration\").filter(pl.col(\"event_name\")==c).first().alias(f\"{c}_ET_first_{feature_suffix}4\") for c in event_name_feature],\n        *[pl.col(\"hover_duration\").filter(pl.col(\"name\")==c).mean().alias(f\"{c}_ET_mean_{feature_suffix}4\") for c in name_feature],\n        *[pl.col(\"hover_duration\").filter(pl.col(\"name\")==c).max().alias(f\"{c}_ET_max_{feature_suffix}4\") for c in name_feature],\n        *[pl.col(\"hover_duration\").filter(pl.col(\"name\")==c).min().alias(f\"{c}_ET_min_{feature_suffix}4\") for c in name_feature],\n        *[pl.col(\"hover_duration\").filter(pl.col(\"name\")==c).std().alias(f\"{c}_ET_std_{feature_suffix}4\") for c in name_feature],\n        *[pl.col(\"hover_duration\").filter(pl.col(\"name\")==c).last().alias(f\"{c}_ET_last_{feature_suffix}4\") for c in name_feature],\n        *[pl.col(\"hover_duration\").filter(pl.col(\"name\")==c).first().alias(f\"{c}_ET_first_{feature_suffix}4\") for c in name_feature],\n\n        *[pl.col(\"event_name\").filter((pl.col(\"fqid\")==c) & (pl.col('event_name')== 'navigate_click') ).count().alias(f\"{c}_ET_count_{feature_suffix}fe1\") for c in FQID],\n        *[pl.col(\"event_name\").filter((pl.col(\"fqid\")==c) & (pl.col('event_name')== 'observation_click') ).count().alias(f\"{c}_ET_count_{feature_suffix}fe2\") for c in FQID],\n        *[pl.col(\"event_name\").filter((pl.col(\"fqid\")==c) & (pl.col('event_name')== 'map_click') ).count().alias(f\"{c}_ET_count_{feature_suffix}fe3\") for c in FQID],\n        *[pl.col(\"event_name\").filter((pl.col(\"fqid\")==c) & (pl.col('event_name')== 'cutscene_click') ).count().alias(f\"{c}_ET_count_{feature_suffix}fe4\") for c in FQID],\n        *[pl.col(\"event_name\").filter((pl.col(\"fqid\")==c) & (pl.col('event_name')== 'person_click') ).count().alias(f\"{c}_ET_count_{feature_suffix}fe5\") for c in FQID],\n        *[pl.col(\"event_name\").filter((pl.col(\"fqid\")==c) & (pl.col('event_name')== 'object_click') ).count().alias(f\"{c}_ET_count_{feature_suffix}fe6\") for c in FQID],\n         \n    ]\n    \n    df = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\").fill_null(-1)\n\n    if feature_suffix == '04':\n        aggs = aggs_04\n    if feature_suffix == '512':\n        aggs = aggs_512\n    if feature_suffix == '1322':\n        aggs = aggs_1322 \n    #aggs = aggs+adddistanceFeats(feature_suffix)\n    tmp = x.groupby(['session_id'], maintain_order=True).agg(aggs).sort(\"session_id\")  \n    df = df.join(tmp, on=\"session_id\", how='left')\n    \n    # 统计notebook_click操作中，每次开启、关闭的时间\n    open = x.filter((pl.col(\"event_name\") == \"notebook_click\") & (pl.col(\"name\") == \"open\")).select(pl.col([\"session_id\",\"elapsed_time\"]))\n    open.columns = [\"session_id\", \"start\"]\n    close = x.filter((pl.col(\"event_name\") == \"notebook_click\") & (pl.col(\"name\") == \"close\")).select(pl.col([\"session_id\",\"elapsed_time\"]))\n    close.columns = [\"session_id\", \"end\"]\n    merged = open.join(close, on=\"session_id\")\n    merged = merged.with_columns((pl.col(\"end\") - pl.col('start')).alias(\"open_close_gap\"))\n    tmp = merged.groupby([\"session_id\"], maintain_order=True).agg([\n        pl.col(\"open_close_gap\").count().alias(f\"open_close_count_{feature_suffix}\"),\n        pl.col('open_close_gap').sum().alias(f\"open_close_ET_sum_{feature_suffix}\"),\n        pl.col('open_close_gap').mean().alias(f\"open_close_ET_mean_{feature_suffix}\"),\n        pl.col('open_close_gap').max().alias(f\"open_close_ET_max_{feature_suffix}\"),\n        # pl.col('open_close_gap').min().alias(\"open_close_ET_min\"),\n    ]).sort(\"session_id\")\n    df = df.join(tmp, on=\"session_id\", how='left')\n\n    \n    # 不同group之间不同的信息\n    if grp == '0-4':\n        aggs = [\n            # 每个level所花时间的方差、平均值、最小值、最大值、总时间\n            # *[pl.col(\"level\").filter(pl.col(\"level\") == l).count().alias(f\"level{l}_counts{feature_suffix}\") for l in range(0,5)],\n            *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == l).std().alias(f\"level{l}_ET_std_{feature_suffix}\") for l in range(0,5)],\n            *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == l).mean().alias(f\"level{l}_ET_mean_{feature_suffix}\") for l in range(0,5)],\n            # *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == l).max().alias(f\"level{l}_ET_max_{feature_suffix}\") for l in range(0,5)],\n            # *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == l).min().alias(f\"level{l}_ET_min_{feature_suffix}\") for l in range(0,5)],\n            *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == l).sum().alias(f\"level{l}_ET_sum_{feature_suffix}\") for l in range(0,5)],\n            # *[pl.col(\"text\").filter(pl.col(\"level\") == l).n_unique().alias(f\"level{l}_text_unique_{feature_suffix}\") for l in range(0,5)],\n            # 是否统计上分位数？\n        ]\n        tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n        df = df.join(tmp, on=\"session_id\", how='left')\n    elif grp == '5-12':\n        aggs = [\n            # 每个level所花时间的方差、平均值、最小值、最大值、总时间\n            # *[pl.col(\"level\").filter(pl.col(\"level\") == l).count().alias(f\"level{l}_counts{feature_suffix}\") for l in range(5,13)],\n            # *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == l).std().alias(f\"level{l}_ET_std_{feature_suffix}\") for l in range(5,13)],\n            *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == l).mean().alias(f\"level{l}_ET_mean_{feature_suffix}\") for l in range(5,13)],\n            # *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == l).max().alias(f\"level{l}_ET_max_{feature_suffix}\") for l in range(5,13)],\n            # *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == l).min().alias(f\"level{l}_ET_min_{feature_suffix}\") for l in range(5,13)],\n            *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == l).sum().alias(f\"level{l}_ET_sum_{feature_suffix}\") for l in range(5,13)],\n            # *[pl.col(\"text\").filter(pl.col(\"level\") == l).n_unique().alias(f\"level{l}_text_unique_{feature_suffix}\") for l in range(5,13)],\n            # 是否统计上分位数？\n        ]\n        tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n        df = df.join(tmp, on=\"session_id\", how='left')\n    elif grp == '13-22':\n        aggs = [\n            # 每个level所花时间的方差、平均值、最小值、最大值、总时间\n            # *[pl.col(\"level\").filter(pl.col(\"level\") == l).count().alias(f\"level{l}_counts{feature_suffix}\") for l in range(13,23)],\n            # *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == l).std().alias(f\"level{l}_ET_std_{feature_suffix}\") for l in range(13,23)],\n            *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == l).mean().alias(f\"level{l}_ET_mean_{feature_suffix}\") for l in range(13,23)],\n            # *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == l).max().alias(f\"level{l}_ET_max_{feature_suffix}\") for l in range(13,23)],\n            # *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == l).min().alias(f\"level{l}_ET_min_{feature_suffix}\") for l in range(13,23)],\n            *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"level\") == l).sum().alias(f\"level{l}_ET_sum_{feature_suffix}\") for l in range(13,23)],\n            # *[pl.col(\"text\").filter(pl.col(\"level\") == l).n_unique().alias(f\"level{l}_text_unique_{feature_suffix}\") for l in range(5,13)],\n            # 是否统计上分位数？\n        ]\n        tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n        df = df.join(tmp, on=\"session_id\", how='left')\n\n    if use_extra:\n        if grp == '0-4':\n            DIALOGS_04 = ['shirt','basketball','slip']\n            aggs = [\n                pl.col(\"elapsed_time\")\n                    .filter((pl.col(\"text\")==\"Now where did I put my notebook?\")|(pl.col(\"fqid\")=='notebook')|(pl.col(\"event_name\")=='notebook_click')).apply(lambda s: s.max()-s.min()).alias(\"notebook_duration\"),\n                pl.col(\"index\")\n                    .filter((pl.col(\"text\")==\"Now where did I put my notebook?\")|(pl.col(\"fqid\")=='notebook')|(pl.col(\"event_name\")=='notebook_click')).apply(lambda s: s.max()-s.min()).alias(\"notebook_indexCount\"),\n\n               pl.col(\"elapsed_time\")\n                    .filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='gramps'))|(pl.col(\"text_fqid\")==\"tunic.historicalsociety.collection.gramps.found\"))\n                    .apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(\"gramps_duration\"),\n                pl.col(\"index\")\n                    .filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='gramps'))|(pl.col(\"text_fqid\")==\"tunic.historicalsociety.collection.gramps.found\"))\n                    .apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(\"gramps_indexCount\"),\n\n                    \n                pl.col(\"index\")\n                    .filter((pl.col(\"text\").str.contains('!'))&(pl.col(\"event_name\")=='person_click'))\n                    .apply(lambda s: s.max()-s.min() ).alias(f\"person_get_duration_{feature_suffix}\"),\n                pl.col(\"elapsed_time\")\n                    .filter((pl.col(\"text\").str.contains('!'))&(pl.col(\"event_name\")=='person_click'))\n                    .apply(lambda s: s.max()-s.min()).alias(f\"person_get_indexCount_{feature_suffix}\"),\n\n                \n                pl.col(\"index\")\n                    .filter((pl.col(\"fqid\").str.contains('tunic'))&(pl.col(\"event_name\")=='navigate_click'))\n                    .apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(\"tunic_get_duration\"),\n                pl.col(\"elapsed_time\")\n                    .filter((pl.col(\"fqid\").str.contains('tunic'))&(pl.col(\"event_name\")=='navigate_click'))\n                    .apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(\"tunic_get_indexCount\"),\n\n\n                *[pl.col(\"index\")\n                    .filter((pl.col(\"text\").str.contains(c)))\n                    .apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(f\"{c}_person_get_indexCount\") for c in DIALOGS_04] ,\n                \n                *[pl.col(\"elapsed_time\")\n                    .filter((pl.col(\"text\").str.contains(c)))\n                    .apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(f\"{c}_person_get_duration\") for c in DIALOGS_04],\n                \n                *[pl.col(\"elapsed_time_diff\").filter((pl.col(\"text\").str.contains(c))).mean().alias(f\"{c}_DIA_mean_{feature_suffix}\") for c in DIALOGS_04],\n                *[pl.col(\"elapsed_time_diff\").filter((pl.col(\"text\").str.contains(c))).max().alias(f\"{c}_DIA_max_{feature_suffix}\") for c in DIALOGS_04],\n                *[pl.col(\"elapsed_time_diff\").filter((pl.col(\"text\").str.contains(c))).min().alias(f\"{c}_DIA_min_{feature_suffix}\") for c in DIALOGS_04],\n                *[pl.col(\"elapsed_time_diff\").filter((pl.col(\"text\").str.contains(c))).std().alias(f\"{c}_DIA_std_{feature_suffix}\") for c in DIALOGS_04],\n                *[pl.col(\"elapsed_time_diff\").filter((pl.col(\"text\").str.contains(c))).last().alias(f\"{c}_DIA_last_{feature_suffix}\") for c in DIALOGS_04],\n                *[pl.col(\"elapsed_time_diff\").filter((pl.col(\"text\").str.contains(c))).first().alias(f\"{c}_DIA_first_{feature_suffix}\") for c in DIALOGS_04],\n                # Time Weighted\n                *[pl.col(\"tw_score\").filter((pl.col(\"event_name\") == c)).sum().alias(f\"{c}_tw_sum_{feature_suffix}\") for c in ev_feature],\n                *[pl.col(\"tw_score\").filter(pl.col(\"name\")==c).sum().alias(f\"{c}_tw_sum_{feature_suffix}\") for c in name_feature],\n                #*[pl.col(\"tw_score\").filter((pl.col(\"text\").str.contains(c))).sum().alias(f\"{c}_DIA_tw_sum_{feature_suffix}\") for c in DIALOGS],\n                *[pl.col(\"tw_score\").filter(pl.col(\"text\")==c).sum().alias(f\"{c}_TW_sum_{feature_suffix}f\") for c in TEXTS],\n            ]\n            tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n            df = df.join(tmp, on=\"session_id\", how='left')\n        if grp=='5-12':\n            aggs = [\n                pl.col(\"elapsed_time\")\n                        .filter((pl.col(\"text\")==\"Here's the log book.\")|(pl.col(\"fqid\")=='logbook.page.bingo'))\n                        .apply(lambda s: s.max()-s.min()).alias(\"logbook_bingo_duration\"),\n                pl.col(\"index\")\n                    .filter((pl.col(\"text\")==\"Here's the log book.\")|(pl.col(\"fqid\")=='logbook.page.bingo'))\n                    .apply(lambda s: s.max()-s.min()).alias(\"logbook_bingo_indexCount\"),\n                pl.col(\"elapsed_time\")\n                    .filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='reader'))|(pl.col(\"fqid\")==\"reader.paper2.bingo\"))\n                    .apply(lambda s: s.max()-s.min()).alias(\"reader_bingo_duration\"),\n                pl.col(\"index\")\n                    .filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='reader'))|(pl.col(\"fqid\")==\"reader.paper2.bingo\"))\n                    .apply(lambda s: s.max()-s.min()).alias(\"reader_bingo_indexCount\"),\n                pl.col(\"elapsed_time\")\n                    .filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='journals'))|(pl.col(\"fqid\")==\"journals.pic_2.bingo\"))\n                    .apply(lambda s: s.max()-s.min()).alias(\"journals_bingo_duration\"),\n                pl.col(\"index\")\n                    .filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='journals'))|(pl.col(\"fqid\")==\"journals.pic_2.bingo\"))\n                    .apply(lambda s: s.max()-s.min()).alias(\"journals_bingo_indexCount\"),\n                pl.col(\"index\")\n                    .filter((pl.col(\"text\").str.contains('!'))&(pl.col(\"event_name\")=='person_click'))\n                    .apply(lambda s: s.max()-s.min()).alias(f\"person_get_duration_{feature_suffix}\"),\n                pl.col(\"elapsed_time\")\n                    .filter((pl.col(\"text\").str.contains('!'))&(pl.col(\"event_name\")=='person_click'))\n                    .apply(lambda s: s.max()-s.min()).alias(f\"person_get_indexCount_{feature_suffix}\"),\n               \n            ]\n            tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n            df = df.join(tmp, on=\"session_id\", how='left')\n\n        if grp=='13-22':\n            aggs = [\n                pl.col(\"elapsed_time\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='reader_flag'))|(pl.col(\"fqid\")==\"tunic.library.microfiche.reader_flag.paper2.bingo\")).apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(\"reader_flag_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='reader_flag'))|(pl.col(\"fqid\")==\"tunic.library.microfiche.reader_flag.paper2.bingo\")).apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(\"reader_flag_indexCount\"),\n                pl.col(\"elapsed_time\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='journals_flag'))|(pl.col(\"fqid\")==\"journals_flag.pic_0.bingo\")).apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(\"journalsFlag_bingo_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='journals_flag'))|(pl.col(\"fqid\")==\"journals_flag.pic_0.bingo\")).apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(\"journalsFlag_bingo_indexCount\"),\n\n\n                # pl.col(\"elapsed_time\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='wells'))|(pl.col(\"text_fqid\")==\"tunic.historicalsociety.entry.wells.flag\")).apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(\"WellFlag_duration\"),\n                # pl.col(\"index\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='wells'))|(pl.col(\"text_fqid\")==\"tunic.historicalsociety.entry.wells.flag\")).apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(\"WellFlag_indexCount\"),\n\n\n                pl.col(\"index\")\n                    .filter((pl.col(\"text\").str.contains('!'))&(pl.col(\"event_name\")=='person_click'))\n                    .apply(lambda s: s.max()-s.min()).alias(f\"person_get_duration_{feature_suffix}\"),\n                pl.col(\"elapsed_time\")\n                    .filter((pl.col(\"text\").str.contains('!'))&(pl.col(\"event_name\")=='person_click'))\n                    .apply(lambda s: s.max()-s.min()).alias(f\"person_get_indexCount_{feature_suffix}\"),\n\n            ]\n            tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n            df = df.join(tmp, on=\"session_id\", how='left')\n    df_out = df.to_pandas()\n    return df\n\ndef time_feature(train):\n    #train[\"year\"] = train[\"session_id\"].apply(lambda x: int(str(x)[:2])).astype(np.uint8)\n    #train[\"month\"] = train[\"session_id\"].apply(lambda x: int(str(x)[2:4])+1).astype(np.uint8)\n    train[\"day\"] = train[\"session_id\"].apply(lambda x: int(str(x)[4:6])).astype(np.uint8)\n    train[\"hour\"] = train[\"session_id\"].apply(lambda x: int(str(x)[6:8])).astype(np.uint8)\n    train[\"minute\"] = train[\"session_id\"].apply(lambda x: int(str(x)[8:10])).astype(np.uint8)  \n    #print(\"Computing and encoding\")\n    return train","metadata":{"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import shutil \n# shutil.copyfile('./working/fe_v2.py', f/working/{CFG.version}/fe_v2.py\") ","metadata":{"execution":{"iopub.execute_input":"2023-06-25T01:28:14.883163Z","iopub.status.busy":"2023-06-25T01:28:14.882803Z","iopub.status.idle":"2023-06-25T01:28:14.894776Z","shell.execute_reply":"2023-06-25T01:28:14.893645Z","shell.execute_reply.started":"2023-06-25T01:28:14.883122Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"#%%time\nimport sys\nsys.path.insert(0,f'./working/{CFG.version}/')  \nfrom fe_v2 import feature_engineer,time_feature\n# no need of h1, music , fullscreen\nnot_null_mask = ((pl.col('screen_coor_x').is_not_null()) & (pl.col('screen_coor_y').is_not_null()) & (pl.col('next_x').is_not_null()) & (pl.col('next_y').is_not_null()))\nnot_null_mask_rm = ((pl.col('room_coor_x').is_not_null()) & (pl.col('room_coor_y').is_not_null()) & (pl.col('room_next_x').is_not_null()) & (pl.col('room_next_y').is_not_null()))\n\ncolumns = [\n    pl.col(\"page\").cast(pl.Float32),\n    ((pl.col(\"elapsed_time\") - pl.col(\"elapsed_time\").shift(1)).fill_null(0).clip(0, 1e9).over([\"session_id\", \"level\"]) .alias(\"elapsed_time_diff\")),\n    # Combine Room and Screen \n    ((pl.col(\"screen_coor_x\") - pl.col(\"screen_coor_x\").shift(1)) .abs().over([\"session_id\", \"level\"]) .alias('screen_coor_x_diff') ),\n    ((pl.col(\"screen_coor_y\") - pl.col(\"screen_coor_y\").shift(1)).abs().over([\"session_id\", \"level\"]) .alias('screen_coor_y_diff') ),\n    #((pl.col(\"room_coor_x\") - pl.col(\"room_coor_x\").shift(1)).abs().over([\"session_id\", \"level\"]) .alias('room_coor_x_diff') ),\n    #((pl.col(\"room_coor_y\") - pl.col(\"room_coor_y\").shift(1)) .abs().over([\"session_id\", \"level\"]) .alias('room_coor_y_diff') ),\n    \n    ((pl.col(\"screen_coor_x\").shift_and_fill(-1, fill_value=None)).abs().over([\"session_id\", \"level\"]).alias('next_x')),\n    ((pl.col(\"screen_coor_y\").shift_and_fill(-1, fill_value=None)).abs().over([\"session_id\", \"level\"]).alias('next_y')), \n    ((pl.col(\"room_coor_x\").shift_and_fill(-1, fill_value=None)).alias('room_next_x')),\n    ((pl.col(\"room_coor_y\").shift_and_fill(-1, fill_value=None)).alias('room_next_y')), \n    \n    ## W2vec Need to make this faster \n    #((pl.col(\"fqid\").fill_null(\"fqid_None\").apply(lambda x: np.mean(w2vec.wv[x]))).alias('fqid_vec_mean')),\n    #((pl.col(\"fqid\").fill_null(\"fqid_None\").apply(lambda x: np.max(w2vec.wv[x]))).alias('fqid_vec_max')),\n    #((pl.col(\"fqid\").fill_null(\"fqid_None\").apply(lambda x: np.min(w2vec.wv[x]))) .alias('fqid_vec_min')),\n    pl.col(\"fqid\").fill_null(\"fqid_None\"), \n    pl.col(\"text_fqid\").fill_null(\"text_fqid_None\"),\n    #((pl.col(\"event_name\")+pl.col(\"name\")).alias(\"event_name_comb\")), \n    ((pl.col(\"elapsed_time\").diff(n=5)) .fill_null(0).abs().over([\"session_id\", \"level_group\"]).alias(\"elapsed_time_diff2\")),\n    ((pl.col(\"elapsed_time\").diff(n=10)) .fill_null(0).abs().over([\"session_id\", \"level_group\"]).alias(\"elapsed_time_diff3\")),\n\n] \nif CFG.GEN_FEAT:\n    df = pl.read_parquet(CFG.train)\n\n    print(f\"Memory usage of dataframe is {round(df.estimated_size('mb'), 2)} MB\") \n    df = df.with_columns(columns) \n    ## Co-ordinate features \n    df = df.with_columns([\n        #( np.exp(-decay_factor *pl.col('elapsed_time_diff'))).fill_null(0).clip(0, 1e9).over([\"session_id\", \"level\"]) .alias(\"time_decay\"), \n        ((pl.col(\"elapsed_time_diff\") - pl.first( \"elapsed_time_diff\"))/(pl.last(\"elapsed_time_diff\") - pl.first( \"elapsed_time_diff\"))*3 +1)\n        .fill_null(0).over([\"session_id\", \"level\"]).alias(\"tw_score\"),  \n    ])  \n    df=df.with_columns([ \n    ])   \n    \n    print(f\"Memory usage of dataframe is {round(df.estimated_size('mb'), 2)} MB\")  \n\n    df1 = df.filter(pl.col(\"level_group\") == '0-4')\n    df2 = df.filter(pl.col(\"level_group\") == '5-12')\n    df3 = df.filter(pl.col(\"level_group\") == '13-22') \n\n    df1 = feature_engineer(df1, grp='0-4', use_extra=True, feature_suffix='04') \n    print(\"=> df1 Complete\",df1.shape)\n    df2 = feature_engineer(df2, grp='5-12', use_extra=True, feature_suffix='512')\n    print(\"Pre Addition\",df2.shape)\n    \n    df2 = df2.join(df1,on=\"session_id\", how='left')\n    print(\"=> df2 Post addition\",df2.shape)\n    # We have all groups\n    df3 = feature_engineer(df3, grp='13-22', use_extra=True, feature_suffix='1322')\n    \n    df3 = df3.join(df2,on=\"session_id\", how='left')\n    print(\"=> df3 Post addition\",df3.shape)\n\n    # Merge old features\n    df1 = df1.to_pandas()\n    df2 = df2.to_pandas()\n    df3 = df3.to_pandas()\n    \n    if CFG.time_feat:\n        df1 = time_feature(df1)\n        df2 = time_feature(df2)\n        df3 = time_feature(df3) \n    \n    df1 = df1.set_index('session_id')\n    df2 = df2.set_index('session_id')\n    df3 = df3.set_index('session_id')\n\n    df1.to_parquet(\"df1.pq\")\n    df2.to_parquet(\"df2.pq\")\n    df3.to_parquet(\"df3.pq\")\n    del df1,df2,df3;gc.collect()\n    print(\"😋😋😋😋 ==== FE DONE ==== 😋😋😋😋\")","metadata":{"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os,gc\nfrom tqdm import tqdm \nimport pandas as pd\nimport numpy as np \n\ndf1=pd.read_parquet(\"df1.pq\")\nnull1 = df1.isnull().sum().sort_values(ascending=False)/len(df1)\ndrop1 = list(null1[null1 > 0.9].index)\nfor col in tqdm(df1.columns):\n    if df1[col].nunique() == 1:\n        drop1.append(col)\ndf1.drop(drop1,inplace=True,axis=1)      \ndel null1\n_=gc.collect()\n\ndf2=pd.read_parquet(\"df2.pq\")\nnull2 = df2.isnull().sum().sort_values(ascending=False)/len(df2)\ndrop2 = list(null2[null2 > 0.9].index)\nfor col in tqdm(df2.columns):\n    if df2[col].nunique() == 1:\n        drop2.append(col)\ndf2.drop(drop2,inplace=True,axis=1)     \ndel null2\n_=gc.collect()\n\ndf3=pd.read_parquet(\"df3.pq\")\nnull3 = df3.isnull().sum().sort_values(ascending=False)/len(df3)\ndrop3 = list(null3[null3 > 0.9].index)\nprint(len(drop1), len(drop2), len(drop3))   \nfor col in tqdm(df3.columns):\n    if df3[col].nunique() == 1:\n        #print(col)\n        drop3.append(col)\ndf3.drop(drop3,inplace=True,axis=1)     \ndel null3\n_=gc.collect()\nFEATURES1 = [c for c in df1.columns if c not in drop1+['level_group']]\nFEATURES2 = [c for c in df2.columns if c not in drop2+['level_group']]\nFEATURES3 = [c for c in df3.columns if c not in drop3+['level_group']]\nprint('We will train with', len(FEATURES1), len(FEATURES2), len(FEATURES3), 'features')\nALL_USERS = df1.index.unique()\nprint('We will train with', len(ALL_USERS), 'users info')\n\n\n# We will train with 2329 6199 10870 features\n# With Recaps -> We will train with 2343 6236 10930 features\n# More agg stats We will train with 2354 6258 10963 features\n# We will train with 2443 6313 10984 features\n","metadata":{"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TRAIN","metadata":{}},{"cell_type":"code","source":"feature_importance_df = pd.DataFrame()\nmodels = {}\nresults = [[[], []] for _ in range(18)]\nbest_ntree_dict= defaultdict(list)\nxgb_params = {\n    'n_estimators':1000,\n    'booster': 'gbtree',\n    'tree_method': 'hist',\n    'objective': 'binary:logistic',\n    'eval_metric':'logloss',\n    'learning_rate': 0.015,\n    'alpha': 8,#8,\n    #'lambda':0,\n    'max_depth': 4,#6,#5,#4,#3, 4, 5, 6, 7, 8\n    'subsample':0.8,\n    'colsample_bytree': 0.5,#0.6,# 0.2, 0.3, 0.4, 0.5, 0.6\n    'seed': CFG.seed, \n    'early_stopping_rounds': 90,\n    'tree_method':'gpu_hist',\n    'gpu_id':0\n    } ","metadata":{"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS, columns=[f'meta_{i}' for i in range(1, 19)])\n\ndef train(cfg,results=results,\n          models=models,\n          feature_importance_df=feature_importance_df,\n          best_ntree_dict=best_ntree_dict,\n          oof_m=oof):\n    gkf = GroupKFold(n_splits=cfg.n_folds)  \n    # Meta Feature from KiKi\n    historical_meta2 = defaultdict(list)\n    all_Q_feature_importance_df = pd.DataFrame()\n    for q in tqdm(range(1, 19)):\n        if q <= 3:\n            grp = '0-4'\n            df = df1\n            FEATURES = FEATURES1.copy()\n        elif q <= 13:\n            grp = '5-12'\n            df = df2\n            FEATURES = FEATURES2.copy()\n        elif q <= 22:\n            grp = '13-22'\n            df = df3\n            FEATURES = FEATURES3.copy()\n        #print(f\"Dropping features [{len(bad_feat_bag[q])}]\")    \n        #FEATURES = [c for c in FEATURES if c not in bad_feat_bag[q]]  \n        if q-1 >0:\n            meta_name = []\n            for tt in range(1,q): \n                df[f'meta_{tt}'] = oof_m[f'meta_{tt}']\n                FEATURES.append(f'meta_{tt}')     \n                meta_name.append(f'meta_{tt}')\n\n        print(\"#\"*25)\n        print(f'question{q}, with [{len(FEATURES)}] features')\n        print('#'*25)\n        \n        feature_importance_df = pd.DataFrame()\n        best_ntree = [] \n        scr = []\n        for fold, (train_idx, valid_idx) in enumerate(gkf.split(X=df, groups=df.index)):\n            if CFG.tune & fold >0:\n                break\n            # TRAIN DATA\n            train_x = df.iloc[train_idx]\n            train_users = train_x.index.values\n            train_y = targets.loc[targets.q == q].set_index('session').loc[train_users]\n            # VALID DATA\n            valid_x = df.iloc[valid_idx]\n            valid_users = valid_x.index.values\n            valid_y = targets.loc[targets.q == q].set_index('session').loc[valid_users]\n            \n            if CFG.model == \"cat\":\n                train_pool = Pool(train_x[FEATURES].astype('float32'), train_y['correct'])\n                valid_pool = Pool(valid_x[FEATURES].astype('float32'), valid_y['correct'])\n\n            if os.path.exists(f'fold{fold}_q{q}.cbm'):#CFG.INFER:\n                model = XGBClassifier()\n                model.load_model(f'fold{fold}_q{q}.cbm')\n            else:    \n                model = XGBClassifier(**xgb_params) if CFG.model == \"xgb\" else CatBoostClassifier(**cat_params) \n                model = model.fit(train_x[FEATURES].astype('float32'), train_y['correct'], \n                                  eval_set=[(valid_x[FEATURES].astype('float32'), valid_y['correct'])] ,verbose=0) if CFG.model == \"xgb\" else model.fit(train_pool, eval_set=valid_pool)\n\n            y = valid_y['correct'] if CFG.model == \"xgb\" else valid_pool.get_label()\n            y_hat = model.predict_proba(valid_x[FEATURES].astype('float32'))[:,1]  if CFG.model == \"xgb\"  else model.predict_proba(valid_pool)[:,1]\n            models[(fold, q)] = model \n            bn = model.best_ntree_limit if CFG.model == \"xgb\" else model.get_best_iteration()\n            sc_val = model.evals_result()['validation_0']['logloss'][-1] if CFG.model == \"xgb\" else model.get_evals_result()['validation']['logloss'][-1]\n            best_ntree.append(bn)\n            scr.append(sc_val)\n            print(f'{q}:{fold}({bn}):(score[{sc_val}]), ',end='')\n            \n            if not CFG.INFER:\n                #importance = clf.get_booster().get_score(importance_type='weight')\n                fold_importance_df = pd.DataFrame()\n                fold_importance_df[\"feature\"] = FEATURES\n                fold_importance_df[\"importance\"] = model.feature_importances_\n                fold_importance_df[\"fold\"] = fold + 1\n                feature_importance_df = pd.concat([feature_importance_df, fold_importance_df], axis=0) \n            results[q - 1][0].append(y)\n            results[q - 1][1].append(y_hat)\n            oof_m.loc[valid_users, f'meta_{q}'] = y_hat\n            del train_x, train_y,valid_x,valid_y;gc.collect()\n        \n        if not CFG.INFER:\n            feature_importance_df = feature_importance_df.groupby(['feature'])['importance'].agg(['mean','std']).sort_values(by='mean', ascending=False)\n            feature_importance_df['q']=q\n            all_Q_feature_importance_df = pd.concat([all_Q_feature_importance_df, feature_importance_df], axis=0) \n        display(feature_importance_df.head(10)) \n        print(f\"Average iteration {q} [{np.mean(best_ntree)}] score [{np.mean(scr)}]\")\n        best_ntree_dict[q].append(best_ntree)\n        del FEATURES,df; gc.collect()\n    results = [[np.concatenate(_) for _ in _] for _ in results]  \n    for (fold,q), model in models.items():\n        model.save_model(f'fold{fold}_q{q}.cbm') \n    joblib.dump(best_ntree_dict,Path(CFG.version) / 'best_ntree_dict.pkl')\n    return all_Q_feature_importance_df,oof_m\n\n_=gc.collect()    \nfeature_importance_df,oof = train(CFG)   \n#1(874):(score[0.49809688556391596]), 1(743):(score[0.486485803802552]), 1(843):(score[0.4907284317500984]), 1(850):(score[0.49717602753578505]), 1(986):(score[0.4814082654165373]), \n#Average iteration 1 [859.2] score [0.49077908281377775]\n#Average iteration 2 [507.4] score [0.08609017022211182]\n#Average iteration 3 [552.2] score [0.21335524704587075]\n#Average iteration 4 [748.2] score [0.4108180240329845]\n#Average iteration 5 [790.6] score [0.6080842347030755]\n#Average iteration 6 [610.0] score [0.45935966602646994]\n#Average iteration 7 [604.2] score [0.5159910968229011]\n#Average iteration 8 [342.6] score [0.6428674762226261]\n#Average iteration 9 [430.4] score [0.5135199741292824]\n#Average iteration 10 [484.8] score [0.6310210063616621]\n#Average iteration 11 [359.8] score [0.6114027321120614]\n#Average iteration 12 [358.0] score [0.373328088147413] \n#Average iteration 13 [687.6] score [0.5297076625646107]\n#Average iteration 14 [631.4] score [0.542130956583122]\n#Average iteration 15 [612.0] score [0.6064371015598072]\n#Average iteration 16 [490.4] score [0.5643109925575281]\n#Average iteration 17 [413.8] score [0.6059012737519548]\n# \n\n# With Recap Feat\n\n# Average iteration 1 [857.4] score [0.49091695346255626]\n# Average iteration 2 [523.0] score [0.085898212972193]\n# Average iteration 3 [549.0] score [0.2137386890753537]\n# Average iteration 4 [669.0] score [0.41045078043776034]\n# Average iteration 5 [785.6] score [0.6073916240831319]\n\n#With Duration feat\n#Average iteration 1 [920.8] score [0.4905195060785834]\n#Average iteration 2 [485.0] score [0.08594213437698957]\n#Average iteration 3 [526.2] score [0.21321908949016474]\n#Average iteration 4 [737.2] score [0.41065453992150536]\n\n\n","metadata":{"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof.to_parquet(Path(CFG.version) / 'oof.pq') \n\nfeature_importance_df=feature_importance_df.reset_index()\nfeature_importance_df.columns = [\"feature\",\"mean\",\"std\",'q']\nfeature_importance_df.to_csv(Path(CFG.version) / \"importance.csv\")","metadata":{"execution":{"iopub.execute_input":"2023-06-25T03:31:44.502038Z","iopub.status.busy":"2023-06-25T03:31:44.501658Z","iopub.status.idle":"2023-06-25T03:31:45.600146Z","shell.execute_reply":"2023-06-25T03:31:45.598893Z","shell.execute_reply.started":"2023-06-25T03:31:44.501998Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PUT TRUE LABELS INTO DATAFRAME WITH 18 COLUMNS\noof = pd.read_parquet(Path(CFG.version) / 'oof.pq')\ntrue = oof.copy()\nfor i in range(1, 19): \n    # GET TRUE LABELS\n    tmp = targets.loc[targets.q==i].set_index('session').loc[ALL_USERS]\n    # tmp = targets.loc[targets.q==i].set_index('session').loc[valid_users]\n    true[f'meta_{i}'] = tmp.correct.values\n    \n# FIND BEST THRESHOLD TO CONVERT PROBS INTO 1s AND 0s\nscores = []; thresholds = []\nbest_score_xgb = 0; best_threshold_xgb = 0\n\nfor threshold in np.arange(0.5,0.81,0.001):\n    #print(f'{threshold:.03f}, ',end='')\n    preds = (oof.values.reshape((-1))>threshold).astype('int')\n    m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n    scores.append(m)\n    thresholds.append(threshold)\n    if m>best_score_xgb:\n        best_score_xgb = m\n        best_threshold_xgb = threshold\n        \n# Via Optuna \nimport optuna \noptuna.logging.set_verbosity(optuna.logging.WARNING)\n\ndef optunaOpt(n_trials=20,direction=\"maximize\") : \n    \n    def run(trials):\n        #threshold=trials.suggest_loguniform(\"threshold\", 0.4,0.9)\n        threshold=trials.suggest_uniform(\"threshold\", 0.4,0.9)\n        preds = (oof.values.reshape((-1))>threshold).astype('int')\n        m = f1_score(true.values.reshape((-1)), preds, average='macro')   \n        return m  \n    \n    study = optuna.create_study(sampler=optuna.samplers.TPESampler(seed=CFG.seed),\n                                direction=direction,\n                                study_name=f\"thres-study\")\n    study.optimize(run, n_trials)\n    print('\\n Best Trial:')\n    print(study.best_trial)\n    print('\\n Best value')\n    print(study.best_value)\n    print('\\n Best Threshold Optuna:')\n    print(study.best_params)\n    return study   \n\nstudy = optunaOpt(n_trials=444,direction=\"maximize\")\n\n# PLOT THRESHOLD VS. F1_SCORE\nplt.figure(figsize=(20,5))\nplt.plot(thresholds,scores,'-o',color='blue')\nplt.scatter([best_threshold_xgb], [best_score_xgb], color='blue', s=300, alpha=1)\nplt.xlabel('Threshold',size=14)\nplt.ylabel('Validation F1 Score',size=14)\nplt.title(f'Threshold vs. F1_Score with Best F1_Score = {best_score_xgb:.5f} at Best Threshold = {best_threshold_xgb:.4}',size=18)\nplt.show()    ","metadata":{"execution":{"iopub.execute_input":"2023-06-25T03:31:45.606377Z","iopub.status.busy":"2023-06-25T03:31:45.605996Z","iopub.status.idle":"2023-06-25T03:33:32.335770Z","shell.execute_reply":"2023-06-25T03:33:32.334473Z","shell.execute_reply.started":"2023-06-25T03:31:45.606337Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = (oof.values.reshape((-1))>0.605).astype('int')\nm = f1_score(true.values.reshape((-1)), preds, average='macro') \nprint(f\"0.605 CV {m}\")\npreds = (oof.values.reshape((-1))>0.61).astype('int')\nm = f1_score(true.values.reshape((-1)), preds, average='macro') \nprint(f\"0.61 CV {m}\")\n\nbest_ntree_dict=joblib.load(Path(CFG.version) / 'best_ntree_dict.pkl')\nbest_ntree_dict","metadata":{"execution":{"iopub.execute_input":"2023-06-25T03:33:32.338400Z","iopub.status.busy":"2023-06-25T03:33:32.337662Z","iopub.status.idle":"2023-06-25T03:33:32.625063Z","shell.execute_reply":"2023-06-25T03:33:32.623818Z","shell.execute_reply.started":"2023-06-25T03:33:32.338353Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tune automation ","metadata":{}},{"cell_type":"markdown","source":"# Train On full data ","metadata":{}},{"cell_type":"code","source":"# Iterate and add off preds meta\noof = pd.DataFrame(data=np.zeros((len(ALL_USERS),18)), index=ALL_USERS, columns=[f'meta_{i}' for i in range(1, 19)])\nfor q in tqdm(range(1, 19)):\n    if q <= 3:\n        grp = '0-4'\n        df = df1\n        FEATURES = FEATURES1.copy()\n    elif q <= 13:\n        grp = '5-12'\n        df = df2\n        FEATURES = FEATURES2.copy()\n    elif q <= 22:\n        grp = '13-22'\n        df = df3\n        FEATURES = FEATURES3.copy()\n    #print(f\"Dropping features [{len(bad_feat_bag[q])}]\")    \n    #FEATURES = [c for c in FEATURES if c not in bad_feat_bag[q]]    \n    if q-1 >0:\n        meta_name = []\n        for tt in range(1,q):\n            # print(df.head())\n            df[f'meta_{tt}'] = oof[f'meta_{tt}']\n            FEATURES.append(f'meta_{tt}')\n            meta_name.append(f'meta_{tt}')\n        if CFG.META_AGG:      \n            df['meta_sum'] = df[meta_name].apply(lambda x: x.sum(), axis=1)\n            df['meta_mean'] = df[meta_name].apply(lambda x: x.mean(), axis=1)\n            #df['meta_outliers_count'] = df[meta_name].apply(lambda x: count_outliers(x, threshold_multiplier), axis=1)\n            FEATURES.append('meta_sum')\n            FEATURES.append('meta_mean')\n            # FEATURES.append('meta_std')\n            #FEATURES.append('meta_outliers_count')\n    print(\"#\"*25)\n    iteration = int(np.median(best_ntree_dict[q]) + 1) if CFG.MODE ==\"median\" else int(np.mean(best_ntree_dict[q]) + 1)\n    print(f'question{q}, with{len(FEATURES)}features on full_data iter {iteration}')\n    print('#'*25) \n    xgb_params['early_stopping_rounds']=None\n    xgb_params['n_estimators']=iteration \n    y = targets.loc[targets.q == q].set_index('session') \n    print(f\"y len {y.correct.shape} X {df[FEATURES].shape}\")\n    model = XGBClassifier(**xgb_params) if CFG.model == \"xgb\" else CatBoostClassifier(**cat_params) \n    model = model.fit(df[FEATURES].astype('float32'), y['correct']) if CFG.model == \"xgb\" else model.fit(Pool(df[FEATURES].astype('float32'), y['correct']) )\n    model.save_model(Path(CFG.version) / f'full_data_q{q}.cbm' )  \n    fi = pd.DataFrame()\n    fi[\"feature\"] = FEATURES\n    fi[\"importance\"] = model.feature_importances_\n    fi = fi.sort_values(by='importance', ascending=False)       \n    fi.to_csv(Path(CFG.version) /f\"fe_q{q}.csv\")                        \n    oof.loc[df.index.values, f'meta_{q}'] = model.predict_proba(df[FEATURES].astype('float32'))[:,1]\n    print(f'{q}({model.best_ntree_limit}), ',end='')  ","metadata":{"execution":{"iopub.execute_input":"2023-06-25T03:33:32.628266Z","iopub.status.busy":"2023-06-25T03:33:32.627809Z","iopub.status.idle":"2023-06-25T03:53:59.900748Z","shell.execute_reply":"2023-06-25T03:53:59.899841Z","shell.execute_reply.started":"2023-06-25T03:33:32.628223Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nimportance_dict = {}\nfor t in range(1, 19):\n    if t<=3: \n        importance_dict[str(t)] = FEATURES1\n    elif t<=13: \n        importance_dict[str(t)] = FEATURES2\n    elif t<=22:\n        importance_dict[str(t)] = FEATURES3\n\nf_save = open(Path(CFG.version) / 'importance_dict.pkl', 'wb')\npickle.dump(importance_dict, f_save)\nf_save.close()","metadata":{"execution":{"iopub.execute_input":"2023-06-25T03:53:59.905159Z","iopub.status.busy":"2023-06-25T03:53:59.902191Z","iopub.status.idle":"2023-06-25T03:53:59.917699Z","shell.execute_reply":"2023-06-25T03:53:59.916538Z","shell.execute_reply.started":"2023-06-25T03:53:59.905096Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_importance_df= pd.read_csv(Path(CFG.version) / \"importance.csv\")\nfeature_importance_df.head(10)\n","metadata":{"execution":{"iopub.execute_input":"2023-06-25T03:54:00.002398Z","iopub.status.busy":"2023-06-25T03:54:00.001711Z","iopub.status.idle":"2023-06-25T03:54:00.181175Z","shell.execute_reply":"2023-06-25T03:54:00.180070Z","shell.execute_reply.started":"2023-06-25T03:54:00.002346Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#feature_importance_df[feature_importance_df.feature.str.contains(\"RCP\")].head(100)","metadata":{"execution":{"iopub.execute_input":"2023-06-25T03:54:00.185518Z","iopub.status.busy":"2023-06-25T03:54:00.184992Z","iopub.status.idle":"2023-06-25T03:54:00.189791Z","shell.execute_reply":"2023-06-25T03:54:00.188570Z","shell.execute_reply.started":"2023-06-25T03:54:00.185485Z"}},"execution_count":null,"outputs":[]}]}