{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This Notebook is based on the job of PR0FESS0ROP \nhttps://www.kaggle.com/code/pr0fess0rop/simple-xgb-model","metadata":{}},{"cell_type":"markdown","source":"# Libraries","metadata":{}},{"cell_type":"code","source":"# !pip install polars","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:04.399196Z","iopub.execute_input":"2023-04-05T15:39:04.399731Z","iopub.status.idle":"2023-04-05T15:39:04.406276Z","shell.execute_reply.started":"2023-04-05T15:39:04.399674Z","shell.execute_reply":"2023-04-05T15:39:04.404954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install pyarrow","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:04.407989Z","iopub.execute_input":"2023-04-05T15:39:04.408962Z","iopub.status.idle":"2023-04-05T15:39:04.425288Z","shell.execute_reply.started":"2023-04-05T15:39:04.408919Z","shell.execute_reply":"2023-04-05T15:39:04.423983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install xgboost","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:04.427199Z","iopub.execute_input":"2023-04-05T15:39:04.427807Z","iopub.status.idle":"2023-04-05T15:39:04.436183Z","shell.execute_reply.started":"2023-04-05T15:39:04.427767Z","shell.execute_reply":"2023-04-05T15:39:04.435222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install lightgbm","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:04.438118Z","iopub.execute_input":"2023-04-05T15:39:04.438714Z","iopub.status.idle":"2023-04-05T15:39:04.447820Z","shell.execute_reply.started":"2023-04-05T15:39:04.438662Z","shell.execute_reply":"2023-04-05T15:39:04.446661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install matplotlib","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:04.450358Z","iopub.execute_input":"2023-04-05T15:39:04.451337Z","iopub.status.idle":"2023-04-05T15:39:04.459629Z","shell.execute_reply.started":"2023-04-05T15:39:04.451284Z","shell.execute_reply":"2023-04-05T15:39:04.458410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install colorama","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:04.461228Z","iopub.execute_input":"2023-04-05T15:39:04.461924Z","iopub.status.idle":"2023-04-05T15:39:04.472001Z","shell.execute_reply.started":"2023-04-05T15:39:04.461883Z","shell.execute_reply":"2023-04-05T15:39:04.470612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reading Training file","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport gc\nimport os\n\nimport pandas as pd\nimport numpy as np\nimport warnings\nimport pickle\nimport polars as pl\n\nfrom collections import defaultdict\nfrom itertools import combinations\nimport pyarrow as pa\n\nfrom xgboost import XGBClassifier\n\nfrom lightgbm import LGBMClassifier\nfrom lightgbm import early_stopping\nfrom lightgbm import log_evaluation\n\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import roc_auc_score, f1_score\n\n\nimport matplotlib.pyplot as plt\nfrom colorama import Fore, Back, Style\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# settings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","tags":[],"execution":{"iopub.status.busy":"2023-04-05T15:39:04.473862Z","iopub.execute_input":"2023-04-05T15:39:04.474510Z","iopub.status.idle":"2023-04-05T15:39:04.488531Z","shell.execute_reply.started":"2023-04-05T15:39:04.474469Z","shell.execute_reply":"2023-04-05T15:39:04.487532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtypes = {\"session_id\": pl.Int64,\n          \"elapsed_time\": pl.Int64,\n          \"event_name\": pl.Categorical,\n          \"name\": pl.Categorical,\n          \"level\": pl.Int8,\n          \"page\": pl.Float32,\n          \"room_coor_x\": pl.Float32,\n          \"room_coor_y\": pl.Float32,\n          \"screen_coor_x\": pl.Float32,\n          \"screen_coor_y\": pl.Float32,\n          \"hover_duration\": pl.Float32,\n          \"text\": pl.Categorical,\n          \"fqid\": pl.Categorical,\n          \"room_fqid\": pl.Categorical,\n          \"text_fqid\": pl.Categorical,\n          \"fullscreen\": pl.Int8,\n          \"hq\": pl.Int8,\n          \"music\": pl.Int8,\n          \"level_group\": pl.Categorical\n          }","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:04.490093Z","iopub.execute_input":"2023-04-05T15:39:04.490672Z","iopub.status.idle":"2023-04-05T15:39:04.502736Z","shell.execute_reply.started":"2023-04-05T15:39:04.490632Z","shell.execute_reply":"2023-04-05T15:39:04.501455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = [\n\n    pl.col(\"page\").cast(pl.Float32),\n    (\n        (pl.col(\"elapsed_time\") - pl.col(\"elapsed_time\").shift(1)) \n         .fill_null(0)\n         .clip(0, 1e9)\n         .over([\"session_id\", \"level_group\"])\n         .alias(\"elapsed_time_diff\")\n    ),\n    (\n        (pl.col(\"screen_coor_x\") - pl.col(\"screen_coor_x\").shift(1)) \n         .abs()\n         .over([\"session_id\", \"level_group\"])\n        .alias(\"location_x_diff\") \n    ),\n    (\n        (pl.col(\"screen_coor_y\") - pl.col(\"screen_coor_y\").shift(1)) \n         .abs()\n         .over([\"session_id\", \"level_group\"])\n        .alias(\"location_y_diff\") \n    )\n]","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:04.504726Z","iopub.execute_input":"2023-04-05T15:39:04.505414Z","iopub.status.idle":"2023-04-05T15:39:04.517445Z","shell.execute_reply.started":"2023-04-05T15:39:04.505369Z","shell.execute_reply":"2023-04-05T15:39:04.516250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pl.StringCache()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:04.520592Z","iopub.execute_input":"2023-04-05T15:39:04.521476Z","iopub.status.idle":"2023-04-05T15:39:04.534854Z","shell.execute_reply.started":"2023-04-05T15:39:04.521433Z","shell.execute_reply":"2023-04-05T15:39:04.533609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Reading from Kaggle Environment","metadata":{}},{"cell_type":"code","source":"%%time\nfiles_status = True\n\ntry:    \n    #train = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\n    train = (pl.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv',dtypes=dtypes)\n                .drop([\"fullscreen\", \"hq\", \"music\"])\n                .with_columns(columns)\n              )\n    targets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\n    #test = pd.read_csv(\"'/kaggle/input/predict-student-performance-from-game-play/test.csv'\")\nexcept OSError as e:\n    print(\"files not found in Kaggle environment: \",e.errno)\n    files_status = False","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:04.536769Z","iopub.execute_input":"2023-04-05T15:39:04.537396Z","iopub.status.idle":"2023-04-05T15:39:44.927497Z","shell.execute_reply.started":"2023-04-05T15:39:04.537357Z","shell.execute_reply":"2023-04-05T15:39:44.926502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Reading from local environment (files folder)","metadata":{}},{"cell_type":"code","source":"%%time\n#If environment is not kaggle notebook, read them from a folder \n\ntry:\n    if files_status == False:\n        train = (pl.read_csv(\"predict-student-performance-from-game-play/train.csv\", dtypes=dtypes)\n                    .drop([\"fullscreen\", \"hq\", \"music\"])\n                    .with_columns(columns)\n                )\n        targets = pd.read_csv('predict-student-performance-from-game-play/train_labels.csv')\nexcept OSError as e:\n    print(\"Files not found\")","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:44.928910Z","iopub.execute_input":"2023-04-05T15:39:44.929496Z","iopub.status.idle":"2023-04-05T15:39:44.937288Z","shell.execute_reply.started":"2023-04-05T15:39:44.929456Z","shell.execute_reply":"2023-04-05T15:39:44.936289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Memory usage of dataframe is {round(train.estimated_size('mb'), 2)} MB\")","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:44.938972Z","iopub.execute_input":"2023-04-05T15:39:44.939641Z","iopub.status.idle":"2023-04-05T15:39:44.953595Z","shell.execute_reply.started":"2023-04-05T15:39:44.939603Z","shell.execute_reply":"2023-04-05T15:39:44.952703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reducing training file","metadata":{}},{"cell_type":"code","source":"def reduce_memory_usage_pl(df, name):\n    \"\"\" Reduce memory usage by polars dataframe {df} with name {name} by changing its data types.\n        Original pandas version of this function: https://www.kaggle.com/code/arjanso/reducing-dataframe-memory-size-by-65 \"\"\"\n    print(f\"Memory usage of dataframe {name} is {round(df.estimated_size('mb'), 2)} MB\")\n    Numeric_Int_types = [pl.Int8,pl.Int16,pl.Int32,pl.Int64]\n    Numeric_Float_types = [pl.Float32,pl.Float64]    \n    for col in df.columns:\n        col_type = df[col].dtype\n        c_min = df[col].min()\n        c_max = df[col].max()\n        if col_type in Numeric_Int_types:\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                df = df.with_columns(df[col].cast(pl.Int8))\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                df = df.with_columns(df[col].cast(pl.Int16))\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                df = df.with_columns(df[col].cast(pl.Int32))\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                df = df.with_columns(df[col].cast(pl.Int64))\n        elif col_type in Numeric_Float_types:\n            if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                df = df.with_columns(df[col].cast(pl.Float32))\n            else:\n                pass\n        elif col_type == pl.Utf8:\n            df = df.with_columns(df[col].cast(pl.Categorical))\n        else:\n            pass\n    \n    print(f\"Memory usage of dataframe {name} became {round(df.estimated_size('mb'), 2)} MB\")\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:44.955409Z","iopub.execute_input":"2023-04-05T15:39:44.956120Z","iopub.status.idle":"2023-04-05T15:39:44.968815Z","shell.execute_reply.started":"2023-04-05T15:39:44.956082Z","shell.execute_reply":"2023-04-05T15:39:44.967873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reducing polar\ntrain = reduce_memory_usage_pl(train, \"train_subset\")","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:44.970497Z","iopub.execute_input":"2023-04-05T15:39:44.971163Z","iopub.status.idle":"2023-04-05T15:39:46.396430Z","shell.execute_reply.started":"2023-04-05T15:39:44.971126Z","shell.execute_reply":"2023-04-05T15:39:46.395516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.columns)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:46.398050Z","iopub.execute_input":"2023-04-05T15:39:46.398752Z","iopub.status.idle":"2023-04-05T15:39:46.404527Z","shell.execute_reply.started":"2023-04-05T15:39:46.398714Z","shell.execute_reply":"2023-04-05T15:39:46.403733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.tail(4)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:46.406044Z","iopub.execute_input":"2023-04-05T15:39:46.406700Z","iopub.status.idle":"2023-04-05T15:39:46.430363Z","shell.execute_reply.started":"2023-04-05T15:39:46.406647Z","shell.execute_reply":"2023-04-05T15:39:46.429085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering for training file","metadata":{}},{"cell_type":"code","source":"CATS = ['event_name', 'name', 'fqid', 'room_fqid', 'text_fqid']\nNUMS = ['page', 'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y',\n        'hover_duration', 'elapsed_time_diff']\n\nname_feature = ['basic', 'undefined', 'close', 'open', 'prev', 'next']\nevent_name_feature = ['cutscene_click', 'person_click', 'navigate_click',\n       'observation_click', 'notification_click', 'object_click',\n       'object_hover', 'map_hover', 'map_click', 'checkpoint',\n       'notebook_click']\n\n# from https://www.kaggle.com/code/leehomhuang/catboost-baseline-with-lots-features-inference :\nfqid_lists = ['worker', 'archivist', 'gramps', 'wells', 'toentry', 'confrontation', 'crane_ranger', 'groupconvo', 'flag_girl', 'tomap', 'tostacks', 'tobasement', 'archivist_glasses', 'boss', 'journals', 'seescratches', 'groupconvo_flag', 'cs', 'teddy', 'expert', 'businesscards', 'ch3start', 'tunic.historicalsociety', 'tofrontdesk', 'savedteddy', 'plaque', 'glasses', 'tunic.drycleaner', 'reader_flag', 'tunic.library', 'tracks', 'tunic.capitol_2', 'trigger_scarf', 'reader', 'directory', 'tunic.capitol_1', 'journals.pic_0.next', 'unlockdoor', 'tunic', 'what_happened', 'tunic.kohlcenter', 'tunic.humanecology', 'colorbook', 'logbook', 'businesscards.card_0.next', 'journals.hub.topics', 'logbook.page.bingo', 'journals.pic_1.next', 'journals_flag', 'reader.paper0.next', 'tracks.hub.deer', 'reader_flag.paper0.next', 'trigger_coffee', 'wellsbadge', 'journals.pic_2.next', 'tomicrofiche', 'journals_flag.pic_0.bingo', 'plaque.face.date', 'notebook', 'tocloset_dirty', 'businesscards.card_bingo.bingo', 'businesscards.card_1.next', 'tunic.wildlife', 'tunic.hub.slip', 'tocage', 'journals.pic_2.bingo', 'tocollectionflag', 'tocollection', 'chap4_finale_c', 'chap2_finale_c', 'lockeddoor', 'journals_flag.hub.topics', 'tunic.capitol_0', 'reader_flag.paper2.bingo', 'photo', 'tunic.flaghouse', 'reader.paper1.next', 'directory.closeup.archivist', 'intro', 'businesscards.card_bingo.next', 'reader.paper2.bingo', 'retirement_letter', 'remove_cup', 'journals_flag.pic_0.next', 'magnify', 'coffee', 'key', 'togrampa', 'reader_flag.paper1.next', 'janitor', 'tohallway', 'chap1_finale', 'report', 'outtolunch', 'journals_flag.hub.topics_old', 'journals_flag.pic_1.next', 'reader.paper2.next', 'chap1_finale_c', 'reader_flag.paper2.next', 'door_block_talk', 'journals_flag.pic_1.bingo', 'journals_flag.pic_2.next', 'journals_flag.pic_2.bingo', 'block_magnify', 'reader.paper0.prev', 'block', 'reader_flag.paper0.prev', 'block_0', 'door_block_clean', 'reader.paper2.prev', 'reader.paper1.prev', 'doorblock', 'tocloset', 'reader_flag.paper2.prev', 'reader_flag.paper1.prev', 'block_tomap2', 'journals_flag.pic_0_old.next', 'journals_flag.pic_1_old.next', 'block_tocollection', 'block_nelson', 'journals_flag.pic_2_old.next', 'block_tomap1', 'block_badge', 'need_glasses', 'block_badge_2', 'fox', 'block_1']\ntext_lists = ['tunic.historicalsociety.cage.confrontation', 'tunic.wildlife.center.crane_ranger.crane', 'tunic.historicalsociety.frontdesk.archivist.newspaper', 'tunic.historicalsociety.entry.groupconvo', 'tunic.wildlife.center.wells.nodeer', 'tunic.historicalsociety.frontdesk.archivist.have_glass', 'tunic.drycleaner.frontdesk.worker.hub', 'tunic.historicalsociety.closet_dirty.gramps.news', 'tunic.humanecology.frontdesk.worker.intro', 'tunic.historicalsociety.frontdesk.archivist_glasses.confrontation', 'tunic.historicalsociety.basement.seescratches', 'tunic.historicalsociety.collection.cs', 'tunic.flaghouse.entry.flag_girl.hello', 'tunic.historicalsociety.collection.gramps.found', 'tunic.historicalsociety.basement.ch3start', 'tunic.historicalsociety.entry.groupconvo_flag', 'tunic.library.frontdesk.worker.hello', 'tunic.library.frontdesk.worker.wells', 'tunic.historicalsociety.collection_flag.gramps.flag', 'tunic.historicalsociety.basement.savedteddy', 'tunic.library.frontdesk.worker.nelson', 'tunic.wildlife.center.expert.removed_cup', 'tunic.library.frontdesk.worker.flag', 'tunic.historicalsociety.frontdesk.archivist.hello', 'tunic.historicalsociety.closet.gramps.intro_0_cs_0', 'tunic.historicalsociety.entry.boss.flag', 'tunic.flaghouse.entry.flag_girl.symbol', 'tunic.historicalsociety.closet_dirty.trigger_scarf', 'tunic.drycleaner.frontdesk.worker.done', 'tunic.historicalsociety.closet_dirty.what_happened', 'tunic.wildlife.center.wells.animals', 'tunic.historicalsociety.closet.teddy.intro_0_cs_0', 'tunic.historicalsociety.cage.glasses.afterteddy', 'tunic.historicalsociety.cage.teddy.trapped', 'tunic.historicalsociety.cage.unlockdoor', 'tunic.historicalsociety.stacks.journals.pic_2.bingo', 'tunic.historicalsociety.entry.wells.flag', 'tunic.humanecology.frontdesk.worker.badger', 'tunic.historicalsociety.stacks.journals_flag.pic_0.bingo', 'tunic.historicalsociety.closet.intro', 'tunic.historicalsociety.closet.retirement_letter.hub', 'tunic.historicalsociety.entry.directory.closeup.archivist', 'tunic.historicalsociety.collection.tunic.slip', 'tunic.kohlcenter.halloffame.plaque.face.date', 'tunic.historicalsociety.closet_dirty.trigger_coffee', 'tunic.drycleaner.frontdesk.logbook.page.bingo', 'tunic.library.microfiche.reader.paper2.bingo', 'tunic.kohlcenter.halloffame.togrampa', 'tunic.capitol_2.hall.boss.haveyougotit', 'tunic.wildlife.center.wells.nodeer_recap', 'tunic.historicalsociety.cage.glasses.beforeteddy', 'tunic.historicalsociety.closet_dirty.gramps.helpclean', 'tunic.wildlife.center.expert.recap', 'tunic.historicalsociety.frontdesk.archivist.have_glass_recap', 'tunic.historicalsociety.stacks.journals_flag.pic_1.bingo', 'tunic.historicalsociety.cage.lockeddoor', 'tunic.historicalsociety.stacks.journals_flag.pic_2.bingo', 'tunic.historicalsociety.collection.gramps.lost', 'tunic.historicalsociety.closet.notebook', 'tunic.historicalsociety.frontdesk.magnify', 'tunic.humanecology.frontdesk.businesscards.card_bingo.bingo', 'tunic.wildlife.center.remove_cup', 'tunic.library.frontdesk.wellsbadge.hub', 'tunic.wildlife.center.tracks.hub.deer', 'tunic.historicalsociety.frontdesk.key', 'tunic.library.microfiche.reader_flag.paper2.bingo', 'tunic.flaghouse.entry.colorbook', 'tunic.wildlife.center.coffee', 'tunic.capitol_1.hall.boss.haveyougotit', 'tunic.historicalsociety.basement.janitor', 'tunic.historicalsociety.collection_flag.gramps.recap', 'tunic.wildlife.center.wells.animals2', 'tunic.flaghouse.entry.flag_girl.symbol_recap', 'tunic.historicalsociety.closet_dirty.photo', 'tunic.historicalsociety.stacks.outtolunch', 'tunic.library.frontdesk.worker.wells_recap', 'tunic.historicalsociety.frontdesk.archivist_glasses.confrontation_recap', 'tunic.capitol_0.hall.boss.talktogramps', 'tunic.historicalsociety.closet.photo', 'tunic.historicalsociety.collection.tunic', 'tunic.historicalsociety.closet.teddy.intro_0_cs_5', 'tunic.historicalsociety.closet_dirty.gramps.archivist', 'tunic.historicalsociety.closet_dirty.door_block_talk', 'tunic.historicalsociety.entry.boss.flag_recap', 'tunic.historicalsociety.frontdesk.archivist.need_glass_0', 'tunic.historicalsociety.entry.wells.talktogramps', 'tunic.historicalsociety.frontdesk.block_magnify', 'tunic.historicalsociety.frontdesk.archivist.foundtheodora', 'tunic.historicalsociety.closet_dirty.gramps.nothing', 'tunic.historicalsociety.closet_dirty.door_block_clean', 'tunic.capitol_1.hall.boss.writeitup', 'tunic.library.frontdesk.worker.nelson_recap', 'tunic.library.frontdesk.worker.hello_short', 'tunic.historicalsociety.stacks.block', 'tunic.historicalsociety.frontdesk.archivist.need_glass_1', 'tunic.historicalsociety.entry.boss.talktogramps', 'tunic.historicalsociety.frontdesk.archivist.newspaper_recap', 'tunic.historicalsociety.entry.wells.flag_recap', 'tunic.drycleaner.frontdesk.worker.done2', 'tunic.library.frontdesk.worker.flag_recap', 'tunic.humanecology.frontdesk.block_0', 'tunic.library.frontdesk.worker.preflag', 'tunic.historicalsociety.basement.gramps.seeyalater', 'tunic.flaghouse.entry.flag_girl.hello_recap', 'tunic.historicalsociety.closet.doorblock', 'tunic.drycleaner.frontdesk.worker.takealook', 'tunic.historicalsociety.basement.gramps.whatdo', 'tunic.library.frontdesk.worker.droppedbadge', 'tunic.historicalsociety.entry.block_tomap2', 'tunic.library.frontdesk.block_nelson', 'tunic.library.microfiche.block_0', 'tunic.historicalsociety.entry.block_tocollection', 'tunic.historicalsociety.entry.block_tomap1', 'tunic.historicalsociety.collection.gramps.look_0', 'tunic.library.frontdesk.block_badge', 'tunic.historicalsociety.cage.need_glasses', 'tunic.library.frontdesk.block_badge_2', 'tunic.kohlcenter.halloffame.block_0', 'tunic.capitol_0.hall.chap1_finale_c', 'tunic.capitol_1.hall.chap2_finale_c', 'tunic.capitol_2.hall.chap4_finale_c', 'tunic.wildlife.center.fox.concern', 'tunic.drycleaner.frontdesk.block_0', 'tunic.historicalsociety.entry.gramps.hub', 'tunic.humanecology.frontdesk.block_1', 'tunic.drycleaner.frontdesk.block_1']\nroom_lists = ['tunic.historicalsociety.entry', 'tunic.wildlife.center', 'tunic.historicalsociety.cage', 'tunic.library.frontdesk', 'tunic.historicalsociety.frontdesk', 'tunic.historicalsociety.stacks', 'tunic.historicalsociety.closet_dirty', 'tunic.humanecology.frontdesk', 'tunic.historicalsociety.basement', 'tunic.kohlcenter.halloffame', 'tunic.library.microfiche', 'tunic.drycleaner.frontdesk', 'tunic.historicalsociety.collection', 'tunic.historicalsociety.closet', 'tunic.flaghouse.entry', 'tunic.historicalsociety.collection_flag', 'tunic.capitol_1.hall', 'tunic.capitol_0.hall', 'tunic.capitol_2.hall']","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:46.432273Z","iopub.execute_input":"2023-04-05T15:39:46.432611Z","iopub.status.idle":"2023-04-05T15:39:46.453451Z","shell.execute_reply.started":"2023-04-05T15:39:46.432580Z","shell.execute_reply":"2023-04-05T15:39:46.451941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer_pl(x, grp, use_extra, feature_suffix):\n        \n    aggs = [\n        pl.col(\"index\").count().alias(f\"session_number_{feature_suffix}\"),\n      \n        *[pl.col(c).drop_nulls().n_unique().alias(f\"{c}_unique_{feature_suffix}\") for c in CATS],\n        [pl.col(c).quantile(0.1, \"nearest\").alias(f\"{c}_quantile1_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).quantile(0.2, \"nearest\").alias(f\"{c}_quantile2_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).quantile(0.4, \"nearest\").alias(f\"{c}_quantile4_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).quantile(0.6, \"nearest\").alias(f\"{c}_quantile6_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).quantile(0.8, \"nearest\").alias(f\"{c}_quantile8_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).quantile(0.9, \"nearest\").alias(f\"{c}_quantile9_{feature_suffix}\") for c in NUMS],\n        \n        *[pl.col(c).mean().alias(f\"{c}_mean_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).std().alias(f\"{c}_std_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).min().alias(f\"{c}_min_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).max().alias(f\"{c}_max_{feature_suffix}\") for c in NUMS],\n        \n        *[pl.col(\"event_name\").filter(pl.col(\"event_name\") == c).count().alias(f\"{c}_event_name_counts{feature_suffix}\")for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).quantile(0.1, \"nearest\").alias(f\"{c}_ET_quantile1_{feature_suffix}\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).quantile(0.2, \"nearest\").alias(f\"{c}_ET_quantile2_{feature_suffix}\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).quantile(0.4, \"nearest\").alias(f\"{c}_ET_quantile4_{feature_suffix}\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).quantile(0.6, \"nearest\").alias(f\"{c}_ET_quantile6_{feature_suffix}\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).quantile(0.8, \"nearest\").alias(f\"{c}_ET_quantile8_{feature_suffix}\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).quantile(0.9, \"nearest\").alias(f\"{c}_ET_quantile9_{feature_suffix}\") for c in event_name_feature],      \n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in event_name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"event_name\")==c).min().alias(f\"{c}_ET_min_{feature_suffix}\") for c in event_name_feature],\n     \n        *[pl.col(\"name\").filter(pl.col(\"name\") == c).count().alias(f\"{c}_name_counts{feature_suffix}\")for c in name_feature],   \n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\")==c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c in name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\")==c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\")==c).min().alias(f\"{c}_ET_min_{feature_suffix}\") for c in name_feature],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"name\")==c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in name_feature],  \n        \n        *[pl.col(\"room_fqid\").filter(pl.col(\"room_fqid\") == c).count().alias(f\"{c}_room_fqid_counts{feature_suffix}\")for c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).min().alias(f\"{c}_ET_min_{feature_suffix}\") for c in room_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"room_fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for c in room_lists],\n                \n        *[pl.col(\"fqid\").filter(pl.col(\"fqid\") == c).count().alias(f\"{c}_fqid_counts{feature_suffix}\")for c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).min().alias(f\"{c}_ET_min_{feature_suffix}\") for c in fqid_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for c in fqid_lists],\n       \n        *[pl.col(\"text_fqid\").filter(pl.col(\"text_fqid\") == c).count().alias(f\"{c}_text_fqid_counts{feature_suffix}\") for c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).std().alias(f\"{c}_ET_std_{feature_suffix}\") for c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).mean().alias(f\"{c}_ET_mean_{feature_suffix}\") for c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).max().alias(f\"{c}_ET_max_{feature_suffix}\") for c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).min().alias(f\"{c}_ET_min_{feature_suffix}\") for c in text_lists],\n        *[pl.col(\"elapsed_time_diff\").filter(pl.col(\"text_fqid\") == c).sum().alias(f\"{c}_ET_sum_{feature_suffix}\") for c in text_lists],\n         \n        *[pl.col(\"location_x_diff\").filter(pl.col(\"event_name\")==c).mean().alias(f\"{c}_ET_mean_x{feature_suffix}\") for c in event_name_feature],\n        *[pl.col(\"location_x_diff\").filter(pl.col(\"event_name\")==c).std().alias(f\"{c}_ET_std_x{feature_suffix}\") for c in event_name_feature],\n        *[pl.col(\"location_x_diff\").filter(pl.col(\"event_name\")==c).max().alias(f\"{c}_ET_max_x{feature_suffix}\") for c in event_name_feature],\n        *[pl.col(\"location_x_diff\").filter(pl.col(\"event_name\")==c).min().alias(f\"{c}_ET_min_x{feature_suffix}\") for c in event_name_feature],\n        ]\n    \n    df = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n  \n    if use_extra:\n        if grp=='5-12':\n            aggs = [\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\")==\"Here's the log book.\")|(pl.col(\"fqid\")=='logbook.page.bingo')).apply(lambda s: s.max()-s.min()).alias(\"logbook_bingo_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\")==\"Here's the log book.\")|(pl.col(\"fqid\")=='logbook.page.bingo')).apply(lambda s: s.max()-s.min()).alias(\"logbook_bingo_indexCount\"),\n                pl.col(\"elapsed_time\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='reader'))|(pl.col(\"fqid\")==\"reader.paper2.bingo\")).apply(lambda s: s.max()-s.min()).alias(\"reader_bingo_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='reader'))|(pl.col(\"fqid\")==\"reader.paper2.bingo\")).apply(lambda s: s.max()-s.min()).alias(\"reader_bingo_indexCount\"),\n                pl.col(\"elapsed_time\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='journals'))|(pl.col(\"fqid\")==\"journals.pic_2.bingo\")).apply(lambda s: s.max()-s.min()).alias(\"journals_bingo_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='journals'))|(pl.col(\"fqid\")==\"journals.pic_2.bingo\")).apply(lambda s: s.max()-s.min()).alias(\"journals_bingo_indexCount\"),\n            ]\n            tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n            df = df.join(tmp, on=\"session_id\", how='left')\n\n        if grp=='13-22':\n            aggs = [\n                pl.col(\"elapsed_time\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='reader_flag'))|(pl.col(\"fqid\")==\"tunic.library.microfiche.reader_flag.paper2.bingo\")).apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(\"reader_flag_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='reader_flag'))|(pl.col(\"fqid\")==\"tunic.library.microfiche.reader_flag.paper2.bingo\")).apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(\"reader_flag_indexCount\"),\n                pl.col(\"elapsed_time\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='journals_flag'))|(pl.col(\"fqid\")==\"journals_flag.pic_0.bingo\")).apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(\"journalsFlag_bingo_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\")=='navigate_click')&(pl.col(\"fqid\")=='journals_flag'))|(pl.col(\"fqid\")==\"journals_flag.pic_0.bingo\")).apply(lambda s: s.max()-s.min() if s.len()>0 else 0).alias(\"journalsFlag_bingo_indexCount\"),\n            ]\n            tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n            df = df.join(tmp, on=\"session_id\", how='left')\n        \n    return df.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:46.459542Z","iopub.execute_input":"2023-04-05T15:39:46.460065Z","iopub.status.idle":"2023-04-05T15:39:46.517073Z","shell.execute_reply.started":"2023-04-05T15:39:46.460022Z","shell.execute_reply":"2023-04-05T15:39:46.515562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = train.filter(pl.col(\"level_group\")=='0-4')\ndf1.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:46.519347Z","iopub.execute_input":"2023-04-05T15:39:46.519809Z","iopub.status.idle":"2023-04-05T15:39:46.888920Z","shell.execute_reply.started":"2023-04-05T15:39:46.519768Z","shell.execute_reply":"2023-04-05T15:39:46.887963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2 = train.filter(pl.col(\"level_group\")=='5-12')\ndf2.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:46.890933Z","iopub.execute_input":"2023-04-05T15:39:46.891271Z","iopub.status.idle":"2023-04-05T15:39:47.415182Z","shell.execute_reply.started":"2023-04-05T15:39:46.891232Z","shell.execute_reply":"2023-04-05T15:39:47.414036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df3 = train.filter(pl.col(\"level_group\")=='13-22')\ndf3.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:47.416626Z","iopub.execute_input":"2023-04-05T15:39:47.417067Z","iopub.status.idle":"2023-04-05T15:39:48.143024Z","shell.execute_reply.started":"2023-04-05T15:39:47.417030Z","shell.execute_reply":"2023-04-05T15:39:48.142002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Delete train to liberate memory\ndel train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:48.144480Z","iopub.execute_input":"2023-04-05T15:39:48.144897Z","iopub.status.idle":"2023-04-05T15:39:48.361197Z","shell.execute_reply.started":"2023-04-05T15:39:48.144858Z","shell.execute_reply":"2023-04-05T15:39:48.360084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf1 = feature_engineer_pl(df1, grp='0-4', use_extra=True, feature_suffix='')\nprint('df1 done, shape: ',df1.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:39:48.362709Z","iopub.execute_input":"2023-04-05T15:39:48.363476Z","iopub.status.idle":"2023-04-05T15:40:01.893870Z","shell.execute_reply.started":"2023-04-05T15:39:48.363432Z","shell.execute_reply":"2023-04-05T15:40:01.891828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.columns","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:40:01.896652Z","iopub.execute_input":"2023-04-05T15:40:01.898709Z","iopub.status.idle":"2023-04-05T15:40:01.910201Z","shell.execute_reply.started":"2023-04-05T15:40:01.898617Z","shell.execute_reply":"2023-04-05T15:40:01.909269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.tail(3)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:40:01.911806Z","iopub.execute_input":"2023-04-05T15:40:01.912519Z","iopub.status.idle":"2023-04-05T15:40:01.950421Z","shell.execute_reply.started":"2023-04-05T15:40:01.912482Z","shell.execute_reply":"2023-04-05T15:40:01.948844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1[\"session_id\"].tail(8)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:40:01.952271Z","iopub.execute_input":"2023-04-05T15:40:01.952648Z","iopub.status.idle":"2023-04-05T15:40:01.964243Z","shell.execute_reply.started":"2023-04-05T15:40:01.952614Z","shell.execute_reply":"2023-04-05T15:40:01.962931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf2 = feature_engineer_pl(df2, grp='5-12', use_extra=True, feature_suffix='')\nprint('df2 done, shape: ',df2.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:40:01.966226Z","iopub.execute_input":"2023-04-05T15:40:01.966561Z","iopub.status.idle":"2023-04-05T15:40:32.693520Z","shell.execute_reply.started":"2023-04-05T15:40:01.966530Z","shell.execute_reply":"2023-04-05T15:40:32.692403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2.tail(5)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:40:32.694842Z","iopub.execute_input":"2023-04-05T15:40:32.696076Z","iopub.status.idle":"2023-04-05T15:40:32.723616Z","shell.execute_reply.started":"2023-04-05T15:40:32.696030Z","shell.execute_reply":"2023-04-05T15:40:32.722607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf3 = feature_engineer_pl(df3, grp='13-22', use_extra=True, feature_suffix='')\nprint('df3 done, shape: ',df3.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:40:32.724913Z","iopub.execute_input":"2023-04-05T15:40:32.729815Z","iopub.status.idle":"2023-04-05T15:41:20.726239Z","shell.execute_reply.started":"2023-04-05T15:40:32.729758Z","shell.execute_reply":"2023-04-05T15:41:20.725232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df3.tail(5)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:41:20.727498Z","iopub.execute_input":"2023-04-05T15:41:20.728644Z","iopub.status.idle":"2023-04-05T15:41:20.758525Z","shell.execute_reply.started":"2023-04-05T15:41:20.728603Z","shell.execute_reply":"2023-04-05T15:41:20.757123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# some cleaning...\nnull1 = df1.isnull().sum().sort_values(ascending=False) / len(df1)\nnull2 = df2.isnull().sum().sort_values(ascending=False) / len(df1)\nnull3 = df3.isnull().sum().sort_values(ascending=False) / len(df1)\n\ndrop1 = list(null1[null1>0.9].index)\ndrop2 = list(null2[null2>0.9].index)\ndrop3 = list(null3[null3>0.9].index)\nprint(len(drop1), len(drop2), len(drop3))\n\nfor col in df1.columns:\n    if df1[col].nunique()==1:\n        #print(col)\n        drop1.append(col)\nprint(\"*********df1 DONE*********\")\nfor col in df2.columns:\n    if df2[col].nunique()==1:\n        #print(col)\n        drop2.append(col)\nprint(\"*********df2 DONE*********\")\nfor col in df3.columns:\n    if df3[col].nunique()==1:\n        #print(col)\n        drop3.append(col)\nprint(\"*********df3 DONE*********\")","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:41:20.760394Z","iopub.execute_input":"2023-04-05T15:41:20.760950Z","iopub.status.idle":"2023-04-05T15:41:23.485286Z","shell.execute_reply.started":"2023-04-05T15:41:20.760904Z","shell.execute_reply":"2023-04-05T15:41:23.483716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = df1.set_index('session_id')\ndf2 = df2.set_index('session_id')\ndf3 = df3.set_index('session_id')\n\nFEATURES1 = [c for c in df1.columns if c not in drop1+['level_group']]\nFEATURES2 = [c for c in df2.columns if c not in drop2+['level_group']]\nFEATURES3 = [c for c in df3.columns if c not in drop3+['level_group']]\nprint('We will train with', len(FEATURES1), len(FEATURES2), len(FEATURES3) ,'features')\nALL_USERS = df1.index.unique()\nprint('We will train with', len(ALL_USERS) ,'users info')","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:41:23.486891Z","iopub.execute_input":"2023-04-05T15:41:23.487411Z","iopub.status.idle":"2023-04-05T15:41:24.230533Z","shell.execute_reply.started":"2023-04-05T15:41:23.487371Z","shell.execute_reply":"2023-04-05T15:41:24.228803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(type(df1))","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:41:24.232798Z","iopub.execute_input":"2023-04-05T15:41:24.233227Z","iopub.status.idle":"2023-04-05T15:41:24.239328Z","shell.execute_reply.started":"2023-04-05T15:41:24.233187Z","shell.execute_reply":"2023-04-05T15:41:24.237709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1[FEATURES1].head(5)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:41:24.241413Z","iopub.execute_input":"2023-04-05T15:41:24.241823Z","iopub.status.idle":"2023-04-05T15:41:24.678201Z","shell.execute_reply.started":"2023-04-05T15:41:24.241786Z","shell.execute_reply":"2023-04-05T15:41:24.676896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train XGBoost Model","metadata":{}},{"cell_type":"code","source":"# With previous training notebook (Kfold with 20 folds as performed in others notebooks) :\nestimators_xgb = [498, 448, 378, 364, 405, 495, 456, 249, 384, 405, 356, 262, 484, 381, 392, 248 ,248, 345]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_params = {\n        'booster': 'gbtree',\n        'tree_method': 'hist',\n        'objective': 'binary:logistic',\n        'eval_metric':'logloss',\n        'learning_rate': 0.02,\n        'alpha': 8,\n        'max_depth': 4,\n        'subsample':0.8,\n        'colsample_bytree': 0.5,\n        'seed': 42\n        }","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwarnings.filterwarnings(\"ignore\")\n\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]))\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]))\npred_xgb = np.zeros((df1.shape[0],18))     \n\nfor t in range(1,19):\n#for t in range(1,2):\n    # USE THIS TRAIN DATA WITH THESE QUESTIONS\n    if t<=3: \n        grp = '0-4'\n        df = df1\n        FEATURES = FEATURES1\n\n    elif t<=13: \n        grp = '5-12'\n        df = df2\n        FEATURES = FEATURES2\n\n    elif t<=22: \n        grp = '13-22'\n        df = df3\n        FEATURES = FEATURES3\n        \n    xgb_params['n_estimators'] = estimators_xgb[t-1]\n     \n    # TRAIN DATA\n    train_users = df.index.values\n    train_y = targets.loc[targets.q==t].set_index('session').loc[train_users] \n\n    clf =  XGBClassifier(**xgb_params)\n    clf.fit(df[FEATURES].astype('float32'), train_y['correct'], verbose = 0)\n    clf.save_model(f'XGB_question{t}.xgb')\n    \n    print(f'model XGB saved for question {t} with iterations = {estimators_xgb[t-1]}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission Jo Wilder","metadata":{}},{"cell_type":"code","source":"import jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:42:03.705149Z","iopub.execute_input":"2023-04-05T15:42:03.705652Z","iopub.status.idle":"2023-04-05T15:42:03.764223Z","shell.execute_reply.started":"2023-04-05T15:42:03.705613Z","shell.execute_reply":"2023-04-05T15:42:03.762540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor test, sample_submission in iter_test:\n    sample_submission['question'] = [int(label.split('_')[1][1:]) for label in sample_submission['session_id']]\n    grp = test.level_group.values[0]\n    a,b = limits[grp]\n    \n    # ------------------- level 0-4 ---------------------------------\n    if a == 1:\n        FEATURES = FEATURES1\n        test = (pl.from_pandas(test)\n                  .drop([\"fullscreen\", \"hq\", \"music\"])\n                  .with_columns(columns))\n        test = feature_engineer_pl(test, grp, use_extra=True, feature_suffix='')\n        test = test[FEATURES]\n            \n\n    # ------------------- level 5-12 ---------------------------------\n    elif a == 4:\n        FEATURES = FEATURES2\n        test = (pl.from_pandas(test)\n                  .drop([\"fullscreen\", \"hq\", \"music\"])\n                  .with_columns(columns))\n        test = feature_engineer_pl(test, grp, use_extra=True, feature_suffix='')\n        test = test[FEATURES]\n\n    # ------------------- level 13-22 ---------------------------------    \n    elif a == 14:\n        FEATURES = FEATURES3\n        test = (pl.from_pandas(test)\n                  .drop([\"fullscreen\", \"hq\", \"music\"])\n                  .with_columns(columns))\n        test = feature_engineer_pl(test, grp, use_extra=True, feature_suffix='')\n        test = test[FEATURES]\n    \n\n    # INFER TEST DATA\n    \n    for t in range(a,b):\n        clf = XGBClassifier()\n        clf.load_model(f'/kaggle/working/XGB_question{t}.xgb')\n        p = clf.predict_proba(test.astype('float32'))[:,1]\n        mask = sample_submission.question == t    \n        sample_submission.loc[mask, 'correct'] = (p > 0.625).astype('int') \n    env.predict(sample_submission[['session_id', 'correct']])","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:42:12.200840Z","iopub.execute_input":"2023-04-05T15:42:12.202413Z","iopub.status.idle":"2023-04-05T15:42:13.462530Z","shell.execute_reply.started":"2023-04-05T15:42:12.202366Z","shell.execute_reply":"2023-04-05T15:42:13.461269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('submission.csv').head(10)\n#print(\"count1: \",count1)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:43:02.478784Z","iopub.execute_input":"2023-04-05T15:43:02.480246Z","iopub.status.idle":"2023-04-05T15:43:02.495474Z","shell.execute_reply.started":"2023-04-05T15:43:02.480201Z","shell.execute_reply":"2023-04-05T15:43:02.494204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}