{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom pandas.api.types import CategoricalDtype\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-19T17:10:28.933793Z","iopub.execute_input":"2023-04-19T17:10:28.934586Z","iopub.status.idle":"2023-04-19T17:10:28.978738Z","shell.execute_reply.started":"2023-04-19T17:10:28.934480Z","shell.execute_reply":"2023-04-19T17:10:28.977547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# import","metadata":{}},{"cell_type":"code","source":"import gc\nfrom IPython.display import display\nfrom tqdm import tqdm\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import StandardScaler\n\nfrom sklearn.decomposition import FactorAnalysis\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import RandomizedSearchCV\n\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import f1_score","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:10:28.980594Z","iopub.execute_input":"2023-04-19T17:10:28.981224Z","iopub.status.idle":"2023-04-19T17:10:30.463475Z","shell.execute_reply.started":"2023-04-19T17:10:28.981187Z","shell.execute_reply":"2023-04-19T17:10:30.462477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Optimization memory","metadata":{}},{"cell_type":"markdown","source":"# Memory optimization function","metadata":{}},{"cell_type":"code","source":"def optimize_memory_usage(df, print_size=True):\n    # Function optimizes memory usage in dataframe.\n    # (RU) Функция оптимизации типов в dataframe.\n    \n    # Types for optimization.\n    # Типы, которые будем проверять на оптимизацию.\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    # Memory usage size before optimize (Mb).\n    # (RU) Размер занимаемой памяти до оптимизации (в Мб).\n    before_size = df.memory_usage().sum() / 1024**2\n    \n    for column in tqdm(df.columns):\n        column_type = df[column].dtypes\n        if column_type in numerics:\n            column_min = df[column].min()\n            column_max = df[column].max()\n            if str(column_type).startswith('int'):\n                if column_min > np.iinfo(np.int8).min and column_max < np.iinfo(np.int8).max:\n                    df[column] = df[column].astype(np.int8)\n                elif column_min > np.iinfo(np.int16).min and column_max < np.iinfo(np.int16).max:\n                    df[column] = df[column].astype(np.int16)\n                elif column_min > np.iinfo(np.int32).min and column_max < np.iinfo(np.int32).max:\n                    df[column] = df[column].astype(np.int32)\n                elif column_min > np.iinfo(np.int64).min and column_max < np.iinfo(np.int64).max:\n                    df[column] = df[column].astype(np.int64) \n                    \n            elif str(column_type).startswith('float'):\n                if column_min > np.finfo(np.float32).min and column_max < np.finfo(np.float32).max:\n                    df[column] = df[column].astype(np.float32)\n                else:\n                    df[column] = df[column].astype(np.float64)\n                    \n        elif str(column_type).startswith('object'):\n            df[column] = df[column].astype('category')\n        else:\n            pass\n    # Memory usage size after optimize (Mb).\n    # (RU) Размер занимаемой памяти после оптимизации (в Мб).\n    after_size = df.memory_usage().sum() / 1024**2\n\n    if print_size: print(f'Memory usage size: before {before_size:5.4f} Mb - after {after_size:5.4f} Mb ({100 * (before_size - after_size) / before_size:.1f}%).')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:10:30.469154Z","iopub.execute_input":"2023-04-19T17:10:30.471440Z","iopub.status.idle":"2023-04-19T17:10:30.488211Z","shell.execute_reply.started":"2023-04-19T17:10:30.471395Z","shell.execute_reply":"2023-04-19T17:10:30.487200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reading train data and conversion type\n# Чтение данных  и оптимизация типа данных ","metadata":{}},{"cell_type":"code","source":"path = \"/kaggle/input/predict-student-performance-from-game-play/\"\nchunksize = 5_000_000\n\nwith pd.read_csv(path + \"train.csv\", chunksize= chunksize) as reader:\n    train = optimize_memory_usage(next(reader))\n    validation = optimize_memory_usage(next(reader))\n\ntrain_labels = optimize_memory_usage( pd.read_csv(path + \"train_labels.csv\"), print_size = True )\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:10:30.492826Z","iopub.execute_input":"2023-04-19T17:10:30.495258Z","iopub.status.idle":"2023-04-19T17:11:33.200406Z","shell.execute_reply.started":"2023-04-19T17:10:30.495193Z","shell.execute_reply":"2023-04-19T17:11:33.199284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape, validation.shape, train_labels.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:11:33.203066Z","iopub.execute_input":"2023-04-19T17:11:33.203470Z","iopub.status.idle":"2023-04-19T17:11:33.210223Z","shell.execute_reply.started":"2023-04-19T17:11:33.203434Z","shell.execute_reply":"2023-04-19T17:11:33.209327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Eda","metadata":{}},{"cell_type":"markdown","source":"### Столбцы\n    session_id - идентификатор сеанса, в котором произошло событие                                           groupby element   -> количественный  -> дискретный\n    name - имя события (например, определяет, открывает или закрывает ли блокнот Notebook_click)             groupby element   -> качественный    -> номинальный\n    level_group — к какой группе уровней — и группе вопросов — принадлежит эта строка (0-4, 5-12, 13-22)     groupby element   -> качественный    -> номинальный\n    \n    \n    elapsed_time — сколько времени прошло (в миллисекундах) между началом сеанса и моментом записи события   -> количественный  -> дискретный\n    \n    level - на каком уровне игры произошло событие (от 0 до 22)                                              -> качественный  -> порядковый\n    \n    event_name - название типа события                                                                       -> качественный  -> номинальный       надо закодирован под бигпрный\n    fqid — полный идентификатор события                                                                      -> качественный    -> номинальный     0.328058\n    room_fqid — полный идентификатор комнаты, в которой произошло событие                                    -> качественный    -> номинальный     надо  закодирован под бигпрный\n    \n    room_coor_x - координаты клика относительно игровой комнаты (только для кликов)                          -> качественный  -> непрерывный       0.097103\n    room_coor_y - координаты клика относительно игровой комнаты (только для кликов)                          -> качественный  -> непрерывный       0.097103\n    screen_coor_x - координаты клика относительно экрана игрока (только для кликов)                          -> качественный  -> непрерывный       0.097103\n    screen_coor_y - координаты клика относительно экрана игрока (только для кликов)                          -> качественный  -> непрерывный       0.097103\n    \n    text - текст, который игрок видит во время этого события                                                 -> text  0.688305    \n​","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:11:33.211519Z","iopub.execute_input":"2023-04-19T17:11:33.212024Z","iopub.status.idle":"2023-04-19T17:11:33.249039Z","shell.execute_reply.started":"2023-04-19T17:11:33.211992Z","shell.execute_reply":"2023-04-19T17:11:33.248204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:11:33.250288Z","iopub.execute_input":"2023-04-19T17:11:33.250786Z","iopub.status.idle":"2023-04-19T17:11:35.286820Z","shell.execute_reply.started":"2023-04-19T17:11:33.250752Z","shell.execute_reply":"2023-04-19T17:11:35.285841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing = train.isna().sum() / train.isna().count()\npercent =  0.1\nprint(\"Lossy data:\\nДанные с потерями:\")\nprint(missing[missing > percent ], \"\\n\")\nprint(missing[missing > percent ].index.to_list(),\"\\n\" )\nprint(\"Data without gaps:\\nДанные без пропусков:\")\nprint(missing[missing == 0])\nprint()\nprint(\"Number of unique table values: \\nКоличество уникальных значений в  таблице : \")\ndisplay( train.nunique())\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:11:35.288149Z","iopub.execute_input":"2023-04-19T17:11:35.288684Z","iopub.status.idle":"2023-04-19T17:11:37.465355Z","shell.execute_reply.started":"2023-04-19T17:11:35.288649Z","shell.execute_reply":"2023-04-19T17:11:37.464260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature engineering","metadata":{}},{"cell_type":"code","source":"# deleting data with large gaps\nDROP_l_v = ['page', 'hover_duration', 'text', 'text_fqid', 'fullscreen', 'hq', 'music'] \ntrain.drop(DROP_l_v, axis = 1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:11:37.466906Z","iopub.execute_input":"2023-04-19T17:11:37.467249Z","iopub.status.idle":"2023-04-19T17:11:37.562314Z","shell.execute_reply.started":"2023-04-19T17:11:37.467205Z","shell.execute_reply":"2023-04-19T17:11:37.561291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Add column question","metadata":{}},{"cell_type":"code","source":"def add_question(data:pd.DataFrame) -> pd.DataFrame:\n    \"\"\"\n    add column session_q -> questions\n    добавить колонку session_q\n    q -> 0-4 = 3 (q1-q3), 5-12 = 10 (q4-q13), 13-22 = 5(q14-q18)\n    \n    old_lvl: all level = True, not all level = False\n    \"\"\"\n    \n    df = pd.DataFrame() \n    \n    missed_questions = [7,10] # вопросы на которые надо розширить датафрейм \n    session_id  = pd.unique(data.session_id) \n    \n    # роспределенние вопросов по групах\n    cut = {'0-4': list(range(1,4)), '5-12': list(range(4,14)), '13-22': list(range(14,19)) }\n    \n    # итерфция по session_id\n    for i in tqdm(session_id):\n        df_s = data.loc[data.session_id == i, : ].copy()\n        level_group =  pd.unique(df_s.level_group)\n        # итерация по level_group\n        for j in level_group:\n        # проверка level_group \n            if j in ('0-4', '13-22'):\n                # данные с груп '0-4' и '13-22' уменшаются в розмере\n                df_l = df_s.loc[ df_s.level_group == j, : ].copy()\n                df_l[\"q\"] = pd.cut( df_l.level, bins = len(cut[j]), labels= cut[j] ) \n                df = pd.concat([df, df_l], axis=0)\n\n            else:\n                # данные с груп ''5-12' увеличеваются в розмере на 2 вопроса\n                # add missing questions\n                # добавить недостающие вопросы\n                df_l = df_s.loc[ df_s.level_group == j, : ].copy() \n                df_l[\"q\"] = pd.cut( df_l.level, bins = len(cut[j]), labels= cut[j])\n                df_l =  pd.concat([ df_l, pd.DataFrame({\"q\": missed_questions }) ], axis=0 , ignore_index = True)\n\n\n                 # completing added questions\n                 # заполнение добавленных вопросов\n                for q in missed_questions:\n                     # qualitative variables\n                    df_l.loc[df_l.q == q , \"session_id\"]  = i\n                    df_l.loc[df_l.q == q , \"level_group\"] = j\n                    df_l.loc[df_l.q == q , \"level\"]       = df_l.loc[df_l.q == q, \"q\"]\n\n                    df_l.loc[df_l.q == q , \"event_name\"] = df_l.loc[df_l.q <= q, \"event_name\" ].mode()[0]\n                    df_l.loc[df_l.q == q , \"name\"]       = df_l.loc[df_l.q <= q, \"name\" ].mode()[0]\n                    df_l.loc[df_l.q == q , \"room_fqid\"]  = df_l.loc[df_l.q <= q, \"room_fqid\" ].mode()[0]\n\n                    # quantitative\n                    df_l.loc[df_l.q == q , \"elapsed_time\"]  = df_l.loc[ df_l.q <= q, \"elapsed_time\"  ].median()\n                    df_l.loc[df_l.q == q , \"room_coor_x\"]   = df_l.loc[ df_l.q <= q, \"room_coor_x\"   ].median()\n                    df_l.loc[df_l.q == q , \"room_coor_y\"]   = df_l.loc[ df_l.q <= q, \"room_coor_x\"   ].median()\n                    df_l.loc[df_l.q == q , \"screen_coor_x\"] = df_l.loc[ df_l.q <= q, \"screen_coor_x\" ].median()\n                    df_l.loc[df_l.q == q , \"screen_coor_y\"] = df_l.loc[ df_l.q <= q, \"screen_coor_y\" ].median()\n\n\n                df = pd.concat([df, df_l], axis=0)\n\n    df = df.astype( {\"session_id\" : np.int64 } )\n    df = optimize_memory_usage(df, print_size=False)\n    gc.collect()\n\n    return df\n\ntrain =  add_question( train )","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:11:37.564098Z","iopub.execute_input":"2023-04-19T17:11:37.564482Z","iopub.status.idle":"2023-04-19T17:26:28.139916Z","shell.execute_reply.started":"2023-04-19T17:11:37.564448Z","shell.execute_reply":"2023-04-19T17:26:28.138796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def future_engineering(data: pd.DataFrame) -> pd.DataFrame :\n    \n    event_n = pd.get_dummies(data['event_name'])\n    data = pd.concat([data, event_n], axis=1)\n\n    binary = ['checkpoint',\n              'map_click',\n              'navigate_click',\n              'notebook_click',\n              'object_hover' ]\n\n    nominal =  ['fqid']\n    discrete = ['elapsed_time']\n    ordinal =  ['level']\n    continuous = ['room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y' ]\n\n    \n    df = pd.DataFrame()\n    \n    columns_to_group = [\"session_id\",\"q\"]\n    group_data = data.groupby(columns_to_group)\n    \n    for c in tqdm( binary):\n        df[ c + \"_\" + \"nunique\" ] = group_data[c].agg('nunique')\n        df[ c + \"_\" + \"count\" ] = group_data[c].agg('count')\n        \n    for c in tqdm(nominal):\n        df[ c + \"_\" + \"nunique\" ] = group_data[c].agg('nunique')\n        \n    for c in tqdm(discrete):\n        df[ c + \"_\" + \"count\" ] = group_data[c].agg('count')\n        df[ c + \"_\" + \"median\" ] = group_data[c].agg('median')\n        df[ c + \"_\" + \"sum\" ] = group_data[c].agg('sum')\n        df[ c + \"_\" + \"min\" ] = group_data[c].agg('min')\n        df[ c + \"_\" + \"max\" ] = group_data[c].agg('max')\n        df[ c + \"_\" + \"mean\" ] = group_data[c].agg('mean')\n        df[ c + \"_\" + \"std\" ] = group_data[c].agg('std')\n        df[ c + \"_\" + \"mad\" ] = group_data[c].agg('mad')\n\n    for c in tqdm(ordinal):\n        df[ c + \"_\" + \"nunique\" ] = group_data[c].agg('nunique')\n        df[ c + \"_\" + \"std\" ] = group_data[c].agg('std')\n        df[ c + \"_\" + \"mad\" ] = group_data[c].agg('mad')\n        \n    for c in tqdm(continuous):\n        df[ c + \"_\" + \"count\" ] = group_data[c].agg('count')\n        df[ c + \"_\" + \"median\" ] = group_data[c].agg('median')\n        df[ c + \"_\" + \"sum\" ] = group_data[c].agg('sum')\n        df[ c + \"_\" + \"min\" ] = group_data[c].agg('min')\n        df[ c + \"_\" + \"max\" ] = group_data[c].agg('max')\n        df[ c + \"_\" + \"mean\" ] = group_data[c].agg('mean')\n        df[ c + \"_\" + \"std\" ] = group_data[c].agg('std')\n        df[ c + \"_\" + \"mad\" ] = group_data[c].agg('mad')\n\n        \n\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df.index = range(len(df.index))\n    \n\n    df = optimize_memory_usage(df, print_size=False)\n    gc.collect()\n\n    return df\n\ntrain = future_engineering( train )","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:26:28.143454Z","iopub.execute_input":"2023-04-19T17:26:28.143777Z","iopub.status.idle":"2023-04-19T17:31:00.300305Z","shell.execute_reply.started":"2023-04-19T17:26:28.143748Z","shell.execute_reply":"2023-04-19T17:31:00.299288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def join_s_q(data: pd.DataFrame ) -> pd.DataFrame:\n    \"\"\"\n    join session_id and question and remove those columns\n    \"\"\"\n    \n    data['session_id_q'] = data.session_id.astype(str) + \"_q\" + data.q.astype(str)\n    \n    data = data.sort_values([\"session_id\",\"q\"])\n    \n    drop = [\"session_id\"]\n    data.drop(drop, axis = 1, inplace = True )\n    \n    data = optimize_memory_usage(data, print_size=False)\n    gc.collect()\n    \n    return data\n\ntrain = join_s_q(train)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:31:00.301700Z","iopub.execute_input":"2023-04-19T17:31:00.302492Z","iopub.status.idle":"2023-04-19T17:31:00.737518Z","shell.execute_reply.started":"2023-04-19T17:31:00.302454Z","shell.execute_reply":"2023-04-19T17:31:00.736526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def merge_data_and_labels(data:pd.DataFrame, labels:pd.Series ) -> pd.DataFrame:\n    data = data.merge(labels, right_on=\"session_id\", left_on =\"session_id_q\", how='inner',suffixes=('_l', '_r'))\n    data.drop(\"session_id_q\", axis = 1, inplace = True)\n    gc.collect()    \n    return data\n    \ntrain = merge_data_and_labels(train, train_labels )","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:31:00.738899Z","iopub.execute_input":"2023-04-19T17:31:00.739450Z","iopub.status.idle":"2023-04-19T17:31:01.411275Z","shell.execute_reply.started":"2023-04-19T17:31:00.739415Z","shell.execute_reply":"2023-04-19T17:31:01.409546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Factor Analysis","metadata":{}},{"cell_type":"code","source":"NUMERIC = [i for i in train.dtypes.keys() if i not in [\"session_id\",\"correct\"] ] # выбираем только интовые значенния\nq = train.q\n\nscaler = StandardScaler()\ndf = scaler.fit_transform( train[NUMERIC] )\ntrain[NUMERIC] = df\n\nfa = FactorAnalysis(n_components= 3,copy= True, random_state= 0, svd_method= 'lapack' )\nfa.fit_transform( train[NUMERIC] )\n\n# correlation between factors and variables\nmissing  = pd.DataFrame(fa.components_, columns= NUMERIC).T","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:31:01.413177Z","iopub.execute_input":"2023-04-19T17:31:01.413692Z","iopub.status.idle":"2023-04-19T17:33:26.983752Z","shell.execute_reply.started":"2023-04-19T17:31:01.413652Z","shell.execute_reply":"2023-04-19T17:33:26.981342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f_1 = missing[(missing[0] > .5 ) | (missing[0] < -.5 )][0].index\nf_2 = missing[(missing[0] > .5 ) | (missing[0] < -.5 )][1].index\nf_3 = missing[(missing[0] > .5 ) | (missing[0] < -.5 )][2].index\n\nprint( missing[(missing[0] > .5 ) | (missing[0] < -.5 )][0] )\nprint()\nprint( missing[(missing[1] > .5 ) | (missing[1] < -.5 )][1] )\nprint()\nprint( missing[(missing[2] > .5 ) | (missing[2] < -.5 )][2] )","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:33:26.987569Z","iopub.execute_input":"2023-04-19T17:33:26.990486Z","iopub.status.idle":"2023-04-19T17:33:27.011076Z","shell.execute_reply.started":"2023-04-19T17:33:26.990443Z","shell.execute_reply":"2023-04-19T17:33:27.010274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# to be removed after FactorAnalysis\nDROP_F = list({*f_1, *f_2, *f_3})\nlen(DROP_F), type(DROP_F)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:33:27.015337Z","iopub.execute_input":"2023-04-19T17:33:27.017847Z","iopub.status.idle":"2023-04-19T17:33:27.027421Z","shell.execute_reply.started":"2023-04-19T17:33:27.017792Z","shell.execute_reply":"2023-04-19T17:33:27.026593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score_fa = pd.DataFrame(fa.transform(train[NUMERIC]), columns=[\"f_1\",\"f_2\",\"f_1\"])\ntrain.drop(DROP_F, axis = 1, inplace = True )\ntrain = pd.concat([ train, score_fa ], axis=1)\ntrain.q = q","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:33:27.032135Z","iopub.execute_input":"2023-04-19T17:33:27.034638Z","iopub.status.idle":"2023-04-19T17:33:27.082251Z","shell.execute_reply.started":"2023-04-19T17:33:27.034593Z","shell.execute_reply":"2023-04-19T17:33:27.081306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data preparation for forecasting","metadata":{}},{"cell_type":"code","source":"def tr_te_split(data):\n    \"\"\"\n    train test split and delit categorial\\str columns \n    \"\"\"\n    # data_for_models ->  0 = X_train,1 = X_test,2 = y_train,3 = y_test\n    data_for_models = dict()\n    \n    for i in tqdm( range(1,19) ):\n        df = data.loc[data.q == i ]\n        drop = [\"correct\",\"q\"] \n        X = df.drop(drop , axis = 1)\n        y = df.correct\n\n        data_for_models[i] = train_test_split( X, y, test_size=0.33, random_state=42) \n    gc.collect()       \n    return data_for_models\n\ndata = train.drop(\"session_id\", axis =1)\ndata_for_models = tr_te_split(data) # tr_te_split(train[LIST_Q_COR])\n\nprint( data_for_models[1][0].shape )  # X_train\nprint( data_for_models[1][1].shape )  # X_test\nprint( data_for_models[1][2].shape )  # y_train\nprint( data_for_models[1][3].shape )  # y_test","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:33:27.086483Z","iopub.execute_input":"2023-04-19T17:33:27.089124Z","iopub.status.idle":"2023-04-19T17:33:27.333033Z","shell.execute_reply.started":"2023-04-19T17:33:27.089081Z","shell.execute_reply":"2023-04-19T17:33:27.332040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# RandomForestClassifier Model preparation","metadata":{}},{"cell_type":"code","source":"rfc = RandomForestClassifier(max_depth=8, criterion='entropy', n_estimators=400, random_state=0)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-04-19T17:33:27.334391Z","iopub.execute_input":"2023-04-19T17:33:27.335009Z","iopub.status.idle":"2023-04-19T17:33:27.340260Z","shell.execute_reply.started":"2023-04-19T17:33:27.334966Z","shell.execute_reply":"2023-04-19T17:33:27.339163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# RandomizedSearchCV","metadata":{}},{"cell_type":"markdown","source":"    %time\n    rfc = RandomForestClassifier(max_depth=5, random_state=0)\n\n\n    params = {\n             \"max_depth\" : list(range(5,15,1)),\n             \"n_estimators\" : list(range(100,600,100)),\n             \"criterion\" : [\"gini\",\"entropy\"],\n             \"bootstrap\" : [True, False ] , \n        } \n\n    rscv = RandomizedSearchCV(estimator = rfc, param_distributions = params, cv=5, n_iter=21, n_jobs = -1, random_state = 0  )\n\n\n    model = dict()\n\n    for i in range(1,19):\n        model[i] = rscv.fit( data_for_models[i][0], data_for_models[i][2] ) \n        \n        \n        \n        \n        \n    for i in range(1,19):\n        print( model[i].best_estimator_)\n        \n        \n     OUTPUT:\n     \n    RandomForestClassifier(max_depth=8, n_estimators=400, random_state=0)\n    RandomForestClassifier(max_depth=8, n_estimators=400, random_state=0)\n    RandomForestClassifier(max_depth=8, n_estimators=400, random_state=0)\n    RandomForestClassifier(max_depth=8, n_estimators=400, random_state=0)\n    RandomForestClassifier(max_depth=8, n_estimators=400, random_state=0)\n    RandomForestClassifier(max_depth=8, n_estimators=400, random_state=0)\n    RandomForestClassifier(max_depth=8, n_estimators=400, random_state=0)\n    RandomForestClassifier(max_depth=8, n_estimators=400, random_state=0)\n    RandomForestClassifier(max_depth=8, n_estimators=400, random_state=0)\n    RandomForestClassifier(max_depth=8, n_estimators=400, random_state=0)\n    RandomForestClassifier(max_depth=8, n_estimators=400, random_state=0)\n    RandomForestClassifier(max_depth=8, n_estimators=400, random_state=0)\n    RandomForestClassifier(max_depth=8, n_estimators=400, random_state=0)\n    RandomForestClassifier(max_depth=8, n_estimators=400, random_state=0)\n    RandomForestClassifier(max_depth=8, n_estimators=400, random_state=0)\n    RandomForestClassifier(max_depth=8, n_estimators=400, random_state=0)\n    RandomForestClassifier(max_depth=8, n_estimators=400, random_state=0)\n    RandomForestClassifier(max_depth=8, n_estimators=400, random_state=0)","metadata":{}},{"cell_type":"markdown","source":"# Fit model","metadata":{}},{"cell_type":"code","source":"%time\ntrained_model = dict()\nfor i in tqdm( range(1,19) ):\n    trained_model[i] = rfc.fit( data_for_models[i][0], data_for_models[i][2] )","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:33:27.341582Z","iopub.execute_input":"2023-04-19T17:33:27.342123Z","iopub.status.idle":"2023-04-19T17:34:31.314978Z","shell.execute_reply.started":"2023-04-19T17:33:27.342084Z","shell.execute_reply":"2023-04-19T17:34:31.313938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Getting and setting forecasts\n# Получение и настройка прогнозов","metadata":{}},{"cell_type":"code","source":"# Setting predict probability and record best probability\npercentage_for_pred_prob = dict()\n\npredict_X_train = dict()\npredict_X_test = dict()\n\nfor i in range(1,19):\n    predict_X_train[i] = trained_model[i].predict(data_for_models[i][0])\n    predict_X_test[i] = trained_model[i].predict(data_for_models[i][1])\n\nfor i in tqdm( range(1,19) ):\n    predict_p = trained_model[i].predict_proba(data_for_models[i][0])\n    \n    percent = 0.01\n    old_bill = 0\n    for j in range(100):\n\n        pred_prob = np.where( predict_p[:, 1] > percent  , 1, 0 )\n        scor = f1_score(data_for_models[i][2], pred_prob, average='macro')\n\n        if scor > old_bill:\n            old_bill = scor\n            percentage_for_pred_prob[i] = percent\n            \n        percent += 0.01\n        \n        \n# Estimating the change in forecasts after fitting the probability forecast\nscor_train = []\nscor_test = []\n\nfor i in range(1,19):\n    \n    predict_train = trained_model[i].predict_proba(data_for_models[i][0])\n    predict_test = trained_model[i].predict_proba(data_for_models[i][1])\n    \n    pred_prob_train = np.where( predict_train[:, 1] > percentage_for_pred_prob[i], 1, 0 )\n    pred_prob_test = np.where( predict_test[:, 1] > percentage_for_pred_prob[i], 1, 0 )\n    \n    scor_train.append(f1_score(data_for_models[i][2], pred_prob_train, average='macro'))\n    scor_test.append(f1_score(data_for_models[i][3], pred_prob_test, average='macro'))\n\n\n    print( f1_score(data_for_models[i][2], predict_X_train[i] , average='macro'))\n    print( f1_score(data_for_models[i][3], predict_X_test[i] , average='macro'))\n    print(\"After promotion\")\n    print( f1_score(data_for_models[i][2], pred_prob_train, average='macro'))\n    print( f1_score(data_for_models[i][3], pred_prob_test, average='macro'))\n    print(\"#\"*30)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:34:31.316459Z","iopub.execute_input":"2023-04-19T17:34:31.316815Z","iopub.status.idle":"2023-04-19T17:34:44.106979Z","shell.execute_reply.started":"2023-04-19T17:34:31.316782Z","shell.execute_reply":"2023-04-19T17:34:44.105881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"train:\", sum(scor_train) / len(scor_train) )\nprint(\"test:\", sum(scor_test) / len(scor_test) )","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:34:44.108456Z","iopub.execute_input":"2023-04-19T17:34:44.108777Z","iopub.status.idle":"2023-04-19T17:34:44.114013Z","shell.execute_reply.started":"2023-04-19T17:34:44.108747Z","shell.execute_reply":"2023-04-19T17:34:44.113041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(data_for_predict: dict,model: dict, percent: list, q_num: list) -> list:\n    \n    answers = []\n    \n    for index in tqdm( range(len(data_for_predict))): \n        predict_p = model[q_num[index]].predict_proba( data_for_predict.iloc[index : index+1] )\n        pred_prob = np.where( predict_p[:, 1] > percent[q_num[index]]  ,1 ,0 )\n        answers.append(*pred_prob)\n     \n    gc.collect()\n    return answers ","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:34:44.115559Z","iopub.execute_input":"2023-04-19T17:34:44.115885Z","iopub.status.idle":"2023-04-19T17:34:44.126713Z","shell.execute_reply.started":"2023-04-19T17:34:44.115854Z","shell.execute_reply":"2023-04-19T17:34:44.125450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Validation data preparation","metadata":{}},{"cell_type":"code","source":"# validation.drop(DROP_l_v, axis = 1, inplace=True)\n\n# validation =  add_question( validation )\n\n# validation = future_engineering( validation )\n# validation = join_s_q(validation)\n\n# validation = merge_data_and_labels(validation, train_labels )\n\n# question_number = validation.q \n# correct = validation.correct\n\n# ###########################################################################################################\n# validation[NUMERIC] = scaler.transform( validation[NUMERIC])\n# ############################################################################################################\n# score_fa = pd.DataFrame(fa.transform(validation[NUMERIC]), columns=[\"f_1\",\"f_2\",\"f_1\"])\n# validation = pd.concat([ validation.drop([*DROP_F,\"q\",\"session_id\",\"correct\"], axis = 1), score_fa ], axis=1)\n\n# ############################################################################################################\n\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:34:44.128521Z","iopub.execute_input":"2023-04-19T17:34:44.128856Z","iopub.status.idle":"2023-04-19T17:34:44.143351Z","shell.execute_reply.started":"2023-04-19T17:34:44.128826Z","shell.execute_reply":"2023-04-19T17:34:44.142301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# answers = predict(data_for_predict = validation, model = trained_model,  \n#                   percent =percentage_for_pred_prob, q_num= question_number.to_list() )","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:34:44.144771Z","iopub.execute_input":"2023-04-19T17:34:44.145130Z","iopub.status.idle":"2023-04-19T17:34:44.155521Z","shell.execute_reply.started":"2023-04-19T17:34:44.145098Z","shell.execute_reply":"2023-04-19T17:34:44.154527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Forecasts for validations","metadata":{}},{"cell_type":"code","source":"# print( f1_score(answers, correct, average='macro'))\n# 0.5778329168472126","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:34:44.157031Z","iopub.execute_input":"2023-04-19T17:34:44.157366Z","iopub.status.idle":"2023-04-19T17:34:44.166146Z","shell.execute_reply.started":"2023-04-19T17:34:44.157337Z","shell.execute_reply":"2023-04-19T17:34:44.165040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"# jo_wilder.make_env.__called__ = False\n# type(env)._state = type(type(env)._state).__dict__['INIT']\n\n# import jo_wilder\n# env = jo_wilder.make_env()\n# iter_test = env.iter_test()\n\n\n# # API доставит два фрейма данных в указанном порядке,\n# # для каждой сессии + группировка на уровне (одна группа на сессию для каждой контрольной точки)\n# for (test, sample_submission) in iter_test:\n    \n    \n#     test = optimize_memory_usage(test)\n#     ################################################\n#     test.drop(DROP_l_v, axis = 1, inplace=True)\n#     ################################################\n#     test =  add_question( test )\n    \n#     test = future_engineering( test )\n#     test = join_s_q(test)\n#     ################################################\n#     session_id = test.session_id_q\n#     question_number = test.q\n#     ################################################\n#     test[NUMERIC] = scaler.transform( test[NUMERIC])\n#     ################################################    \n#     score_fa = pd.DataFrame(fa.transform(test[NUMERIC]), columns=[\"f_1\",\"f_2\",\"f_1\"])\n#     test.drop([*DROP_F, \"q\", \"session_id_q\"], axis = 1, inplace = True )\n#     test = pd.concat([test, score_fa ], axis=1)\n    \n#     answer = predict(data_for_predict = test, model = trained_model,  \n#                   percent =percentage_for_pred_prob, q_num= question_number.to_list() )\n        \n#     for i, v in enumerate( answer ):\n#         sample_submission.loc[sample_submission.session_id == session_id[i], \"correct\"] = v.astype(int)\n    \n\n        \n# #     display(sample_submission)\n#     env.predict(sample_submission)\n#     gc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:34:44.167951Z","iopub.execute_input":"2023-04-19T17:34:44.168285Z","iopub.status.idle":"2023-04-19T17:34:44.180653Z","shell.execute_reply.started":"2023-04-19T17:34:44.168255Z","shell.execute_reply":"2023-04-19T17:34:44.179520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# jo_wilder.make_env.__called__ = False\n# type(env)._state = type(type(env)._state).__dict__['INIT']\n\nimport jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n\n\nfor (test, sample_submission) in iter_test:\n\n    test = optimize_memory_usage(test)\n    ################################################\n    test.drop(DROP_l_v, axis = 1, inplace=True)\n    ################################################\n    test =  add_question( test )\n    \n    test = future_engineering( test )\n    test = join_s_q(test)\n    ################################################\n    session_id = test.session_id_q\n    question_number = test.q\n    ################################################\n    test[NUMERIC] = scaler.transform( test[NUMERIC])\n    ################################################    \n    score_fa = pd.DataFrame(fa.transform(test[NUMERIC]), columns=[\"f_1\",\"f_2\",\"f_1\"])\n    test.drop([*DROP_F, \"q\", \"session_id_q\"], axis = 1, inplace = True )\n    test = pd.concat([test, score_fa ], axis=1)\n    \n    \n    \n    answer = predict(data_for_predict = test, model = trained_model,  \n                  percent =percentage_for_pred_prob, q_num= question_number.to_list() )\n    \n    try:\n        for i, v in enumerate( answer ):\n            sample_submission.loc[sample_submission.session_id == session_id[i], \"correct\"] = int(v)\n    except:\n        sample_submission.loc[:, \"correct\"] = 999\n    finally :\n        env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:34:44.182349Z","iopub.execute_input":"2023-04-19T17:34:44.182682Z","iopub.status.idle":"2023-04-19T17:34:49.298646Z","shell.execute_reply.started":"2023-04-19T17:34:44.182651Z","shell.execute_reply":"2023-04-19T17:34:49.297508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA submission.csv","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('submission.csv')\nprint( df.shape )\ndisplay(df.info())\ndf.tail(54)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:34:49.303270Z","iopub.execute_input":"2023-04-19T17:34:49.303611Z","iopub.status.idle":"2023-04-19T17:34:49.328479Z","shell.execute_reply.started":"2023-04-19T17:34:49.303580Z","shell.execute_reply":"2023-04-19T17:34:49.327432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Search questions present in the game","metadata":{}}]}