{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This code is for those who are interested in how dataframe features were made for https://www.kaggle.com/code/vadimkamaev/catboost\nI'm sorry, but the code here will not work. It was made in pycharm on the local computer.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import KFold, GroupKFold\nimport xgboost as xgb\nfrom catboost import CatBoostClassifier\nfrom sklearn.metrics import f1_score\n\ndef read_csv_loc(file):\n    dtypes = {\"session_id\": 'int64',\n              \"index\": np.int16,\n              \"elapsed_time\": np.int32,\n              \"event_name\": 'category',\n              \"name\": 'category',\n              \"level\": np.int8,\n              \"page\": np.float16,\n              \"room_coor_x\": np.float16,\n              \"room_coor_y\": np.float16,\n              # \"screen_coor_x\": np.float16,\n              # \"screen_coor_y\": np.float16,\n              \"hover_duration\": np.float32,\n              \"text\": 'category',\n              \"fqid\": 'category',\n              \"room_fqid\": 'category',\n              \"text_fqid\": 'category',\n              # \"fullscreen\": np.int8,\n              # \"hq\": np.int8,\n              # \"music\": np.int8,\n              \"level_group\": 'category'\n              }\n    train = pd.read_csv(file, dtype=dtypes)\n    return train\n\n# ___________ CREATING A DATAFRAME WITH A LIST OF FEATURES _________________\n\ndef fich1():\n    global train, feature_df, l_quest, nabor\n    CATS = ['name', 'event_name', 'fqid', 'room_fqid', 'text_fqid', 'level', 'page', 'text']\n    nabor = 0\n    for cat in CATS:\n        nabor += 1\n        for quest in l_quest:\n            for val in train[cat].unique():\n                # feature_df.loc[len(feature_df.index)] = (nabor, 't', quest, 1, cat, val, 0, 0, 0, 0, 0)\n                feature_df.loc[len(feature_df.index)] = (nabor, ' ', quest, 1, cat, val, 0, 0, 0, 0, 0)\n\ndef fich2():\n    global train, feature_df, l_quest, nabor\n    CATS = [['room_fqid', 'level'],['text_fqid', 'level'], ['fqid', 'level'], ['room_fqid', 'fqid']]\n    for cat in CATS:\n        nabor += 1\n        for quest in l_quest:\n            lcet0 = train[cat[0]]\n            for val0 in lcet0.unique():\n                l_cat1 = train[lcet0==val0][cat[1]].unique()\n                if len(l_cat1) > 1:\n                    for val1 in l_cat1:\n                        # feature_df.loc[len(feature_df.index)]=(nabor,'t',quest,2,cat[0],val0,cat[1],val1,0,0,0)\n                        feature_df.loc[len(feature_df.index)]=(nabor,' ',quest,2,cat[0],val0,cat[1],val1,0,0,0)\n\n# def create_df_feature():\n#     global train, feature_df, l_quest\n#     feature_df = pd.DataFrame(columns=['nabor', 'tip', 'quest', 'kol_col', 'col1', 'val1', 'col2', 'val2',\n#                                        'col3', 'val3', 'kach', 'rez'])\n\n#     train = read_csv_loc(\"/kaggle/input/predict-student-performance-from-game-play/train_0_4t.csv\")\n#     l_quest = [1,2,3]\n#     fich1()\n#     fich2()\n\n#     train = read_csv_loc(\"C:\\\\kaggle\\\\ОбучИгра\\\\train_5_12t.csv\")\n#     l_quest = [4,5,6,7,8,9,10,11,12,13]\n#     fich1()\n#     fich2()\n\n#     train = read_csv_loc(\"C:\\\\kaggle\\\\ОбучИгра\\\\train_13_22t.csv\")\n#     l_quest = [14,15,16,17,18]\n#     fich1()\n#     fich2()\n#     feature_df.sort_values(by = 'quest', inplace=True)\n#     feature_df.to_csv(\"C:\\\\kaggle\\\\ОбучИгра\\\\feature.csv\", index=False)\n\n# def deftarget():\n#     global targets\n#     targets = pd.read_csv('C:\\\\kaggle\\\\ОбучИгра\\\\train_labels.csv')\n#     targets['q'] = targets['session_id'].apply(lambda x: int(x.split('_')[-1][1:]))\n#     targets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]))\n\n# # _________________ FURTHER TESTING AND SORTING FEATURES BY QUALITY ______________\n\n# def def_delt_time():\n#     global train\n#     train.sort_values(by=['session_id', 'elapsed_time'], inplace=True)\n#     train['delt_time'] = train['elapsed_time'].diff(1)\n#     train['delt_time'].fillna(0, inplace=True)\n#     train['delt_time'].clip(0, 100000, inplace=True)\n\n# def feature_engineer(row_f):\n#     global train, new_train\n#     # tip = row_f['tip']\n#     col1 = row_f['col1']\n#     val1 = row_f['val1']\n#     l = len(new_train.columns)\n#     if row_f['kol_col'] == 1: \n#         maska = (train[col1] == val1)\n#         new_train[f'{l}'] = train[maska].groupby(['session_id'])['delt_time'].sum()\n#         new_train[f'{l+1}'] = train[maska].groupby(['session_id'])['index'].count()\n#     elif row_f['kol_col'] == 2: \n#         col2 = row_f['col2']\n#         val2 = row_f['val2']\n#         maska = (train[col1] == val1) & (train[col2] == val2)\n#         new_train[f'{l}'] = train[maska].groupby(['session_id'])['delt_time'].sum()\n#         new_train[f'{l+1}'] = train[maska].groupby(['session_id'])['index'].count()\n\n\n# def feature_engineer_new(feature_rez_not_0):\n#     global train, new_train\n#     for i, row_f in feature_rez_not_0.iterrows(): \n#         feature_engineer(row_f)\n\n#     # tip = row_f['tip']\n#     # col1 = row_f['col1']\n#     # val1 = row_f['val1']\n#     # if row_f['kol_col'] == 1: \n#     #     maska = (train[col1] == val1)\n#     #     new_train['1'] = train[maska].groupby(['session_id'])['delt_time'].sum()\n#     #     new_train['2'] = train[maska].groupby(['session_id'])['index'].count()\n#     # elif row_f['kol_col'] == 2: \n#     #     col2 = row_f['col2']\n#     #     val2 = row_f['val2']\n#     #     maska = (train[col1] == val1) & (train[col2] == val2)\n#     #     new_train['1'] = train[maska].groupby(['session_id'])['delt_time'].sum()\n#     #     new_train['2'] = train[maska].groupby(['session_id'])['index'].count()\n\n# def one_vopros(train_index, test_index):\n#     global new_train, targets, quest, pred_skaz\n#     # TRAIN DATA\n#     train_x = new_train.iloc[train_index]\n#     train_users = train_x.index.values\n#     train_y = targets.loc[targets.q == quest].set_index('session').loc[train_users]\n\n#     # VALID DATA\n#     valid_x = new_train.iloc[test_index]\n#     valid_users = valid_x.index.values\n#     valid_y = targets.loc[targets.q == quest].set_index('session').loc[valid_users]\n\n#     # TRAIN MODEL\n#     model = CatBoostClassifier(\n#         n_estimators = 40,\n#         learning_rate= 0.05,\n#         depth = 6,\n#         l2_leaf_reg = 1.4,\n#     )\n\n#     X = train_x.astype('float32')\n#     Y = train_y['correct']\n#     model.fit(X, Y, verbose=False)\n\n#     # SAVE MODEL, PREDICT VALID OOF\n#     pred_skaz.loc[valid_users, quest] = model.predict_proba(valid_x.astype('float32'))[:, 1]\n#     return pred_skaz\n\n# def preds():\n#     global quest, new_train, targets, pred_skaz, true\n#     ALL_USERS = new_train.index.unique()\n\n#     gkf = KFold(n_splits=5)\n#     pred_skaz = pd.DataFrame(data=np.zeros((len(ALL_USERS), 1)), index=ALL_USERS)\n#     true = pred_skaz.copy()  \n\n\n#     for i, (train_index, test_index) in enumerate(gkf.split(X=new_train)):\n#         # print(' ', i + 1, end='')\n#         pred_skaz = one_vopros(train_index, test_index)\n#     # print()\n\n#     # GET TRUE LABELS\n#     tmp = targets.loc[targets.q == quest].set_index('session').loc[ALL_USERS]\n#     true[quest] = tmp.correct.values\n#     return true\n\n# def otvet():\n#     global quest, pred_skaz, true, best_thresholds, kol_stuk, b_thresholds\n#     best_threshold = 0.61\n#     # Считаем F1 SCORE \n#     tru = true[quest].values\n#     y_pred = pred_skaz[quest].values\n#     m = f1_score(tru, (y_pred > best_threshold).astype('int'), average='macro')\n\n#     best_score = 0\n#     best_threshold = 0\n#     for threshold in np.arange(best_thresholds[quest-1], best_thresholds[quest-1] + 2.5, 0.1):\n#         preds = (y_pred > threshold).astype('int')\n#         m = f1_score(tru, preds, average='macro')\n#         if m > best_score:\n#             best_score = m\n#             best_threshold = threshold\n#     b_thresholds += best_threshold\n#     kol_stuk += 1\n#     print('Результат для вопроса:', quest, 'F1 =', best_score, 'best_threshold =',\n#           best_threshold, 'средний best_threshold =', b_thresholds/kol_stuk)\n#     return best_score\n\n\n# def main_ml():\n#     global train, feature_df, quest, new_train, best_thresholds, kol_stuk, b_thresholds\n#     best_thresholds = [0.6, 0.85, 0.80]\n#     best_thresholds += [0.6, 0.55, 0.65, 0.65, 0.55, 0.65, 0.55, 0.55, 0.65, 0.4]\n#     best_thresholds += [0.6, 0.5, 0.6, 0.6, 0.7]\n#     best_thresholds += [0.55, 0.5, 0.6, 0.6, 0.7]\n\n#     # l_quest = [1, 2, 3]\n#     # l_quest = [4, 5, 6, 7, 8, 9, 10, 11, 12, 13]\n#     l_quest = [14, 15, 16, 17, 18]\n#     col = feature_df.columns\n#     if not ('kach1' in col):\n#         feature_df['kach1'] = 0\n#         feature_df['rez'] = 0\n#     for quest in l_quest: \n#         b_thresholds = 0\n#         kol_stuk = 0\n#         maska = feature_df['quest'] == quest\n#         for i, row_f in feature_df[maska].iterrows(): \n#             print('i=', i, 'kol_col=', row_f['kol_col'], 'col1=', row_f['col1'], row_f['val1'])\n#             new_train = pd.DataFrame(index=train['session_id'].unique(), columns=[])\n#             feature_engineer(row_f) \n#             preds() \n#             m = otvet()\n#             feature_df.loc[i,'kach'] = m\n#             feature_df.loc[i, 'kach1'] = m\n#         feature_df.sort_values(by=['quest','kach'], ascending=False, inplace=True,)\n#         old_max = feature_df[maska]['kach'].max()\n#         feature_df.loc[maska & (feature_df['kach']==old_max), 'rez'] = 1\n#         print(feature_df[feature_df['quest']==quest].head(50))\n#         feature_df.to_csv(\"C:\\\\kaggle\\\\ОбучИгра\\\\feature7.csv\", index=False)\n\n#     for k in [2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13]:#range(5,14,):\n\n#     # if not ('kach1' in col):\n#     #     feature_df['kach1'] = 0\n#         ###### feature_df['rez'] = 0\n#     # for quest in l_quest: \n#     #     b_thresholds = 0\n#     #     kol_stuk = 0\n#     #     maska = feature_df['quest'] == quest\n#     #     for i, row_f in feature_df[maska].iterrows(): \n#     #         print('i=', i, 'kol_col=', row_f['kol_col'], 'col1=', row_f['col1'], row_f['val1'])\n#     #         new_train = pd.DataFrame(index=train['session_id'].unique(), columns=[])\n#     #         feature_engineer(row_f) \n#     #         preds() \n#     #         m = otvet()\n#     #         feature_df.loc[i,'kach'] = m\n#     #         feature_df.loc[i, 'kach1'] = m\n#     #     feature_df.sort_values(by=['quest','kach'], ascending=False, inplace=True,)\n#     #     old_max = feature_df[maska]['kach'].max()\n#     #     feature_df.loc[maska & (feature_df['kach']==old_max), 'rez'] = 1\n#     #     print(feature_df[feature_df['quest']==quest].head(50))\n#     #     feature_df.to_csv(\"C:\\\\kaggle\\\\ОбучИгра\\\\feature9.csv\", index=False)\n\n#     # for k in []:#range(5,14,):\n#         kach = f'kach{k}'\n#         # if kach in col:\n#         #     continue\n#         # feature_df[kach] = 0\n#         for quest in l_quest:  \n#             b_thresholds = 0\n#             kol_stuk = 0\n#             maska = feature_df['quest'] == quest\n#             new_train = pd.DataFrame(index=train['session_id'].unique(), columns=[])\n#             feature_engineer_new(feature_df[maska & (feature_df['rez'] > 0.5)])\n#             new_train0 = new_train.copy()\n#             for i, row_f in feature_df[maska].iterrows():  \n#                 if row_f [kach] < 0.001:\n#                     print('i=', i, 'kol_col=', row_f['kol_col'], 'col1=', row_f['col1'], row_f['val1'])\n#                     feature_engineer(row_f)  \n#                     preds()  \n#                     m = otvet()\n#                     feature_df.loc[i, kach] = m\n#                     new_train = new_train0.copy()\n#             feature_df.sort_values(by=['quest', kach], ascending=False, inplace=True, )\n#             # old_max = feature_df[maska][kach].max()\n#             # feature_df[maska & (feature_df['rez'] < 0.5)].iloc[0,'rez'] = k\n#             ind = feature_df[maska & (feature_df['rez'] < 0.5)].index\n#             feature_df.loc[ind[0],'rez'] = k\n#             print(feature_df[feature_df['quest'] == quest].head(50))\n#             feature_df.to_csv(\"C:\\\\kaggle\\\\ОбучИгра\\\\feature7.csv\", index=False)\n\n\n# feature_df = pd.read_csv(\"C:\\\\kaggle\\\\ОбучИгра\\\\feature6.csv\")\n# # feature_df.to_csv(\"C:\\\\kaggle\\\\ОбучИгра\\\\feature9.csv\", index=False)\n\n\n# feature_df = pd.read_csv(\"C:\\\\kaggle\\\\ОбучИгра\\\\feature8.csv\")\n# # train = read_csv_loc(\"C:\\\\kaggle\\\\ОбучИгра\\\\train_0_4t.csv\")\n# train = read_csv_loc(\"C:\\\\kaggle\\\\ОбучИгра\\\\train_13_22t.csv\")\n# deftarget()\n# def_delt_time()\n# main_ml()","metadata":{},"execution_count":null,"outputs":[]}]}