{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import array\nimport codecs\nimport copy\nimport csv\nimport gc\nimport os\nimport pickle\nimport random\nimport time\nfrom typing import Dict, List, Tuple, Union\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:20:57.301154Z","iopub.execute_input":"2022-08-05T06:20:57.302074Z","iopub.status.idle":"2022-08-05T06:20:57.318649Z","shell.execute_reply.started":"2022-08-05T06:20:57.301961Z","shell.execute_reply":"2022-08-05T06:20:57.317048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MISSING_VALUE = -10000","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:20:57.327937Z","iopub.execute_input":"2022-08-05T06:20:57.328811Z","iopub.status.idle":"2022-08-05T06:20:57.334684Z","shell.execute_reply.started":"2022-08-05T06:20:57.328764Z","shell.execute_reply":"2022-08-05T06:20:57.333404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random.seed(42)\nnp.random.seed(42)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:20:57.336635Z","iopub.execute_input":"2022-08-05T06:20:57.337501Z","iopub.status.idle":"2022-08-05T06:20:57.345722Z","shell.execute_reply.started":"2022-08-05T06:20:57.337457Z","shell.execute_reply":"2022-08-05T06:20:57.344357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-05T06:20:57.350277Z","iopub.execute_input":"2022-08-05T06:20:57.351093Z","iopub.status.idle":"2022-08-05T06:20:57.361176Z","shell.execute_reply.started":"2022-08-05T06:20:57.351056Z","shell.execute_reply":"2022-08-05T06:20:57.359584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_dir = '/kaggle/input/amex-default-prediction'\nassert os.path.isdir(dataset_dir)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:20:57.363400Z","iopub.execute_input":"2022-08-05T06:20:57.364438Z","iopub.status.idle":"2022-08-05T06:20:57.372842Z","shell.execute_reply.started":"2022-08-05T06:20:57.364387Z","shell.execute_reply":"2022-08-05T06:20:57.370372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_subsample(dname: str, probability: float) -> \\\n        Tuple[np.ndarray,\n              Dict[str, Tuple[bool, array.array]],\n              Dict[str, Tuple[bool, array.array]],\n              List[int], Dict[int, List[str]]\n        ]:\n    header = []\n    inputs = []\n    data = dict()\n    set_of_IDs = set()\n    line_idx = 1\n    sample_idx = 0\n    customer_ID_col = -1\n    selected_sample_indices = set()\n    categorical_features = []\n    datetime_features = []\n    numerical_features = []\n    names_of_categorical_features = {'B_30', 'B_38', 'D_114', 'D_116', 'D_117',\n                                     'D_120', 'D_126', 'D_63', 'D_64', 'D_66',\n                                     'D_68'}\n    dicts_of_categorical_features = dict(\n        [(val, []) for val in names_of_categorical_features]\n    )\n    names_of_datetime_features = {'S_2'}\n    fname = os.path.join(dname, 'train_data.csv')\n    with codecs.open(fname, mode='r', encoding='utf-8', errors='ignore') as fp:\n        data_reader = csv.reader(fp, quotechar='\"', delimiter=',')\n        for row in data_reader:\n            if len(row) > 0:\n                err_msg = f'The file {fname}: line {line_idx} is wrong!'\n                if len(header) == 0:\n                    header = copy.copy(row)\n                    ok = True\n                    try:\n                        customer_ID_col = row.index('customer_ID')\n                    except:\n                        ok = False\n                    if not ok:\n                        err_msg += ' The column \"customer_ID\" is not found!'\n                        raise ValueError(err_msg)\n                    for cat_ft in names_of_categorical_features:\n                        try:\n                            cat_idx = row.index(cat_ft)\n                        except:\n                            cat_idx = -1\n                        if cat_idx < 0:\n                            err_msg += f' The column \"{cat_ft}\" is not found!'\n                            ok = False\n                            break\n                    if not ok:\n                        raise ValueError(err_msg)\n                    for datetime_ft in names_of_datetime_features:\n                        try:\n                            datetime_idx = row.index(datetime_ft)\n                        except:\n                            datetime_idx = -1\n                        if datetime_idx < 0:\n                            err_msg += f' The column \"{datetime_ft}\" is not found!'\n                            ok = False\n                            break\n                    if not ok:\n                        raise ValueError(err_msg)\n                    all_ft_names = set()\n                    for col_name in header:\n                        if col_name.startswith('D_'):\n                            all_ft_names.add(col_name)\n                        elif col_name.startswith('S_'):\n                            all_ft_names.add(col_name)\n                        elif col_name.startswith('P_'):\n                            all_ft_names.add(col_name)\n                        elif col_name.startswith('B_'):\n                            all_ft_names.add(col_name)\n                        elif col_name.startswith('R_'):\n                            all_ft_names.add(col_name)\n                    if len(all_ft_names) <= 180:\n                        err_msg += ' Columns number = '\n                        err_msg += f'{len(all_ft_names)}'\n                        err_msg += ' is too small!'\n                        raise ValueError(err_msg)\n                    categorical_features = list(map(\n                        lambda ft2: row.index(ft2),\n                        filter(\n                            lambda ft1: ft1 in names_of_categorical_features,\n                            header\n                        )\n                    ))\n                    datetime_features = list(map(\n                        lambda ft2: row.index(ft2),\n                        filter(\n                            lambda ft1: ft1 in names_of_datetime_features,\n                            header\n                        )\n                    ))\n                    numerical_features = list(map(\n                        lambda ft2: row.index(ft2),\n                        filter(\n                            lambda ft1: ft1 in (all_ft_names - \\\n                                                names_of_datetime_features - \\\n                                                names_of_categorical_features),\n                            header\n                        )\n                    ))\n                else:\n                    if len(header) != len(row):\n                        raise ValueError(err_msg)\n                    ok = True\n                    new_sample = []\n                    for ft_idx in numerical_features:\n                        if len(row[ft_idx]) > 0:\n                            try:\n                                col_value = [float(row[ft_idx])]\n                            except:\n                                col_value = None\n                            if col_value is None:\n                                ok = False\n                                err_msg += f' Column {header[ft_idx]} has '\n                                err_msg += f'impossible value {row[ft_idx]}!'\n                                break\n                        else:\n                            col_value = [MISSING_VALUE]\n                        new_sample += col_value\n                    if not ok:\n                        raise ValueError(err_msg)\n                    for ft_idx in categorical_features:\n                        col_value = []\n                        if len(row[ft_idx]) > 0:\n                            ft_name = header[ft_idx]\n                            ft_val = row[ft_idx]\n                            if not isinstance(ft_val, str):\n                                ft_val = str(ft_val)\n                            if ft_val not in dicts_of_categorical_features[ft_name]:\n                                dicts_of_categorical_features[ft_name].append(ft_val)\n                            col_value.append(\n                                dicts_of_categorical_features[ft_name].index(ft_val)\n                            )\n                        else:\n                            col_value.append(MISSING_VALUE)\n                        new_sample += col_value\n                    for ft_idx in datetime_features:\n                        if len(row[ft_idx]) > 0:\n                            try:\n                                time_obj = time.strptime(row[ft_idx], '%Y-%m-%d')\n                                col_value = [time_obj.tm_mon, time_obj.tm_mday,\n                                             time_obj.tm_wday]\n                            except:\n                                col_value = None\n                            if col_value is None:\n                                ok = False\n                                err_msg += f' Column {header[ft_idx]} has '\n                                err_msg += f'impossible value {row[ft_idx]}!'\n                                break\n                        else:\n                            col_value = [MISSING_VALUE, MISSING_VALUE, MISSING_VALUE]\n                        new_sample += col_value\n                    if not ok:\n                        raise ValueError(err_msg)\n                    new_sample = array.array('f', new_sample)\n                    new_customer_ID = row[customer_ID_col]\n                    set_of_IDs.add(new_customer_ID)\n                    if new_customer_ID in data:\n                        inputs.append(new_sample)\n                        data[new_customer_ID][0].append(len(inputs) - 1)\n                        selected_sample_indices.add(sample_idx)\n                    elif random.random() > (1.0 - probability):\n                        inputs.append(new_sample)\n                        data[new_customer_ID] = [[len(inputs) - 1], 0]\n                        selected_sample_indices.add(sample_idx)\n                    del new_sample\n                    sample_idx += 1\n            if line_idx % 100000 == 0:\n                print(f'{line_idx} lines are processed...')\n                gc.collect()\n            line_idx += 1\n    if (line_idx - 1) % 100000 != 0:\n        print(f'{line_idx - 1} lines are processed...')\n    if (len(data) == 0) or (len(selected_sample_indices) == 0):\n        raise ValueError(f'The file \"{fname}\" is empty!')\n    if len(inputs) != len(selected_sample_indices):\n        raise ValueError(f'The file \"{fname}\" is empty!')\n    print(f'There are {len(set_of_IDs)} unique customers.')\n    print(f'{len(data)} customers ({len(selected_sample_indices)} samples) are selected.')\n    print(f'Number of numerical features is {len(numerical_features)}.')\n    print(f'Number of categorical features is {len(categorical_features)}.')\n    print(f'Number of datetime features is {len(datetime_features) * 3}.')\n    dicts_of_categorical_features_ = dict()\n    for ft_idx in range(len(categorical_features)):\n        ft_name = header[categorical_features[ft_idx]]\n        dicts_of_categorical_features_[len(numerical_features) + ft_idx] = copy.copy(\n            dicts_of_categorical_features[ft_name]\n        )\n    for ft_idx in range(len(datetime_features)):\n        ft_base_name = header[datetime_features[ft_idx]]\n        ft_key = len(numerical_features) + len(categorical_features)\n        ft_key += ft_idx * 3\n        dicts_of_categorical_features_[ft_key] = [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12]\n        dicts_of_categorical_features_[ft_key + 1] = [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11,\n                                                      12, 13, 14, 15, 16, 17, 18, 19, 20,\n                                                      21, 22, 23, 24, 25, 26, 27, 28, 29, 30,\n                                                      31]\n        dicts_of_categorical_features_[ft_key + 2] = [0, 1, 2, 3, 4, 5, 6]\n    del header\n    inputs = np.array(inputs, dtype=np.float32)\n    gc.collect()\n    fname = os.path.join(dname, 'train_labels.csv')\n    true_header = ['customer_ID', 'target']\n    header = []\n    line_idx = 1\n    set_of_target_IDs = set()\n    with codecs.open(fname, mode='r', encoding='utf-8', errors='ignore') as fp:\n        data_reader = csv.reader(fp, quotechar='\"', delimiter=',')\n        for row in data_reader:\n            if len(row) > 0:\n                err_msg = f'The file {fname}: line {line_idx} is wrong!'\n                if len(header) == 0:\n                    header = copy.copy(row)\n                    if header != true_header:\n                        err_msg += f' {header} != {true_header}'\n                        raise ValueError(err_msg)\n                else:\n                    if len(row) != len(header):\n                        raise ValueError(err_msg)\n                    new_customer_ID = row[0]\n                    if new_customer_ID not in set_of_IDs:\n                        err_msg += f' The customer {new_customer_ID} is unknown!'\n                        raise ValueError(err_msg)\n                    if new_customer_ID in set_of_target_IDs:\n                        err_msg += f' The customer {new_customer_ID} is duplicated!'\n                        raise ValueError(err_msg)\n                    set_of_target_IDs.add(new_customer_ID)\n                    try:\n                        target_val = int(row[1])\n                    except:\n                        target_val = -1\n                    if target_val not in {0, 1}:\n                        raise ValueError(err_msg)\n                    if new_customer_ID in data:\n                        data[new_customer_ID][1] = target_val\n            line_idx += 1\n    del set_of_target_IDs\n    gc.collect()\n    IDs_for_validation = set(\n        random.sample(\n            population=sorted(list(data.keys())),\n            k=int(round(0.1 * len(data)))\n        )\n    )\n    IDs_for_training = set(data.keys()) - IDs_for_validation\n    data_for_training = dict()\n    data_for_evaluation = dict()\n    for customer_ID in IDs_for_training:\n        data_for_training[customer_ID] = (\n            True if data[customer_ID][1] > 0 else False,\n            array.array('l', data[customer_ID][0])\n        )\n        del data[customer_ID]\n    for customer_ID in IDs_for_validation:\n        data_for_evaluation[customer_ID] = (\n            True if data[customer_ID][1] > 0 else False,\n            array.array('l', data[customer_ID][0])\n        )\n        del data[customer_ID]\n    del data, IDs_for_training, IDs_for_validation\n    gc.collect()\n    cat_feature_indices = list(range(len(numerical_features), inputs.shape[1]))\n    return inputs, data_for_training, data_for_evaluation, \\\n           cat_feature_indices, dicts_of_categorical_features_","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:20:57.375580Z","iopub.execute_input":"2022-08-05T06:20:57.376670Z","iopub.status.idle":"2022-08-05T06:20:57.432402Z","shell.execute_reply.started":"2022-08-05T06:20:57.376616Z","shell.execute_reply":"2022-08-05T06:20:57.430974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_inputs, trainset, evalset, cat_features, cat_feature_vals = read_subsample(\n    dname=dataset_dir,\n    probability=0.06\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:20:57.435918Z","iopub.execute_input":"2022-08-05T06:20:57.437101Z","iopub.status.idle":"2022-08-05T06:41:57.910900Z","shell.execute_reply.started":"2022-08-05T06:20:57.437044Z","shell.execute_reply":"2022-08-05T06:41:57.909809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'all_inputs.shape = {all_inputs.shape}')\nprint(f'len(trainset) = {len(trainset)}')\nprint(f'len(evalset) = {len(evalset)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:41:57.912543Z","iopub.execute_input":"2022-08-05T06:41:57.912947Z","iopub.status.idle":"2022-08-05T06:41:57.919662Z","shell.execute_reply.started":"2022-08-05T06:41:57.912908Z","shell.execute_reply":"2022-08-05T06:41:57.918576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'cat_features = {cat_features}')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:55:02.322843Z","iopub.execute_input":"2022-08-05T06:55:02.323235Z","iopub.status.idle":"2022-08-05T06:55:02.329154Z","shell.execute_reply.started":"2022-08-05T06:55:02.323198Z","shell.execute_reply":"2022-08-05T06:55:02.327991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Categorical features:')\nfor ft_idx in cat_features:\n    print('    {0:>4}: '.format(ft_idx) + f'{cat_feature_vals[ft_idx]}')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:55:02.330920Z","iopub.execute_input":"2022-08-05T06:55:02.331975Z","iopub.status.idle":"2022-08-05T06:55:02.341119Z","shell.execute_reply.started":"2022-08-05T06:55:02.331936Z","shell.execute_reply":"2022-08-05T06:55:02.339948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pos_class_weight = sum(map(lambda it: int(trainset[it][0]), trainset.keys()))\npos_class_weight /= float(len(trainset))\npos_class_weight *= 100.0\npos_class_weight = int(round(pos_class_weight))\nprint(f'Positive class ratio in the training data is {pos_class_weight}%.')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:55:02.342975Z","iopub.execute_input":"2022-08-05T06:55:02.343496Z","iopub.status.idle":"2022-08-05T06:55:02.467700Z","shell.execute_reply.started":"2022-08-05T06:55:02.343459Z","shell.execute_reply":"2022-08-05T06:55:02.466649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pos_class_weight = sum(map(lambda it: int(evalset[it][0]), evalset.keys()))\npos_class_weight /= float(len(evalset))\npos_class_weight *= 100.0\npos_class_weight = int(round(pos_class_weight))\nprint(f'Positive class ratio in the evaluation data is {pos_class_weight}%.')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:55:02.472379Z","iopub.execute_input":"2022-08-05T06:55:02.472671Z","iopub.status.idle":"2022-08-05T06:55:02.491991Z","shell.execute_reply.started":"2022-08-05T06:55:02.472644Z","shell.execute_reply":"2022-08-05T06:55:02.490942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_history = sorted([len(trainset[it][1]) for it in trainset.keys()])\nn = (len(customers_history) - 1) // 2\nprint(f'Minimal customer history in the training data is {customers_history[0]}.')\nprint(f'Maximal customer history in the training data is {customers_history[-1]}.')\nprint(f'Median customer history in the training data is {customers_history[n]}.')\nprint(f'Mean customer history in the training data is {np.mean(customers_history)}.')\ndel customers_history","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:55:02.493223Z","iopub.execute_input":"2022-08-05T06:55:02.494259Z","iopub.status.idle":"2022-08-05T06:55:02.624940Z","shell.execute_reply.started":"2022-08-05T06:55:02.494200Z","shell.execute_reply":"2022-08-05T06:55:02.623891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_history = sorted([len(evalset[it][1]) for it in evalset.keys()])\nn = (len(customers_history) - 1) // 2\nprint(f'Minimal customer history in the evaluation data is {customers_history[0]}.')\nprint(f'Maximal customer history in the evaluation data is {customers_history[-1]}.')\nprint(f'Median customer history in the evaluation data is {customers_history[n]}.')\nprint(f'Mean customer history in the evaluation data is {np.mean(customers_history)}.')\ndel customers_history","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:55:02.626528Z","iopub.execute_input":"2022-08-05T06:55:02.626908Z","iopub.status.idle":"2022-08-05T06:55:02.649782Z","shell.execute_reply.started":"2022-08-05T06:55:02.626872Z","shell.execute_reply":"2022-08-05T06:55:02.648639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"source_ft_vector_size = all_inputs.shape[1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import OneHotEncoder, RobustScaler","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:55:02.651467Z","iopub.execute_input":"2022-08-05T06:55:02.652222Z","iopub.status.idle":"2022-08-05T06:55:03.347389Z","shell.execute_reply.started":"2022-08-05T06:55:02.652187Z","shell.execute_reply":"2022-08-05T06:55:03.346408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocessor = Pipeline(steps=[\n    (\n        'imputer',\n        SimpleImputer(missing_values=MISSING_VALUE, strategy='median',\n                      copy=False)\n    ),\n    (\n        'transformer',\n        ColumnTransformer(transformers=[\n            (\n                'numerical',\n                RobustScaler(copy=False),\n                list(set(range(all_inputs.shape[1])) - set(cat_features))\n            ),\n            (\n                'categorical',\n                OneHotEncoder(\n                    drop='if_binary',\n                    handle_unknown='ignore',\n                    dtype=np.float32,\n                    sparse=False\n                ),\n                cat_features\n            ),\n        ], n_jobs=1)\n    )\n])","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:55:03.348892Z","iopub.execute_input":"2022-08-05T06:55:03.349225Z","iopub.status.idle":"2022-08-05T06:57:36.443741Z","shell.execute_reply.started":"2022-08-05T06:55:03.349189Z","shell.execute_reply":"2022-08-05T06:57:36.442624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npreprocessor.fit(all_inputs)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_inputs = preprocessor.transform(all_inputs)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_NN_NAME = 'amex-default-pred-1'","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:57:42.002642Z","iopub.execute_input":"2022-08-05T06:57:42.002997Z","iopub.status.idle":"2022-08-05T06:57:42.012667Z","shell.execute_reply.started":"2022-08-05T06:57:42.002962Z","shell.execute_reply":"2022-08-05T06:57:42.011823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(BASE_NN_NAME + '-preprocessor.pkl', 'wb') as fp:\n    pickle.dump(obj=preprocessor, file=fp, protocol=pickle.HIGHEST_PROTOCOL)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'all_inputs.shape = {all_inputs.shape}')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:57:36.445506Z","iopub.execute_input":"2022-08-05T06:57:36.445902Z","iopub.status.idle":"2022-08-05T06:57:36.453654Z","shell.execute_reply.started":"2022-08-05T06:57:36.445865Z","shell.execute_reply":"2022-08-05T06:57:36.452308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processed_ft_vector_size = all_inputs.shape[1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.python.framework import tensor_util\nfrom tensorflow.python.keras.utils import losses_utils, tf_utils\nfrom tensorflow.python.ops.losses import util as tf_losses_util\nimport tensorflow_addons as tfa\n\ntf.random.set_seed(42)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:57:36.455480Z","iopub.execute_input":"2022-08-05T06:57:36.455876Z","iopub.status.idle":"2022-08-05T06:57:41.959151Z","shell.execute_reply.started":"2022-08-05T06:57:36.455839Z","shell.execute_reply":"2022-08-05T06:57:41.958135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class LossFunctionWrapper(tf.keras.losses.Loss):\n    def __init__(self,\n                 fn,\n                 reduction=losses_utils.ReductionV2.AUTO,\n                 name=None,\n                 **kwargs):\n        super(LossFunctionWrapper, self).__init__(reduction=reduction, name=name)\n        self.fn = fn\n        self._fn_kwargs = kwargs\n\n    def call(self, y_true, y_pred):\n        if tensor_util.is_tensor(y_pred) and tensor_util.is_tensor(y_true):\n            y_pred, y_true = tf_losses_util.squeeze_or_expand_dimensions(y_pred, y_true)\n        return self.fn(y_true, y_pred, **self._fn_kwargs)\n\n    def get_config(self):\n        config = {}\n        for k, v in six.iteritems(self._fn_kwargs):\n            config[k] = tf.keras.backend.eval(v) if tf_utils.is_tensor_or_variable(v) \\\n                else v\n        base_config = super(LossFunctionWrapper, self).get_config()\n        return dict(list(base_config.items()) + list(config.items()))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def npairs_loss(labels, feature_vectors):\n    feature_vectors_normalized = tf.math.l2_normalize(feature_vectors, axis=1)\n    logits = tf.divide(\n        tf.matmul(\n            feature_vectors_normalized, tf.transpose(feature_vectors_normalized)\n        ),\n        0.5  # temperature\n    )\n    return tfa.losses.npairs_loss(tf.squeeze(labels), logits)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class NPairsLoss(LossFunctionWrapper):\n    def __init__(self, reduction=losses_utils.ReductionV2.AUTO,\n                 name='n_pairs_loss'):\n        super(NPairsLoss, self).__init__(npairs_loss, name=name,\n                                         reduction=reduction)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_random_seed() -> int:\n    return random.randint(0, 2147483646)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:57:41.961140Z","iopub.execute_input":"2022-08-05T06:57:41.961932Z","iopub.status.idle":"2022-08-05T06:57:41.969755Z","shell.execute_reply.started":"2022-08-05T06:57:41.961892Z","shell.execute_reply":"2022-08-05T06:57:41.968540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_nn(ft_size: int, base_name: str) -> tf.keras.Model:\n    nn_input = tf.keras.layers.Input(\n        shape=(None, ft_size),\n        dtype=tf.float32,\n        name=f'customer_input_{base_name}'\n    )\n    nn_mask = tf.keras.layers.Input(\n        shape=(None,),\n        dtype=tf.bool,\n        name=f'customer_input_mask_{base_name}'\n    )\n    rnn_layer1 = tf.keras.layers.GRU(\n        units=512,\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        recurrent_initializer=\"orthogonal\",\n        bias_initializer=\"zeros\",\n        return_sequences=True,\n        dropout=0.5, recurrent_dropout=0.3,\n        name=f'rnn1_{base_name}'\n    )(nn_input, mask=nn_mask)\n    rnn_layer2 = tf.keras.layers.GRU(\n        units=512,\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        recurrent_initializer=\"orthogonal\",\n        bias_initializer=\"zeros\",\n        return_sequences=True,\n        dropout=0.5, recurrent_dropout=0.3,\n        name=f'rnn2_{base_name}'\n    )(rnn_layer1, mask=nn_mask)\n    rnn_layer3 = tf.keras.layers.GRU(\n        units=512,\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        recurrent_initializer=\"orthogonal\",\n        bias_initializer=\"zeros\",\n        return_sequences=False,\n        dropout=0.5, recurrent_dropout=0.3,\n        name=f'rnn3_{base_name}'\n    )(rnn_layer2, mask=nn_mask)\n    dropout_layer1 = tf.keras.layers.Dropout(\n        rate=0.5,\n        seed=generate_random_seed(),\n        name=f'dropout1_{base_name}'\n    )(rnn_layer3)\n    prj_layer = tf.keras.layers.Dense(\n        units=64, activation=None, use_bias=False,\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        name=f'prj_{base_name}'\n    )(dropout_layer1)\n    hidden_layer1 = tf.keras.layers.Dense(\n        units=512,\n        activation='tanh',\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        bias_initializer=\"zeros\",\n        name=f'hidden1_{base_name}'\n    )(dropout_layer1)\n    dropout_layer2 = tf.keras.layers.Dropout(\n        rate=0.5,\n        seed=generate_random_seed(),\n        name=f'dropout2_{base_name}'\n    )(hidden_layer1)\n    hidden_layer2 = tf.keras.layers.Dense(\n        units=512,\n        activation='tanh',\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        bias_initializer=\"zeros\",\n        name=f'hidden2_{base_name}'\n    )(dropout_layer2)\n    dropout_layer3 = tf.keras.layers.Dropout(\n        rate=0.5,\n        seed=generate_random_seed(),\n        name=f'dropout3_{base_name}'\n    )(hidden_layer2)\n    hidden_layer3 = tf.keras.layers.Dense(\n        units=512,\n        activation='tanh',\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        bias_initializer=\"zeros\",\n        name=f'hidden3_{base_name}'\n    )(dropout_layer3)\n    sum_layer1 = tf.keras.layers.Add(\n        name=f'add1_{base_name}'\n    )([hidden_layer3, hidden_layer1])\n    dropout_layer4 = tf.keras.layers.Dropout(\n        rate=0.5,\n        seed=generate_random_seed(),\n        name=f'dropout4_{base_name}'\n    )(sum_layer1)\n    hidden_layer4 = tf.keras.layers.Dense(\n        units=512,\n        activation='tanh',\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        bias_initializer=\"zeros\",\n        name=f'hidden4_{base_name}'\n    )(dropout_layer4)\n    dropout_layer5 = tf.keras.layers.Dropout(\n        rate=0.5,\n        seed=generate_random_seed(),\n        name=f'dropout5_{base_name}'\n    )(hidden_layer4)\n    hidden_layer5 = tf.keras.layers.Dense(\n        units=512,\n        activation='tanh',\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        bias_initializer=\"zeros\",\n        name=f'hidden5_{base_name}'\n    )(dropout_layer5)\n    dropout_layer6 = tf.keras.layers.Dropout(\n        rate=0.5,\n        seed=generate_random_seed(),\n        name=f'dropout6_{base_name}'\n    )(hidden_layer5)\n    hidden_layer6 = tf.keras.layers.Dense(\n        units=512,\n        activation='tanh',\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        bias_initializer=\"zeros\",\n        name=f'hidden6_{base_name}'\n    )(dropout_layer6)\n    sum_layer2 = tf.keras.layers.Add(\n        name=f'add2_{base_name}'\n    )([hidden_layer6, hidden_layer4])\n    dropout_layer7 = tf.keras.layers.Dropout(\n        rate=0.5,\n        seed=generate_random_seed(),\n        name=f'dropout7_{base_name}'\n    )(sum_layer2)\n    cls_layer = tf.keras.layers.Dense(\n        units=1,\n        activation='sigmoid',\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        bias_initializer=\"zeros\",\n        name=f'output_{base_name}'\n    )(dropout_layer7)\n    classifier = tf.keras.Model(\n        inputs=[nn_input, nn_mask],\n        outputs=[cls_layer, prj_layer],\n        name=f'Classifier_{base_name}'\n    )\n    radam = tfa.optimizers.RectifiedAdam(learning_rate=1e-3)\n    ranger = tfa.optimizers.Lookahead(radam, sync_period=6, slow_step_size=0.5)\n    losses = {\n        f'output_{base_name}': tf.keras.losses.BinaryCrossentropy(label_smoothing=0.001),\n        f'prj_{base_name}': NPairsLoss()\n    }\n    loss_weights = {\n        f'output_{base_name}': 1.0,\n        f'prj_{base_name}': 0.7\n    }\n    metrics = {\n        f'output_{base_name}': [\n            tf.keras.metrics.AUC(name='auc'),\n            tf.keras.metrics.BinaryAccuracy()\n        ]\n    }\n    classifier.compile(\n        optimizer=ranger,\n        loss=losses,\n        loss_weights=loss_weights,\n        metrics=metrics\n    )\n    return classifier","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:57:41.971556Z","iopub.execute_input":"2022-08-05T06:57:41.972299Z","iopub.status.idle":"2022-08-05T06:57:41.985784Z","shell.execute_reply.started":"2022-08-05T06:57:41.972257Z","shell.execute_reply":"2022-08-05T06:57:41.984790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TrainsetGenerator(tf.keras.utils.Sequence):\n    def __init__(self, inputs: np.ndarray,\n                 data_for_training: Dict[str, Tuple[bool, array.array]],\n                 batch_size: int, for_training: bool):\n        self.inputs = inputs\n        self.data_for_training = data_for_training\n        self.batch_size = batch_size\n        self.for_training = for_training\n        self.customer_IDs_ = sorted(list(data_for_training.keys()))\n        self.positive_customer_IDs_ = list(filter(\n            lambda customer_ID: data_for_training[customer_ID][0],\n            self.customer_IDs_\n        ))\n        self.negative_customer_IDs_ = list(filter(\n            lambda customer_ID: not data_for_training[customer_ID][0],\n            self.customer_IDs_\n        ))\n        assert len(set(self.positive_customer_IDs_) & set(self.negative_customer_IDs_)) == 0\n        assert len(set(self.positive_customer_IDs_) | set(self.negative_customer_IDs_)) == len(self.customer_IDs_)\n    \n    def __len__(self):\n        return int(np.ceil(len(self.customer_IDs_) / self.batch_size))\n    \n    def __getitem__(self, idx):\n        if self.for_training:\n            customer_IDs_for_batch = random.sample(\n                population=self.positive_customer_IDs_,\n                k=self.batch_size // 2\n            )\n            customer_IDs_for_batch += random.sample(\n                population=self.negative_customer_IDs_,\n                k=self.batch_size - (self.batch_size // 2)\n            )\n            random.shuffle(customer_IDs_for_batch)\n        else:\n            batch_start = idx * self.batch_size\n            batch_end = min(len(self.customer_IDs_), batch_start + self.batch_size)\n            customer_IDs_for_batch = self.customer_IDs_[batch_start:batch_end]\n        batch_size = len(customer_IDs_for_batch)\n        max_seq_len = max(map(\n            lambda customer_ID: len(self.data_for_training[customer_ID][1]),\n            customer_IDs_for_batch\n        ))\n        X = [\n            np.zeros((batch_size, max_seq_len, self.inputs.shape[1]), dtype=np.float32),\n            np.zeros((batch_size, max_seq_len), dtype=np.bool_)\n        ]\n        y = np.zeros((batch_size,), dtype=np.float32)\n        for idx, customer_ID in enumerate(customer_IDs_for_batch):\n            customer_data = self.data_for_training[customer_ID]\n            if customer_data[0]:\n                y[idx] = 1.0\n            for t, input_idx in enumerate(customer_data[1]):\n                X[0][idx, t] = self.inputs[input_idx]\n                X[1][idx, t] = True\n        return X, [y, y]","metadata":{"execution":{"iopub.status.busy":"2022-08-05T07:11:16.849276Z","iopub.execute_input":"2022-08-05T07:11:16.850699Z","iopub.status.idle":"2022-08-05T07:11:16.866796Z","shell.execute_reply.started":"2022-08-05T07:11:16.850580Z","shell.execute_reply":"2022-08-05T07:11:16.865773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnn = build_nn(ft_size=all_inputs.shape[1], base_name=BASE_NN_NAME)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:57:42.013844Z","iopub.execute_input":"2022-08-05T06:57:42.014150Z","iopub.status.idle":"2022-08-05T06:58:28.432687Z","shell.execute_reply.started":"2022-08-05T06:57:42.014125Z","shell.execute_reply":"2022-08-05T06:58:28.431658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:58:28.434014Z","iopub.execute_input":"2022-08-05T06:58:28.434354Z","iopub.status.idle":"2022-08-05T06:58:28.442297Z","shell.execute_reply.started":"2022-08-05T06:58:28.434320Z","shell.execute_reply":"2022-08-05T06:58:28.441004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(nn, BASE_NN_NAME + '.png', show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:58:28.443566Z","iopub.execute_input":"2022-08-05T06:58:28.444676Z","iopub.status.idle":"2022-08-05T06:58:29.494960Z","shell.execute_reply.started":"2022-08-05T06:58:28.444637Z","shell.execute_reply":"2022-08-05T06:58:29.493623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MINIBATCH_SIZE = 512","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainset_gen = TrainsetGenerator(\n    inputs=all_inputs,\n    data_for_training=trainset,\n    batch_size=MINIBATCH_SIZE,\n    for_training=True\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:58:29.496788Z","iopub.execute_input":"2022-08-05T06:58:29.497539Z","iopub.status.idle":"2022-08-05T06:58:29.651548Z","shell.execute_reply.started":"2022-08-05T06:58:29.497496Z","shell.execute_reply":"2022-08-05T06:58:29.650425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nX_batch, y_batch = trainset_gen[0]\nprint(X_batch[0])\nprint(X_batch[1])\nprint(y_batch)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T06:58:29.653380Z","iopub.execute_input":"2022-08-05T06:58:29.654302Z","iopub.status.idle":"2022-08-05T06:58:29.681805Z","shell.execute_reply.started":"2022-08-05T06:58:29.654260Z","shell.execute_reply":"2022-08-05T06:58:29.679846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evalset_gen = TrainsetGenerator(\n    inputs=all_inputs,\n    data_for_training=evalset,\n    batch_size=MINIBATCH_SIZE,\n    for_training=False\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T07:11:30.983279Z","iopub.execute_input":"2022-08-05T07:11:30.983923Z","iopub.status.idle":"2022-08-05T07:11:31.007969Z","shell.execute_reply.started":"2022-08-05T07:11:30.983875Z","shell.execute_reply":"2022-08-05T07:11:31.006893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nX_batch, y_batch = evalset_gen[0]\nprint(X_batch[0])\nprint(X_batch[1])\nprint(y_batch)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T07:11:33.538622Z","iopub.execute_input":"2022-08-05T07:11:33.539011Z","iopub.status.idle":"2022-08-05T07:11:33.551458Z","shell.execute_reply.started":"2022-08-05T07:11:33.538979Z","shell.execute_reply":"2022-08-05T07:11:33.550350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\nimport umap\n\ndef reduce_dimensions_of_data(datagen: TrainsetGenerator,\n                              fe: tf.keras.Model) -> Tuple[np.ndarray, np.ndarray]:\n    features = []\n    targets = []\n    for batch_idx in range(len(datagen)):\n        X, y = datagen[batch_idx]\n        output = fe.predict_on_batch(X)\n        if isinstance(output[1], np.ndarray):\n            features.append(output[1])\n        else:\n            features.append(output[1].numpy())\n        targets.append(y[0])\n    features = np.vstack(features)\n    targets = np.concatenate(targets)\n    preprocessed_features = Pipeline(\n        steps=[\n            ('scaler', StandardScaler()),\n            ('pca', PCA(n_components=features.shape[1] // 3,\n                        random_state=42))\n        ]\n    ).fit_transform(features)\n    print('Features are preprocessed.')\n    reduced_features = umap.UMAP(\n        low_memory=False,\n        n_jobs=-1,\n        random_state=42,\n        verbose=True\n    ).fit_transform(preprocessed_features)\n    print('Feature space is reduced.')\n    del preprocessed_features\n    return reduced_features, targets","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_dimensions_of_submission_data(features: np.ndarray) -> np.ndarray:\n    preprocessed_features = Pipeline(\n        steps=[\n            ('scaler', StandardScaler()),\n            ('pca', PCA(n_components=features.shape[1] // 3,\n                        random_state=42))\n        ]\n    ).fit_transform(features)\n    print('Features are preprocessed.')\n    reduced_features = umap.UMAP(\n        low_memory=False,\n        n_jobs=-1,\n        random_state=42,\n        verbose=True\n    ).fit_transform(preprocessed_features)\n    print('Feature space is reduced.')\n    del preprocessed_features\n    return reduced_features, targets","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndef show_projections(X: np.ndarray, y: np.ndarray, figure_id: int,\n                     additional: str=''):\n    plt.figure(num=figure_id, figsize=(9, 9))\n    indices_of_negative_classes = list(filter(\n        lambda sample_idx: y[sample_idx] == 0,\n        range(len(y))\n    ))\n    indices_of_positive_classes = list(filter(\n        lambda sample_idx: y[sample_idx] > 0,\n        range(len(y))\n    ))\n    if len(indices_of_negative_classes) > len(indices_of_positive_classes):\n        positive_markersize = 6\n        negative_markersize = 4\n    else:\n        positive_markersize = 4\n        negative_markersize = 6\n    xy = X[indices_of_negative_classes]\n    plt.plot(xy[:, 0], xy[:, 1], 'o', color='g', markersize=negative_markersize,\n             label='Negative samples')\n    indices_of_positive_classes = list(filter(\n        lambda sample_idx: y[sample_idx] > 0,\n        range(len(y))\n    ))\n    xy = X[indices_of_positive_classes]\n    plt.plot(xy[:, 0], xy[:, 1], 'o', color='r', markersize=positive_markersize,\n             label='Positive samples')\n    if len(additional) > 0:\n        plt.title('RNN projections ' + additional)\n    else:\n        plt.title('RNN projections')\n    plt.legend(loc='best')\n    plt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_emb, y_emb = reduce_dimensions_of_data(evalset_gen, nn)\nshow_projections(X_emb, y_emb, 2, 'before training')\ndel X_emb, y_emb","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callbacks = [\n    tf.keras.callbacks.EarlyStopping(\n        monitor=f'val_output_{BASE_NN_NAME}_auc',\n        patience=7,\n        mode='max',\n        verbose=1\n    ),\n    tf.keras.callbacks.ModelCheckpoint(\n        filepath=BASE_NN_NAME + '.h5',\n        monitor=f'val_output_{BASE_NN_NAME}_auc',\n        mode='max',\n        save_best_only=True, save_weights_only=True,\n        verbose=1\n    ),\n    tfa.callbacks.TimeStopping(\n        seconds=3600,\n        verbose=1\n    )\n]","metadata":{"execution":{"iopub.status.busy":"2022-08-05T07:11:39.269473Z","iopub.execute_input":"2022-08-05T07:11:39.270078Z","iopub.status.idle":"2022-08-05T08:16:37.912883Z","shell.execute_reply.started":"2022-08-05T07:11:39.270041Z","shell.execute_reply":"2022-08-05T08:16:37.910667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnn.fit(trainset_gen, validation_data=evalset_gen,\n       epochs=1000, callbacks=callbacks)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn.load_weights(BASE_NN_NAME + '.h5')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_emb, y_emb = reduce_dimensions_of_data(evalset_gen, nn)\nshow_projections(X_emb, y_emb, 3, 'after training')\ndel X_emb, y_emb","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del trainset_gen, evalset_gen, callbacks","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del all_inputs, trainset, evalset","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_inputs, trainset, evalset, cat_features_, cat_feature_vals_ = read_subsample(\n    dname=dataset_dir,\n    probability=0.03\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert cat_features_ == cat_features\nassert cat_feature_vals_ == cat_feature_vals\nassert source_ft_vector_size == all_inputs.shape[1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_inputs = preprocessor.transform(all_inputs)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert processed_ft_vector_size == all_inputs.shape[1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_NN_NAME = 'amex-default-pred-2'","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnn2 = build_nn(ft_size=all_inputs.shape[1], base_name=BASE_NN_NAME)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn2.summary()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(nn2, BASE_NN_NAME + '.png', show_shapes=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainset_gen = TrainsetGenerator(\n    inputs=all_inputs,\n    data_for_training=trainset,\n    batch_size=MINIBATCH_SIZE,\n    for_training=True\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evalset_gen = TrainsetGenerator(\n    inputs=all_inputs,\n    data_for_training=evalset,\n    batch_size=MINIBATCH_SIZE,\n    for_training=False\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callbacks = [\n    tf.keras.callbacks.EarlyStopping(\n        monitor=f'val_output_{BASE_NN_NAME}_auc',\n        patience=7,\n        mode='max',\n        verbose=1\n    ),\n    tf.keras.callbacks.ModelCheckpoint(\n        filepath=BASE_NN_NAME + '.h5',\n        monitor=f'val_output_{BASE_NN_NAME}_auc',\n        mode='max',\n        save_best_only=True, save_weights_only=True,\n        verbose=1\n    ),\n    tfa.callbacks.TimeStopping(\n        seconds=3600,\n        verbose=1\n    )\n]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnn2.fit(trainset_gen, validation_data=evalset_gen,\n        epochs=1000, callbacks=callbacks)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn2.load_weights(BASE_NN_NAME + '.h5')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del trainset_gen, evalset_gen, callbacks","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del all_inputs, trainset, evalset","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_inputs, trainset, evalset, cat_features_, cat_feature_vals_ = read_subsample(\n    dname=dataset_dir,\n    probability=0.03\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert cat_features_ == cat_features\nassert cat_feature_vals_ == cat_feature_vals\nassert source_ft_vector_size == all_inputs.shape[1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_inputs = preprocessor.transform(all_inputs)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert processed_ft_vector_size == all_inputs.shape[1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_NN_NAME = 'amex-default-pred-3'","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnn3 = build_nn(ft_size=all_inputs.shape[1], base_name=BASE_NN_NAME)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn3.summary()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(nn3, BASE_NN_NAME + '.png', show_shapes=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainset_gen = TrainsetGenerator(\n    inputs=all_inputs,\n    data_for_training=trainset,\n    batch_size=MINIBATCH_SIZE,\n    for_training=True\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evalset_gen = TrainsetGenerator(\n    inputs=all_inputs,\n    data_for_training=evalset,\n    batch_size=MINIBATCH_SIZE,\n    for_training=False\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callbacks = [\n    tf.keras.callbacks.EarlyStopping(\n        monitor=f'val_output_{BASE_NN_NAME}_auc',\n        patience=7,\n        mode='max',\n        verbose=1\n    ),\n    tf.keras.callbacks.ModelCheckpoint(\n        filepath=BASE_NN_NAME + '.h5',\n        monitor=f'val_output_{BASE_NN_NAME}_auc',\n        mode='max',\n        save_best_only=True, save_weights_only=True,\n        verbose=1\n    ),\n    tfa.callbacks.TimeStopping(\n        seconds=3600,\n        verbose=1\n    )\n]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnn3.fit(trainset_gen, validation_data=evalset_gen,\n        epochs=1000, callbacks=callbacks)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn3.load_weights(BASE_NN_NAME + '.h5')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del trainset_gen, evalset_gen, callbacks","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del all_inputs, trainset, evalset","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_inputs, trainset, evalset, cat_features_, cat_feature_vals_ = read_subsample(\n    dname=dataset_dir,\n    probability=0.03\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert cat_features_ == cat_features\nassert cat_feature_vals_ == cat_feature_vals\nassert source_ft_vector_size == all_inputs.shape[1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_inputs = preprocessor.transform(all_inputs)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert processed_ft_vector_size == all_inputs.shape[1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_NN_NAME = 'amex-default-pred-4'","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnn4 = build_nn(ft_size=all_inputs.shape[1], base_name=BASE_NN_NAME)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn4.summary()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(nn4, BASE_NN_NAME + '.png', show_shapes=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainset_gen = TrainsetGenerator(\n    inputs=all_inputs,\n    data_for_training=trainset,\n    batch_size=MINIBATCH_SIZE,\n    for_training=True\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evalset_gen = TrainsetGenerator(\n    inputs=all_inputs,\n    data_for_training=evalset,\n    batch_size=MINIBATCH_SIZE,\n    for_training=False\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callbacks = [\n    tf.keras.callbacks.EarlyStopping(\n        monitor=f'val_output_{BASE_NN_NAME}_auc',\n        patience=7,\n        mode='max',\n        verbose=1\n    ),\n    tf.keras.callbacks.ModelCheckpoint(\n        filepath=BASE_NN_NAME + '.h5',\n        monitor=f'val_output_{BASE_NN_NAME}_auc',\n        mode='max',\n        save_best_only=True, save_weights_only=True,\n        verbose=1\n    ),\n    tfa.callbacks.TimeStopping(\n        seconds=3600,\n        verbose=1\n    )\n]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnn4.fit(trainset_gen, validation_data=evalset_gen,\n        epochs=1000, callbacks=callbacks)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn4.load_weights(BASE_NN_NAME + '.h5')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del trainset_gen, evalset_gen, callbacks","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del all_inputs, trainset, evalset","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_inputs, trainset, evalset, cat_features_, cat_feature_vals_ = read_subsample(\n    dname=dataset_dir,\n    probability=0.03\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert cat_features_ == cat_features\nassert cat_feature_vals_ == cat_feature_vals\nassert source_ft_vector_size == all_inputs.shape[1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_inputs = preprocessor.transform(all_inputs)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert processed_ft_vector_size == all_inputs.shape[1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_NN_NAME = 'amex-default-pred-5'","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnn5 = build_nn(ft_size=all_inputs.shape[1], base_name=BASE_NN_NAME)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn5.summary()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(nn5, BASE_NN_NAME + '.png', show_shapes=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainset_gen = TrainsetGenerator(\n    inputs=all_inputs,\n    data_for_training=trainset,\n    batch_size=MINIBATCH_SIZE,\n    for_training=True\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evalset_gen = TrainsetGenerator(\n    inputs=all_inputs,\n    data_for_training=evalset,\n    batch_size=MINIBATCH_SIZE,\n    for_training=False\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callbacks = [\n    tf.keras.callbacks.EarlyStopping(\n        monitor=f'val_output_{BASE_NN_NAME}_auc',\n        patience=7,\n        mode='max',\n        verbose=1\n    ),\n    tf.keras.callbacks.ModelCheckpoint(\n        filepath=BASE_NN_NAME + '.h5',\n        monitor=f'val_output_{BASE_NN_NAME}_auc',\n        mode='max',\n        save_best_only=True, save_weights_only=True,\n        verbose=1\n    ),\n    tfa.callbacks.TimeStopping(\n        seconds=3600,\n        verbose=1\n    )\n]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnn5.fit(trainset_gen, validation_data=evalset_gen,\n        epochs=1000, callbacks=callbacks)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn5.load_weights(BASE_NN_NAME + '.h5')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del trainset_gen, evalset_gen, callbacks","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del all_inputs, trainset, evalset","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_test_inputs(fname: str, cat_ft_values: Dict[int, List[str]],\n                         ft_size: int):\n    header = []\n    samples = []\n    set_of_IDs = set()\n    line_idx = 1\n    sample_idx = 0\n    customer_ID_col = -1\n    categorical_features = []\n    datetime_features = []\n    numerical_features = []\n    names_of_categorical_features = {'B_30', 'B_38', 'D_114', 'D_116', 'D_117',\n                                     'D_120', 'D_126', 'D_63', 'D_64', 'D_66',\n                                     'D_68'}\n    names_of_datetime_features = {'S_2'}\n    prev_customer_ID = ''\n    with codecs.open(fname, mode='r', encoding='utf-8', errors='ignore') as fp:\n        data_reader = csv.reader(fp, quotechar='\"', delimiter=',')\n        for row in data_reader:\n            if len(row) > 0:\n                err_msg = f'The file {fname}: line {line_idx} is wrong!'\n                if len(header) == 0:\n                    header = copy.copy(row)\n                    ok = True\n                    try:\n                        customer_ID_col = row.index('customer_ID')\n                    except:\n                        ok = False\n                    if not ok:\n                        err_msg += ' The column \"customer_ID\" is not found!'\n                        raise ValueError(err_msg)\n                    for cat_ft in names_of_categorical_features:\n                        try:\n                            cat_idx = row.index(cat_ft)\n                        except:\n                            cat_idx = -1\n                        if cat_idx < 0:\n                            err_msg += f' The column \"{cat_ft}\" is not found!'\n                            ok = False\n                            break\n                    if not ok:\n                        raise ValueError(err_msg)\n                    for datetime_ft in names_of_datetime_features:\n                        try:\n                            datetime_idx = row.index(datetime_ft)\n                        except:\n                            datetime_idx = -1\n                        if datetime_idx < 0:\n                            err_msg += f' The column \"{datetime_ft}\" is not found!'\n                            ok = False\n                            break\n                    if not ok:\n                        raise ValueError(err_msg)\n                    all_ft_names = set()\n                    for col_name in header:\n                        if col_name.startswith('D_'):\n                            all_ft_names.add(col_name)\n                        elif col_name.startswith('S_'):\n                            all_ft_names.add(col_name)\n                        elif col_name.startswith('P_'):\n                            all_ft_names.add(col_name)\n                        elif col_name.startswith('B_'):\n                            all_ft_names.add(col_name)\n                        elif col_name.startswith('R_'):\n                            all_ft_names.add(col_name)\n                    if len(all_ft_names) <= 180:\n                        err_msg += ' Columns number = '\n                        err_msg += f'{len(all_ft_names)}'\n                        err_msg += ' is too small!'\n                        raise ValueError(err_msg)\n                    categorical_features = list(map(\n                        lambda ft2: row.index(ft2),\n                        filter(\n                            lambda ft1: ft1 in names_of_categorical_features,\n                            header\n                        )\n                    ))\n                    datetime_features = list(map(\n                        lambda ft2: row.index(ft2),\n                        filter(\n                            lambda ft1: ft1 in names_of_datetime_features,\n                            header\n                        )\n                    ))\n                    numerical_features = list(map(\n                        lambda ft2: row.index(ft2),\n                        filter(\n                            lambda ft1: ft1 in (all_ft_names - \\\n                                                names_of_datetime_features - \\\n                                                names_of_categorical_features),\n                            header\n                        )\n                    ))\n                else:\n                    if len(header) != len(row):\n                        raise ValueError(err_msg)\n                    ok = True\n                    new_sample = []\n                    for ft_idx in numerical_features:\n                        if len(row[ft_idx]) > 0:\n                            try:\n                                col_value = [float(row[ft_idx])]\n                            except:\n                                col_value = None\n                            if col_value is None:\n                                ok = False\n                                err_msg += f' Column {header[ft_idx]} has '\n                                err_msg += f'impossible value {row[ft_idx]}!'\n                                break\n                        else:\n                            col_value = [MISSING_VALUE]\n                        new_sample += col_value\n                    if not ok:\n                        raise ValueError(err_msg)\n                    for ft_idx in categorical_features:\n                        col_value = []\n                        if len(row[ft_idx]) > 0:\n                            ft_name = header[ft_idx]\n                            ft_key = len(new_sample)\n                            if ft_key not in cat_ft_values:\n                                err_msg += f' Categorical feature {ft_key} is unknown!'\n                                ok = False\n                                break\n                            ft_val = row[ft_idx]\n                            if not isinstance(ft_val, str):\n                                ft_val = str(ft_val)\n                            if ft_val not in cat_ft_values[ft_key]:\n                                ft_val_ = cat_ft_values[ft_key][0]\n                            else:\n                                ft_val_ = cat_ft_values[ft_key].index(ft_val)\n                            col_value.append(ft_val_)\n                        else:\n                            col_value.append(MISSING_VALUE)\n                        new_sample += col_value\n                    if not ok:\n                        raise ValueError(err_msg)\n                    for ft_idx in datetime_features:\n                        if len(row[ft_idx]) > 0:\n                            try:\n                                time_obj = time.strptime(row[ft_idx], '%Y-%m-%d')\n                                col_value = [time_obj.tm_mon, time_obj.tm_mday,\n                                             time_obj.tm_wday]\n                            except:\n                                col_value = None\n                            if col_value is None:\n                                ok = False\n                                err_msg += f' Column {header[ft_idx]} has '\n                                err_msg += f'impossible value {row[ft_idx]}!'\n                                break\n                            ft_key = len(new_sample)\n                            if ft_key not in cat_ft_values:\n                                err_msg += f' Categorical feature {ft_key} is unknown!'\n                                ok = False\n                                break\n                            if ft_val not in cat_ft_values[ft_key]:\n                                ft_val_ = cat_ft_values[ft_key][0]\n                            else:\n                                ft_val_ = cat_ft_values[ft_key].index(ft_val)\n                            col_value[0] = ft_val_\n                            ft_key += 1\n                            if ft_key not in cat_ft_values:\n                                err_msg += f' Categorical feature {ft_key} is unknown!'\n                                ok = False\n                                break\n                            if ft_val not in cat_ft_values[ft_key]:\n                                ft_val_ = cat_ft_values[ft_key][0]\n                            else:\n                                ft_val_ = cat_ft_values[ft_key].index(ft_val)\n                            col_value[1] = ft_val_\n                            ft_key += 1\n                            if ft_key not in cat_ft_values:\n                                err_msg += f' Categorical feature {ft_key} is unknown!'\n                                ok = False\n                                break\n                            if ft_val not in cat_ft_values[ft_key]:\n                                ft_val_ = cat_ft_values[ft_key][0]\n                            else:\n                                ft_val_ = cat_ft_values[ft_key].index(ft_val)\n                            col_value[2] = ft_val_\n                        else:\n                            col_value = [MISSING_VALUE, MISSING_VALUE, MISSING_VALUE]\n                        new_sample += col_value\n                    if not ok:\n                        raise ValueError(err_msg)\n                    new_customer_ID = row[customer_ID_col]\n                    if len(new_customer_ID) == 0:\n                        err_msg += ' Customer ID is empty!'\n                        raise ValueError(err_msg)\n                    if new_customer_ID == prev_customer_ID:\n                        samples.append(new_sample)\n                    else:\n                        if len(prev_customer_ID) > 0:\n                            if len(samples) == 0:\n                                err_msg += f' There are no samples for {prev_customer_ID}.'\n                                raise ValueError(err_msg)\n                            set_of_IDs.add(prev_customer_ID)\n                            samples = np.array(samples, dtype=np.float32)\n                            if samples.shape[1] != ft_size:\n                                err_msg += ' Feature vector size is incorrect! '\n                                err_msg += f'Expected {ft_size}, got {samples.shape[1]}.'\n                                raise ValueError(err_msg)\n                            yield (prev_customer_ID, samples)\n                        if new_customer_ID in set_of_IDs:\n                            err_msg += f' Customer {new_customer_ID} is duplicated!'\n                            raise ValueError(err_msg)\n                        del samples\n                        samples = [new_sample]\n                        prev_customer_ID = new_customer_ID\n                    sample_idx += 1\n                    del new_sample\n            if line_idx % 100000 == 0:\n                print(f'{line_idx} lines are processed...')\n                gc.collect()\n            line_idx += 1\n    if (line_idx - 1) % 100000 != 0:\n        print(f'{line_idx - 1} lines are processed...')\n    if len(prev_customer_ID) > 0:\n        if len(samples) == 0:\n            err_msg += f' There are no samples for {prev_customer_ID}.'\n            raise ValueError(err_msg)\n        set_of_IDs.add(prev_customer_ID)\n        samples = np.array(samples, dtype=np.float32)\n        if samples.shape[1] != ft_size:\n            err_msg += ' Feature vector size is incorrect! '\n            err_msg += f'Expected {ft_size}, got {samples.shape[1]}.'\n            raise ValueError(err_msg)\n        yield (prev_customer_ID, samples)\n    print(f'There are {len(set_of_IDs)} unique customers.')\n    print(f'Number of numerical features is {len(numerical_features)}.')\n    print(f'Number of categorical features is {len(categorical_features)}.')\n    print(f'Number of datetime features is {len(datetime_features) * 3}.')\n    del header, samples","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:48:11.383182Z","iopub.execute_input":"2022-08-05T08:48:11.383668Z","iopub.status.idle":"2022-08-05T08:48:12.086859Z","shell.execute_reply.started":"2022-08-05T08:48:11.383615Z","shell.execute_reply":"2022-08-05T08:48:12.085491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs_for_submission = dict()\ntest_predictions = dict()\ntest_gen = generate_test_inputs(\n    fname=os.path.join(dataset_dir, 'test_data.csv'),\n    cat_ft_values=cat_feature_vals,\n    ft_size=source_ft_vector_size\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T08:48:17.561208Z","iopub.execute_input":"2022-08-05T08:48:17.561572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualization = False\n%%time\nfor cur_ID, cur_data in test_gen:\n    assert len(cur_data.shape) == 2\n    assert cur_data.shape[1] == source_ft_vector_size\n    assert cur_ID not in inputs_for_submission\n    inputs_for_submission[cur_ID] = cur_data\n    if len(inputs_for_submission) >= 20000:\n        all_batch_IDs = sorted(list(inputs_for_submission.keys()))\n        max_inp_len = max(map(\n            lambda it: inputs_for_submission[it].shape[0], all_batch_IDs\n        ))\n        print(f'Number of customers in the batch is {len(all_batch_IDs)}.')\n        print(f'Maximal customer history is {max_inp_len}.')\n        new_mask_for_nn = np.zeros(\n            (len(inputs_for_submission), max_inp_len),\n            dtype=np.bool_\n        )\n        new_sample_for_nn = np.zeros(\n            (len(inputs_for_submission), max_inp_len, processed_ft_vector_size),\n            dtype=np.float32\n        )\n        print(f'new_sample_for_nn.shape = {new_sample_for_nn.shape}')\n        print(f'new_mask_for_nn.shape = {new_mask_for_nn.shape}')\n        for batch_idx, batch_ID in enumerate(all_batch_IDs):\n            customer_history = preprocessor.transform(inputs_for_submission[batch_ID])\n            for time_idx in range(customer_history.shape[0]):\n                new_mask_for_nn[batch_idx, time_idx] = True\n                new_sample_for_nn[batch_idx, time_idx] = customer_history[time_idx]\n        pred = nn.predict([new_sample_for_nn, new_mask_for_nn], batch_size=MINIBATCH_SIZE)\n        probas = pred[0]\n        projections = probas[1]\n        if not visualization:\n            visualization = True\n            X_emb = reduce_dimensions_of_submission_data(projections)\n            y_emb = np.asarray(probas[:, 0] >= 0.5, dtype=np.int32)\n            show_projections(X_emb, y_emb, 4, 'for submission data')\n            del X_emb, y_emb\n        del new_sample_for_nn, new_mask_for_nn, pred\n        gc.collect()\n        new_mask_for_nn = np.zeros(\n            (len(inputs_for_submission), max_inp_len),\n            dtype=np.bool_\n        )\n        new_sample_for_nn = np.zeros(\n            (len(inputs_for_submission), max_inp_len, processed_ft_vector_size),\n            dtype=np.float32\n        )\n        for batch_idx, batch_ID in enumerate(all_batch_IDs):\n            customer_history = preprocessor.transform(inputs_for_submission[batch_ID])\n            for time_idx in range(customer_history.shape[0]):\n                new_mask_for_nn[batch_idx, time_idx] = True\n                new_sample_for_nn[batch_idx, time_idx] = customer_history[time_idx]\n        probas2 = nn2.predict([new_sample_for_nn, new_mask_for_nn], batch_size=MINIBATCH_SIZE)[0]\n        probas3 = nn3.predict([new_sample_for_nn, new_mask_for_nn], batch_size=MINIBATCH_SIZE)[0]\n        probas4 = nn4.predict([new_sample_for_nn, new_mask_for_nn], batch_size=MINIBATCH_SIZE)[0]\n        probas5 = nn5.predict([new_sample_for_nn, new_mask_for_nn], batch_size=MINIBATCH_SIZE)[0]\n        del new_sample_for_nn, new_mask_for_nn\n        gc.collect()\n        for batch_idx, batch_ID in enumerate(all_batch_IDs):\n            test_predictions[batch_ID] = probas[batch_idx, 0]\n            test_predictions[batch_ID] += probas2[batch_idx, 0]\n            test_predictions[batch_ID] += probas3[batch_idx, 0]\n            test_predictions[batch_ID] += probas4[batch_idx, 0]\n            test_predictions[batch_ID] += probas5[batch_idx, 0]\n            test_predictions[batch_ID] /= 5.0\n        del inputs_for_submission\n        gc.collect()\n        inputs_for_submission = dict()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nif len(inputs_for_submission) > 0:\n    all_batch_IDs = sorted(list(inputs_for_submission.keys()))\n    max_inp_len = max(map(\n        lambda it: inputs_for_submission[it].shape[0], all_batch_IDs\n    ))\n    print(f'Number of customers in the batch is {len(all_batch_IDs)}.')\n    print(f'Maximal customer history is {max_inp_len}.')\n    new_mask_for_nn = np.zeros(\n        (len(inputs_for_submission), max_inp_len),\n        dtype=np.bool_\n    )\n    new_sample_for_nn = np.zeros(\n        (len(inputs_for_submission), max_inp_len, processed_ft_vector_size),\n        dtype=np.float32\n    )\n    print(f'new_sample_for_nn.shape = {new_sample_for_nn.shape}')\n    print(f'new_mask_for_nn.shape = {new_mask_for_nn.shape}')\n    for batch_idx, batch_ID in enumerate(all_batch_IDs):\n        customer_history = preprocessor.transform(inputs_for_submission[batch_ID])\n        for time_idx in range(customer_history.shape[0]):\n            new_mask_for_nn[batch_idx, time_idx] = True\n            new_sample_for_nn[batch_idx, time_idx] = customer_history[time_idx]\n    probas = nn.predict([new_sample_for_nn, new_mask_for_nn], batch_size=MINIBATCH_SIZE)[0]\n    probas2 = nn2.predict([new_sample_for_nn, new_mask_for_nn], batch_size=MINIBATCH_SIZE)[0]\n    probas3 = nn3.predict([new_sample_for_nn, new_mask_for_nn], batch_size=MINIBATCH_SIZE)[0]\n    probas4 = nn4.predict([new_sample_for_nn, new_mask_for_nn], batch_size=MINIBATCH_SIZE)[0]\n    probas5 = nn5.predict([new_sample_for_nn, new_mask_for_nn], batch_size=MINIBATCH_SIZE)[0]\n    del new_sample_for_nn, new_mask_for_nn\n    gc.collect()\n    for batch_idx, batch_ID in enumerate(all_batch_IDs):\n        test_predictions[batch_ID] = probas[batch_idx, 0]\n        test_predictions[batch_ID] += probas2[batch_idx, 0]\n        test_predictions[batch_ID] += probas3[batch_idx, 0]\n        test_predictions[batch_ID] += probas4[batch_idx, 0]\n        test_predictions[batch_ID] += probas5[batch_idx, 0]\n        test_predictions[batch_ID] /= 5.0\n    del inputs_for_submission","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_gen\ngc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Number of customers for submission is {len(test_predictions)}.')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with codecs.open('submission.csv', mode='w', encoding='utf-8') as res_fp:\n    data_writer = csv.writer(res_fp, delimiter=',', quotechar='\"')\n    data_writer.writerow(['customer_ID', 'prediction'])\n    for cur_ID in sorted(list(test_predictions.keys())):\n        data_writer.writerow([cur_ID, f'{test_predictions[cur_ID]}'])","metadata":{},"execution_count":null,"outputs":[]}]}