{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import array\nimport codecs\nimport copy\nimport csv\nimport gc\nimport os\nimport pickle\nimport random\nimport time\nfrom typing import Dict, List, Set, Tuple, Union\nimport numpy as np","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MISSING_VALUE = -10000","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random.seed(42)\nnp.random.seed(42)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_dir = '/kaggle/input/amex-default-prediction'\nassert os.path.isdir(dataset_dir)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_subsample(dname: str, probability: float) -> \\\n        Tuple[np.ndarray,\n              Dict[str, Tuple[bool, array.array]],\n              Dict[str, Tuple[bool, array.array]],\n              List[int], Dict[int, List[str]], Set[str]\n        ]:\n    header = []\n    inputs = []\n    data = dict()\n    customer_history = []\n    set_of_IDs = set()\n    line_idx = 1\n    customer_ID_col = -1\n    categorical_features = []\n    datetime_features = []\n    numerical_features = []\n    names_of_categorical_features = {'B_30', 'B_38', 'D_114', 'D_116', 'D_117',\n                                     'D_120', 'D_126', 'D_63', 'D_64', 'D_66',\n                                     'D_68'}\n    dicts_of_categorical_features = dict(\n        [(val, []) for val in names_of_categorical_features]\n    )\n    names_of_datetime_features = {'S_2'}\n    fname = os.path.join(dname, 'train_data.csv')\n    prev_customer_ID = ''\n    with codecs.open(fname, mode='r', encoding='utf-8', errors='ignore') as fp:\n        data_reader = csv.reader(fp, quotechar='\"', delimiter=',')\n        for row in data_reader:\n            if len(row) > 0:\n                err_msg = f'The file {fname}: line {line_idx} is wrong!'\n                if len(header) == 0:\n                    header = copy.copy(row)\n                    ok = True\n                    try:\n                        customer_ID_col = row.index('customer_ID')\n                    except:\n                        ok = False\n                    if not ok:\n                        err_msg += ' The column \"customer_ID\" is not found!'\n                        raise ValueError(err_msg)\n                    for cat_ft in names_of_categorical_features:\n                        try:\n                            cat_idx = row.index(cat_ft)\n                        except:\n                            cat_idx = -1\n                        if cat_idx < 0:\n                            err_msg += f' The column \"{cat_ft}\" is not found!'\n                            ok = False\n                            break\n                    if not ok:\n                        raise ValueError(err_msg)\n                    for datetime_ft in names_of_datetime_features:\n                        try:\n                            datetime_idx = row.index(datetime_ft)\n                        except:\n                            datetime_idx = -1\n                        if datetime_idx < 0:\n                            err_msg += f' The column \"{datetime_ft}\" is not found!'\n                            ok = False\n                            break\n                    if not ok:\n                        raise ValueError(err_msg)\n                    all_ft_names = set()\n                    for col_name in header:\n                        if col_name.startswith('D_'):\n                            all_ft_names.add(col_name)\n                        elif col_name.startswith('S_'):\n                            all_ft_names.add(col_name)\n                        elif col_name.startswith('P_'):\n                            all_ft_names.add(col_name)\n                        elif col_name.startswith('B_'):\n                            all_ft_names.add(col_name)\n                        elif col_name.startswith('R_'):\n                            all_ft_names.add(col_name)\n                    if len(all_ft_names) <= 180:\n                        err_msg += ' Columns number = '\n                        err_msg += f'{len(all_ft_names)}'\n                        err_msg += ' is too small!'\n                        raise ValueError(err_msg)\n                    categorical_features = list(map(\n                        lambda ft2: row.index(ft2),\n                        filter(\n                            lambda ft1: ft1 in names_of_categorical_features,\n                            header\n                        )\n                    ))\n                    datetime_features = list(map(\n                        lambda ft2: row.index(ft2),\n                        filter(\n                            lambda ft1: ft1 in names_of_datetime_features,\n                            header\n                        )\n                    ))\n                    numerical_features = list(map(\n                        lambda ft2: row.index(ft2),\n                        filter(\n                            lambda ft1: ft1 in (all_ft_names - \\\n                                                names_of_datetime_features - \\\n                                                names_of_categorical_features),\n                            header\n                        )\n                    ))\n                else:\n                    if len(header) != len(row):\n                        raise ValueError(err_msg)\n                    ok = True\n                    new_sample = []\n                    for ft_idx in numerical_features:\n                        if len(row[ft_idx]) > 0:\n                            try:\n                                col_value = [float(row[ft_idx])]\n                            except:\n                                col_value = None\n                            if col_value is None:\n                                ok = False\n                                err_msg += f' Column {header[ft_idx]} has '\n                                err_msg += f'impossible value {row[ft_idx]}!'\n                                break\n                        else:\n                            col_value = [MISSING_VALUE]\n                        new_sample += col_value\n                    if not ok:\n                        raise ValueError(err_msg)\n                    for ft_idx in categorical_features:\n                        col_value = []\n                        if len(row[ft_idx]) > 0:\n                            ft_name = header[ft_idx]\n                            ft_val = row[ft_idx]\n                            if not isinstance(ft_val, str):\n                                ft_val = str(ft_val)\n                            if ft_val not in dicts_of_categorical_features[ft_name]:\n                                dicts_of_categorical_features[ft_name].append(ft_val)\n                            col_value.append(\n                                dicts_of_categorical_features[ft_name].index(ft_val)\n                            )\n                        else:\n                            col_value.append(MISSING_VALUE)\n                        new_sample += col_value\n                    for ft_idx in datetime_features:\n                        if len(row[ft_idx]) > 0:\n                            try:\n                                time_obj = time.strptime(row[ft_idx], '%Y-%m-%d')\n                                col_value = [time_obj.tm_mon, time_obj.tm_mday,\n                                             time_obj.tm_wday]\n                            except:\n                                col_value = None\n                            if col_value is None:\n                                ok = False\n                                err_msg += f' Column {header[ft_idx]} has '\n                                err_msg += f'impossible value {row[ft_idx]}!'\n                                break\n                        else:\n                            col_value = [MISSING_VALUE, MISSING_VALUE, MISSING_VALUE]\n                        new_sample += col_value\n                    if not ok:\n                        raise ValueError(err_msg)\n                    new_sample = array.array('f', new_sample)\n                    new_customer_ID = row[customer_ID_col]\n                    if prev_customer_ID != new_customer_ID:\n                        if new_customer_ID in set_of_IDs:\n                            err_msg += f' The customer {new_customer_ID} is duplicated!'\n                            raise ValueError(err_msg)\n                        if len(prev_customer_ID) > 0:\n                            if len(customer_history) == 0:\n                                err_msg += f' The customer {prev_customer_ID} has not a history!'\n                                raise ValueError(err_msg)\n                            if random.random() > (1.0 - probability):\n                                data[prev_customer_ID] = [\n                                    list(range(\n                                        len(inputs),\n                                        len(inputs) + len(customer_history)\n                                    )),\n                                    0\n                                ]\n                                inputs += customer_history\n                        del customer_history\n                        customer_history = []\n                        set_of_IDs.add(new_customer_ID)\n                        prev_customer_ID = new_customer_ID\n                    customer_history.append(new_sample)\n                    del new_sample\n            if line_idx % 100000 == 0:\n                print(f'{line_idx} lines are processed...')\n                gc.collect()\n            line_idx += 1\n    if len(prev_customer_ID) > 0:\n        if len(customer_history) == 0:\n            err_msg += f' The customer {prev_customer_ID} has not a history!'\n            raise ValueError(err_msg)\n        if random.random() > (1.0 - probability):\n            data[prev_customer_ID] = [\n                list(range(\n                    len(inputs),\n                    len(inputs) + len(customer_history)\n                )),\n                0\n            ]\n            inputs += customer_history\n    if (line_idx - 1) % 100000 != 0:\n        print(f'{line_idx - 1} lines are processed...')\n    if len(data) == 0:\n        raise ValueError(f'The file \"{fname}\" is empty!')\n    print(f'There are {len(set_of_IDs)} unique customers.')\n    print(f'{len(data)} customers ({len(inputs)} samples) are selected.')\n    print(f'Number of numerical features is {len(numerical_features)}.')\n    print(f'Number of categorical features is {len(categorical_features)}.')\n    print(f'Number of datetime features is {len(datetime_features) * 3}.')\n    dicts_of_categorical_features_ = dict()\n    for ft_idx in range(len(categorical_features)):\n        ft_name = header[categorical_features[ft_idx]]\n        dicts_of_categorical_features_[len(numerical_features) + ft_idx] = copy.copy(\n            dicts_of_categorical_features[ft_name]\n        )\n    for ft_idx in range(len(datetime_features)):\n        ft_base_name = header[datetime_features[ft_idx]]\n        ft_key = len(numerical_features) + len(categorical_features)\n        ft_key += ft_idx * 3\n        dicts_of_categorical_features_[ft_key] = [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12]\n        dicts_of_categorical_features_[ft_key + 1] = [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11,\n                                                      12, 13, 14, 15, 16, 17, 18, 19, 20,\n                                                      21, 22, 23, 24, 25, 26, 27, 28, 29, 30,\n                                                      31]\n        dicts_of_categorical_features_[ft_key + 2] = [0, 1, 2, 3, 4, 5, 6]\n    del header\n    inputs = np.array(inputs, dtype=np.float32)\n    gc.collect()\n    fname = os.path.join(dname, 'train_labels.csv')\n    true_header = ['customer_ID', 'target']\n    header = []\n    line_idx = 1\n    set_of_target_IDs = set()\n    with codecs.open(fname, mode='r', encoding='utf-8', errors='ignore') as fp:\n        data_reader = csv.reader(fp, quotechar='\"', delimiter=',')\n        for row in data_reader:\n            if len(row) > 0:\n                err_msg = f'The file {fname}: line {line_idx} is wrong!'\n                if len(header) == 0:\n                    header = copy.copy(row)\n                    if header != true_header:\n                        err_msg += f' {header} != {true_header}'\n                        raise ValueError(err_msg)\n                else:\n                    if len(row) != len(header):\n                        raise ValueError(err_msg)\n                    new_customer_ID = row[0]\n                    if new_customer_ID not in set_of_IDs:\n                        err_msg += f' The customer {new_customer_ID} is unknown!'\n                        raise ValueError(err_msg)\n                    if new_customer_ID in set_of_target_IDs:\n                        err_msg += f' The customer {new_customer_ID} is duplicated!'\n                        raise ValueError(err_msg)\n                    set_of_target_IDs.add(new_customer_ID)\n                    try:\n                        target_val = int(row[1])\n                    except:\n                        target_val = -1\n                    if target_val not in {0, 1}:\n                        raise ValueError(err_msg)\n                    if new_customer_ID in data:\n                        data[new_customer_ID][1] = target_val\n            line_idx += 1\n    del set_of_target_IDs\n    gc.collect()\n    IDs_for_validation = set(\n        random.sample(\n            population=sorted(list(data.keys())),\n            k=int(round(0.1 * len(data)))\n        )\n    )\n    IDs_for_training = set(data.keys()) - IDs_for_validation\n    data_for_training = dict()\n    data_for_evaluation = dict()\n    for customer_ID in IDs_for_training:\n        data_for_training[customer_ID] = (\n            True if data[customer_ID][1] > 0 else False,\n            array.array('l', data[customer_ID][0])\n        )\n        del data[customer_ID]\n    for customer_ID in IDs_for_validation:\n        data_for_evaluation[customer_ID] = (\n            True if data[customer_ID][1] > 0 else False,\n            array.array('l', data[customer_ID][0])\n        )\n        del data[customer_ID]\n    del data, IDs_for_training, IDs_for_validation\n    gc.collect()\n    cat_feature_indices = list(range(len(numerical_features), inputs.shape[1]))\n    return inputs, data_for_training, data_for_evaluation, \\\n           cat_feature_indices, dicts_of_categorical_features_, set_of_IDs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_inputs, trainset, evalset, \\\ncat_features, cat_feature_vals, \\\nall_customer_IDs = read_subsample(\n    dname=dataset_dir,\n    probability=0.3\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert len(set(trainset.keys()) & set(evalset.keys())) == 0","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"used_customers = set(trainset.keys())\nused_customers |= set(evalset.keys())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'all_inputs.shape = {all_inputs.shape}')\nprint(f'len(trainset) = {len(trainset)}')\nprint(f'len(evalset) = {len(evalset)}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'cat_features = {cat_features}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Categorical features:')\nfor ft_idx in cat_features:\n    print('    {0:>4}: '.format(ft_idx) + f'{cat_feature_vals[ft_idx]}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pos_class_weight = sum(map(lambda it: int(trainset[it][0]), trainset.keys()))\npos_class_weight /= float(len(trainset))\npos_class_weight *= 100.0\npos_class_weight = int(round(pos_class_weight))\nprint(f'Positive class ratio in the training data is {pos_class_weight}%.')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pos_class_weight = sum(map(lambda it: int(evalset[it][0]), evalset.keys()))\npos_class_weight /= float(len(evalset))\npos_class_weight *= 100.0\npos_class_weight = int(round(pos_class_weight))\nprint(f'Positive class ratio in the evaluation data is {pos_class_weight}%.')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_history = sorted([len(trainset[it][1]) for it in trainset.keys()])\nn = (len(customers_history) - 1) // 2\nprint(f'Minimal customer history in the training data is {customers_history[0]}.')\nprint(f'Maximal customer history in the training data is {customers_history[-1]}.')\nprint(f'Median customer history in the training data is {customers_history[n]}.')\nprint(f'Mean customer history in the training data is {np.mean(customers_history)}.')\ndel customers_history","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers_history = sorted([len(evalset[it][1]) for it in evalset.keys()])\nn = (len(customers_history) - 1) // 2\nprint(f'Minimal customer history in the evaluation data is {customers_history[0]}.')\nprint(f'Maximal customer history in the evaluation data is {customers_history[-1]}.')\nprint(f'Median customer history in the evaluation data is {customers_history[n]}.')\nprint(f'Mean customer history in the evaluation data is {np.mean(customers_history)}.')\ndel customers_history","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"source_ft_vector_size = all_inputs.shape[1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\nfrom sklearn.decomposition import PCA\nfrom sklearn.feature_selection import VarianceThreshold\nfrom sklearn.feature_selection import SelectPercentile, f_classif\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocessor = Pipeline(steps=[\n    (\n        'imputer',\n        SimpleImputer(missing_values=MISSING_VALUE, strategy='median',\n                      copy=False)\n    ),\n    (\n        'transformer',\n        ColumnTransformer(transformers=[\n            (\n                'numerical',\n                Pipeline(steps=[\n                    ('selector', VarianceThreshold(threshold=1e-4)),\n                    ('scaler', StandardScaler(copy=False, with_mean=True, with_std=False)),\n                    ('pca', PCA(copy=False, whiten=True, random_state=42))\n                ]),\n                list(set(range(all_inputs.shape[1])) - set(cat_features))\n            ),\n            (\n                'categorical',\n                OneHotEncoder(\n                    drop='if_binary',\n                    handle_unknown='ignore',\n                    dtype=np.float32,\n                    sparse=False\n                ),\n                cat_features\n            ),\n        ], n_jobs=1)\n    ),\n    (\n        'selector1',\n        VarianceThreshold(threshold=1e-4) \n    ),\n    (\n        'selector2',\n        SelectPercentile(f_classif, percentile=80)\n    )\n])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ny_tmp = np.zeros((all_inputs.shape[0]), dtype=np.int32)\nfor customer_ID in trainset:\n    is_default, sample_indices = trainset[customer_ID]\n    if is_default:\n        for idx in sample_indices:\n            y_tmp[idx] = 1\nfor customer_ID in evalset:\n    is_default, sample_indices = evalset[customer_ID]\n    if is_default:\n        for idx in sample_indices:\n            y_tmp[idx] = 1","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npreprocessor.fit(all_inputs, y_tmp)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del y_tmp","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_inputs = preprocessor.transform(all_inputs)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_NN_NAME = 'amex-default-pred-1'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(BASE_NN_NAME + '-preprocessor.pkl', 'wb') as fp:\n    pickle.dump(obj=preprocessor, file=fp, protocol=pickle.HIGHEST_PROTOCOL)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'all_inputs.shape = {all_inputs.shape}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processed_ft_vector_size = all_inputs.shape[1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.python.framework import tensor_util\nfrom tensorflow.python.keras.utils import losses_utils, tf_utils\nfrom tensorflow.python.ops.losses import util as tf_losses_util\nimport tensorflow_addons as tfa\n\ntf.random.set_seed(42)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class LossFunctionWrapper(tf.keras.losses.Loss):\n    def __init__(self,\n                 fn,\n                 reduction=losses_utils.ReductionV2.AUTO,\n                 name=None,\n                 **kwargs):\n        super(LossFunctionWrapper, self).__init__(reduction=reduction, name=name)\n        self.fn = fn\n        self._fn_kwargs = kwargs\n\n    def call(self, y_true, y_pred):\n        if tensor_util.is_tensor(y_pred) and tensor_util.is_tensor(y_true):\n            y_pred, y_true = tf_losses_util.squeeze_or_expand_dimensions(y_pred, y_true)\n        return self.fn(y_true, y_pred, **self._fn_kwargs)\n\n    def get_config(self):\n        config = {}\n        for k, v in six.iteritems(self._fn_kwargs):\n            config[k] = tf.keras.backend.eval(v) if tf_utils.is_tensor_or_variable(v) \\\n                else v\n        base_config = super(LossFunctionWrapper, self).get_config()\n        return dict(list(base_config.items()) + list(config.items()))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def npairs_loss(labels, feature_vectors):\n    feature_vectors_normalized = tf.math.l2_normalize(feature_vectors, axis=1)\n    logits = tf.divide(\n        tf.matmul(\n            feature_vectors_normalized, tf.transpose(feature_vectors_normalized)\n        ),\n        0.5  # temperature\n    )\n    return tfa.losses.npairs_loss(tf.squeeze(labels), logits)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class NPairsLoss(LossFunctionWrapper):\n    def __init__(self, reduction=losses_utils.ReductionV2.AUTO,\n                 name='n_pairs_loss'):\n        super(NPairsLoss, self).__init__(npairs_loss, name=name,\n                                         reduction=reduction)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_random_seed() -> int:\n    return random.randint(0, 2147483646)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_nn(ft_size: int, base_name: str) -> tf.keras.Model:\n    nn_input = tf.keras.layers.Input(\n        shape=(None, ft_size),\n        dtype=tf.float32,\n        name=f'customer_input_{base_name}'\n    )\n    nn_mask = tf.keras.layers.Input(\n        shape=(None,),\n        dtype=tf.bool,\n        name=f'customer_input_mask_{base_name}'\n    )\n    number_of_rnn_units = 512\n    number_of_hidden_units = 2048\n    number_of_projection_units = 128\n    rnn_layer1 = tf.keras.layers.GRU(\n        units=number_of_rnn_units,\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        recurrent_initializer=\"orthogonal\",\n        bias_initializer=\"zeros\",\n        return_sequences=True,\n        return_state=False,\n        name=f'rnn1_{base_name}'\n    )(nn_input, mask=nn_mask)\n    rnn_layer2 = tf.keras.layers.GRU(\n        units=number_of_rnn_units,\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        recurrent_initializer=\"orthogonal\",\n        bias_initializer=\"zeros\",\n        return_sequences=True,\n        return_state=False,\n        name=f'rnn2_{base_name}'\n    )(rnn_layer1, mask=nn_mask)\n    rnn_layer3 = tf.keras.layers.GRU(\n        units=number_of_rnn_units,\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        recurrent_initializer=\"orthogonal\",\n        bias_initializer=\"zeros\",\n        return_sequences=True,\n        return_state=False,\n        name=f'rnn3_{base_name}'\n    )(rnn_layer2, mask=nn_mask)\n    sum_layer1 = tf.keras.layers.Add(\n        name=f'add1_{base_name}'\n    )([rnn_layer3, rnn_layer1])\n    rnn_layer4 = tf.keras.layers.GRU(\n        units=number_of_rnn_units,\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        recurrent_initializer=\"orthogonal\",\n        bias_initializer=\"zeros\",\n        return_sequences=True,\n        return_state=False,\n        name=f'rnn4_{base_name}'\n    )(sum_layer1, mask=nn_mask)\n    rnn_layer5 = tf.keras.layers.GRU(\n        units=number_of_rnn_units,\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        recurrent_initializer=\"orthogonal\",\n        bias_initializer=\"zeros\",\n        return_sequences=True,\n        return_state=False,\n        name=f'rnn5_{base_name}'\n    )(rnn_layer4, mask=nn_mask)\n    rnn_layer6 = tf.keras.layers.GRU(\n        units=number_of_rnn_units,\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        recurrent_initializer=\"orthogonal\",\n        bias_initializer=\"zeros\",\n        return_sequences=True,\n        return_state=False,\n        name=f'rnn6_{base_name}'\n    )(rnn_layer5, mask=nn_mask)\n    sum_layer2 = tf.keras.layers.Add(\n        name=f'add2_{base_name}'\n    )([rnn_layer6, rnn_layer4])\n    rnn_layer7, state = tf.keras.layers.GRU(\n        units=number_of_rnn_units,\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        recurrent_initializer=\"orthogonal\",\n        bias_initializer=\"zeros\",\n        return_sequences=True,\n        return_state=True,\n        name=f'rnn7_{base_name}'\n    )(sum_layer2, mask=nn_mask)\n    pooling_layer1 = tf.keras.layers.GlobalAveragePooling1D(\n        name=f'pooling1_{base_name}'\n    )(rnn_layer4, mask=nn_mask)\n    pooling_layer2 = tf.keras.layers.GlobalAveragePooling1D(\n        name=f'pooling2_{base_name}'\n    )(rnn_layer7, mask=nn_mask)\n    feature_layer = tf.keras.layers.Concatenate(\n        name=f'concat_{base_name}'\n    )([pooling_layer2, state])\n    dropout_layer1 = tf.keras.layers.Dropout(\n        rate=0.5,\n        seed=generate_random_seed(),\n        name=f'dropout1_{base_name}'\n    )(pooling_layer1)\n    dropout_layer2 = tf.keras.layers.Dropout(\n        rate=0.5,\n        seed=generate_random_seed(),\n        name=f'dropout2_{base_name}'\n    )(feature_layer)\n    prj_layer = tf.keras.layers.Dense(\n        units=number_of_projection_units, activation=None, use_bias=False,\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        name=f'prj_{base_name}'\n    )(dropout_layer1)\n    cls_layer = tf.keras.layers.Dense(\n        units=1,\n        activation='sigmoid',\n        kernel_initializer=tf.keras.initializers.GlorotUniform(\n            seed=generate_random_seed()\n        ),\n        bias_initializer=\"zeros\",\n        name=f'output_{base_name}'\n    )(dropout_layer2)\n    classifier = tf.keras.Model(\n        inputs=[nn_input, nn_mask],\n        outputs=[cls_layer, prj_layer],\n        name=f'Classifier_{base_name}'\n    )\n    radam = tfa.optimizers.RectifiedAdam(learning_rate=3e-4)\n    ranger = tfa.optimizers.Lookahead(radam, sync_period=6, slow_step_size=0.5)\n    losses = {\n        f'output_{base_name}': tf.keras.losses.BinaryCrossentropy(label_smoothing=0.001),\n        f'prj_{base_name}': NPairsLoss()\n    }\n    loss_weights = {\n        f'output_{base_name}': 1.0,\n        f'prj_{base_name}': 0.5\n    }\n    metrics = {\n        f'output_{base_name}': [\n            tf.keras.metrics.AUC(name='auc'),\n            tf.keras.metrics.BinaryAccuracy()\n        ]\n    }\n    classifier.compile(\n        optimizer=ranger,\n        loss=losses,\n        loss_weights=loss_weights,\n        metrics=metrics\n    )\n    return classifier","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TrainsetGenerator(tf.keras.utils.Sequence):\n    def __init__(self, inputs: np.ndarray,\n                 data_for_training: Dict[str, Tuple[bool, array.array]],\n                 batch_size: int, for_training: bool):\n        self.inputs = inputs\n        self.data_for_training = data_for_training\n        self.batch_size = batch_size\n        self.for_training = for_training\n        self.customer_IDs_ = sorted(list(data_for_training.keys()))\n    \n    def __len__(self):\n        return int(np.ceil(len(self.customer_IDs_) / self.batch_size))\n    \n    def __getitem__(self, idx):\n        if self.for_training:\n            customer_IDs_for_batch = random.sample(\n                population=self.customer_IDs_,\n                k=self.batch_size\n            )\n        else:\n            batch_start = idx * self.batch_size\n            batch_end = min(len(self.customer_IDs_), batch_start + self.batch_size)\n            customer_IDs_for_batch = self.customer_IDs_[batch_start:batch_end]\n        batch_size = len(customer_IDs_for_batch)\n        max_seq_len = max(map(\n            lambda customer_ID: len(self.data_for_training[customer_ID][1]),\n            customer_IDs_for_batch\n        ))\n        X = [\n            np.zeros((batch_size, max_seq_len, self.inputs.shape[1]), dtype=np.float32),\n            np.zeros((batch_size, max_seq_len), dtype=np.bool_)\n        ]\n        y = np.zeros((batch_size,), dtype=np.float32)\n        for idx, customer_ID in enumerate(customer_IDs_for_batch):\n            customer_data = self.data_for_training[customer_ID]\n            if customer_data[0]:\n                y[idx] = 1.0\n            samples_of_customer_history = customer_data[1]\n            if (len(samples_of_customer_history) > 2) and self.for_training:\n                n = random.randint(max(2, len(samples_of_customer_history) // 2),\n                                   len(samples_of_customer_history))\n                samples_of_customer_history = samples_of_customer_history[-n:]\n            for t, input_idx in enumerate(samples_of_customer_history):\n                X[0][idx, t] = self.inputs[input_idx]\n                X[1][idx, t] = True\n        return X, [y, y]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnn = build_nn(ft_size=all_inputs.shape[1], base_name=BASE_NN_NAME)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(nn, BASE_NN_NAME + '.png', show_shapes=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MINIBATCH_SIZE = 64","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainset_gen = TrainsetGenerator(\n    inputs=all_inputs,\n    data_for_training=trainset,\n    batch_size=MINIBATCH_SIZE,\n    for_training=True\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nX_batch, y_batch = trainset_gen[0]\nprint(X_batch[0])\nprint(X_batch[1])\nprint(y_batch)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evalset_gen = TrainsetGenerator(\n    inputs=all_inputs,\n    data_for_training=evalset,\n    batch_size=MINIBATCH_SIZE,\n    for_training=False\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nX_batch, y_batch = evalset_gen[0]\nprint(X_batch[0])\nprint(X_batch[1])\nprint(y_batch)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import umap\n\ndef reduce_dimensions_of_data(datagen: TrainsetGenerator,\n                              fe: tf.keras.Model) -> Tuple[np.ndarray, np.ndarray]:\n    features = []\n    targets = []\n    for batch_idx in range(len(datagen)):\n        X, y = datagen[batch_idx]\n        output = fe.predict_on_batch(X)\n        if isinstance(output[1], np.ndarray):\n            features.append(output[1])\n        else:\n            features.append(output[1].numpy())\n        targets.append(y[0])\n    features = np.vstack(features)\n    targets = np.concatenate(targets)\n    preprocessed_features = Pipeline(\n        steps=[\n            ('scaler', StandardScaler()),\n            ('pca', PCA(n_components=features.shape[1] // 3,\n                        random_state=42))\n        ]\n    ).fit_transform(features)\n    print('Features are preprocessed.')\n    reduced_features = umap.UMAP(\n        low_memory=False,\n        n_jobs=-1,\n        random_state=42,\n        verbose=True\n    ).fit_transform(preprocessed_features)\n    print('Feature space is reduced.')\n    del preprocessed_features\n    return reduced_features, targets","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_dimensions_of_submission_data(features: np.ndarray) -> np.ndarray:\n    preprocessed_features = Pipeline(\n        steps=[\n            ('scaler', StandardScaler()),\n            ('pca', PCA(n_components=features.shape[1] // 3,\n                        random_state=42))\n        ]\n    ).fit_transform(features)\n    print('Features are preprocessed.')\n    reduced_features = umap.UMAP(\n        low_memory=False,\n        n_jobs=-1,\n        random_state=42,\n        verbose=True\n    ).fit_transform(preprocessed_features)\n    print('Feature space is reduced.')\n    del preprocessed_features\n    return reduced_features","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndef show_projections(X: np.ndarray, y: np.ndarray, figure_id: int,\n                     additional: str=''):\n    plt.figure(num=figure_id, figsize=(9, 9))\n    indices_of_negative_classes = list(filter(\n        lambda sample_idx: y[sample_idx] == 0,\n        range(len(y))\n    ))\n    indices_of_positive_classes = list(filter(\n        lambda sample_idx: y[sample_idx] > 0,\n        range(len(y))\n    ))\n    if len(indices_of_negative_classes) > len(indices_of_positive_classes):\n        positive_markersize = 6\n        negative_markersize = 4\n    else:\n        positive_markersize = 4\n        negative_markersize = 6\n    xy = X[indices_of_negative_classes]\n    plt.plot(xy[:, 0], xy[:, 1], 'o', color='g', markersize=negative_markersize,\n             label='Negative samples')\n    indices_of_positive_classes = list(filter(\n        lambda sample_idx: y[sample_idx] > 0,\n        range(len(y))\n    ))\n    xy = X[indices_of_positive_classes]\n    plt.plot(xy[:, 0], xy[:, 1], 'o', color='r', markersize=positive_markersize,\n             label='Positive samples')\n    if len(additional) > 0:\n        plt.title('RNN projections ' + additional)\n    else:\n        plt.title('RNN projections')\n    plt.legend(loc='best')\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_emb, y_emb = reduce_dimensions_of_data(evalset_gen, nn)\nshow_projections(X_emb, y_emb, 2, 'before training')\ndel X_emb, y_emb","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callbacks = [\n    tf.keras.callbacks.EarlyStopping(\n        monitor=f'val_output_{BASE_NN_NAME}_auc',\n        patience=3,\n        mode='max',\n        verbose=1\n    ),\n    tf.keras.callbacks.ModelCheckpoint(\n        filepath=BASE_NN_NAME + '.h5',\n        monitor=f'val_output_{BASE_NN_NAME}_auc',\n        mode='max',\n        save_best_only=True, save_weights_only=True,\n        verbose=1\n    ),\n    tfa.callbacks.TimeStopping(\n        seconds=int(round(3600 * 1.3)),\n        verbose=1\n    )\n]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnn.fit(trainset_gen, validation_data=evalset_gen,\n       epochs=1000, callbacks=callbacks)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn.load_weights(BASE_NN_NAME + '.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_emb, y_emb = reduce_dimensions_of_data(evalset_gen, nn)\nshow_projections(X_emb, y_emb, 3, 'after training')\ndel X_emb, y_emb","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del trainset_gen, evalset_gen, callbacks","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prev_customers = set(trainset.keys()) | set(evalset.keys())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del all_inputs, trainset, evalset","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_inputs, trainset, evalset, \\\ncat_features_, cat_feature_vals_, \\\nall_customer_IDs_ = read_subsample(\n    dname=dataset_dir,\n    probability=0.25\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert len(set(trainset.keys()) & set(evalset.keys())) == 0\nassert all_customer_IDs == all_customer_IDs_\nassert cat_features_ == cat_features\nassert cat_feature_vals_ == cat_feature_vals\nassert source_ft_vector_size == all_inputs.shape[1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_customers = set(trainset.keys()) | set(evalset.keys())\noverlap_rate = len(new_customers & prev_customers) / float(len(new_customers))\nprint(f'Overlap rate is {round(overlap_rate * 100.0)}%.')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"used_customers |= set(trainset.keys())\nused_customers |= set(evalset.keys())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_inputs = preprocessor.transform(all_inputs)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert processed_ft_vector_size == all_inputs.shape[1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_NN_NAME = 'amex-default-pred-2'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnn2 = build_nn(ft_size=all_inputs.shape[1], base_name=BASE_NN_NAME)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn2.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(nn2, BASE_NN_NAME + '.png', show_shapes=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainset_gen = TrainsetGenerator(\n    inputs=all_inputs,\n    data_for_training=trainset,\n    batch_size=MINIBATCH_SIZE,\n    for_training=True\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evalset_gen = TrainsetGenerator(\n    inputs=all_inputs,\n    data_for_training=evalset,\n    batch_size=MINIBATCH_SIZE,\n    for_training=False\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callbacks = [\n    tf.keras.callbacks.EarlyStopping(\n        monitor=f'val_output_{BASE_NN_NAME}_auc',\n        patience=3,\n        mode='max',\n        verbose=1\n    ),\n    tf.keras.callbacks.ModelCheckpoint(\n        filepath=BASE_NN_NAME + '.h5',\n        monitor=f'val_output_{BASE_NN_NAME}_auc',\n        mode='max',\n        save_best_only=True, save_weights_only=True,\n        verbose=1\n    ),\n    tfa.callbacks.TimeStopping(\n        seconds=int(round(3600 * 1.3)),\n        verbose=1\n    )\n]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnn2.fit(trainset_gen, validation_data=evalset_gen,\n        epochs=1000, callbacks=callbacks)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn2.load_weights(BASE_NN_NAME + '.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prev_customers = set(trainset.keys()) | set(evalset.keys())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del trainset_gen, evalset_gen, callbacks","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del all_inputs, trainset, evalset","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_inputs, trainset, evalset, \\\ncat_features_, cat_feature_vals_, \\\nall_customer_IDs_ = read_subsample(\n    dname=dataset_dir,\n    probability=0.25\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_customers = set(trainset.keys()) | set(evalset.keys())\noverlap_rate = len(new_customers & prev_customers) / float(len(new_customers))\nprint(f'Overlap rate is {round(overlap_rate * 100.0)}%.')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert len(set(trainset.keys()) & set(evalset.keys())) == 0\nassert all_customer_IDs == all_customer_IDs_\nassert cat_features_ == cat_features\nassert cat_feature_vals_ == cat_feature_vals\nassert source_ft_vector_size == all_inputs.shape[1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"used_customers |= set(trainset.keys())\nused_customers |= set(evalset.keys())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"used_customers_rate = float(len(used_customers))\nused_customers_rate /= float(len(all_customer_IDs))\nused_customers_rate *= 100.0\nprint(f'Total rate of used customers is {round(used_customers_rate)}%.')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_inputs = preprocessor.transform(all_inputs)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert processed_ft_vector_size == all_inputs.shape[1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_NN_NAME = 'amex-default-pred-3'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnn3 = build_nn(ft_size=all_inputs.shape[1], base_name=BASE_NN_NAME)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn3.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(nn3, BASE_NN_NAME + '.png', show_shapes=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainset_gen = TrainsetGenerator(\n    inputs=all_inputs,\n    data_for_training=trainset,\n    batch_size=MINIBATCH_SIZE,\n    for_training=True\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evalset_gen = TrainsetGenerator(\n    inputs=all_inputs,\n    data_for_training=evalset,\n    batch_size=MINIBATCH_SIZE,\n    for_training=False\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callbacks = [\n    tf.keras.callbacks.EarlyStopping(\n        monitor=f'val_output_{BASE_NN_NAME}_auc',\n        patience=3,\n        mode='max',\n        verbose=1\n    ),\n    tf.keras.callbacks.ModelCheckpoint(\n        filepath=BASE_NN_NAME + '.h5',\n        monitor=f'val_output_{BASE_NN_NAME}_auc',\n        mode='max',\n        save_best_only=True, save_weights_only=True,\n        verbose=1\n    ),\n    tfa.callbacks.TimeStopping(\n        seconds=int(round(3600 * 1.3)),\n        verbose=1\n    )\n]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnn3.fit(trainset_gen, validation_data=evalset_gen,\n        epochs=1000, callbacks=callbacks)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn3.load_weights(BASE_NN_NAME + '.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prev_customers = set(trainset.keys()) | set(evalset.keys())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del trainset_gen, evalset_gen, callbacks","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del all_inputs, trainset, evalset","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_inputs, trainset, evalset, \\\ncat_features_, cat_feature_vals_, \\\nall_customer_IDs_ = read_subsample(\n    dname=dataset_dir,\n    probability=0.2\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_customers = set(trainset.keys()) | set(evalset.keys())\noverlap_rate = len(new_customers & prev_customers) / float(len(new_customers))\nprint(f'Overlap rate is {round(overlap_rate * 100.0)}%.')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert len(set(trainset.keys()) & set(evalset.keys())) == 0\nassert all_customer_IDs == all_customer_IDs_\nassert cat_features_ == cat_features\nassert cat_feature_vals_ == cat_feature_vals\nassert source_ft_vector_size == all_inputs.shape[1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"used_customers |= set(trainset.keys())\nused_customers |= set(evalset.keys())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"used_customers_rate = float(len(used_customers))\nused_customers_rate /= float(len(all_customer_IDs))\nused_customers_rate *= 100.0\nprint(f'Total rate of used customers is {round(used_customers_rate)}%.')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nall_inputs = preprocessor.transform(all_inputs)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert processed_ft_vector_size == all_inputs.shape[1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_NN_NAME = 'amex-default-pred-4'","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnn4 = build_nn(ft_size=all_inputs.shape[1], base_name=BASE_NN_NAME)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn4.summary()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(nn4, BASE_NN_NAME + '.png', show_shapes=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainset_gen = TrainsetGenerator(\n    inputs=all_inputs,\n    data_for_training=trainset,\n    batch_size=MINIBATCH_SIZE,\n    for_training=True\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evalset_gen = TrainsetGenerator(\n    inputs=all_inputs,\n    data_for_training=evalset,\n    batch_size=MINIBATCH_SIZE,\n    for_training=False\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callbacks = [\n    tf.keras.callbacks.EarlyStopping(\n        monitor=f'val_output_{BASE_NN_NAME}_auc',\n        patience=3,\n        mode='max',\n        verbose=1\n    ),\n    tf.keras.callbacks.ModelCheckpoint(\n        filepath=BASE_NN_NAME + '.h5',\n        monitor=f'val_output_{BASE_NN_NAME}_auc',\n        mode='max',\n        save_best_only=True, save_weights_only=True,\n        verbose=1\n    ),\n    tfa.callbacks.TimeStopping(\n        seconds=int(round(3600 * 1.3)),\n        verbose=1\n    )\n]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnn4.fit(trainset_gen, validation_data=evalset_gen,\n        epochs=1000, callbacks=callbacks)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nn4.load_weights(BASE_NN_NAME + '.h5')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del used_customers, all_customer_IDs, all_customer_IDs_\ndel cat_features_, cat_feature_vals_","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del trainset_gen, evalset_gen, callbacks","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del all_inputs, trainset, evalset","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_test_inputs(fname: str, cat_ft_values: Dict[int, List[str]],\n                         ft_size: int):\n    header = []\n    samples = []\n    set_of_IDs = set()\n    line_idx = 1\n    sample_idx = 0\n    customer_ID_col = -1\n    categorical_features = []\n    datetime_features = []\n    numerical_features = []\n    names_of_categorical_features = {'B_30', 'B_38', 'D_114', 'D_116', 'D_117',\n                                     'D_120', 'D_126', 'D_63', 'D_64', 'D_66',\n                                     'D_68'}\n    names_of_datetime_features = {'S_2'}\n    prev_customer_ID = ''\n    with codecs.open(fname, mode='r', encoding='utf-8', errors='ignore') as fp:\n        data_reader = csv.reader(fp, quotechar='\"', delimiter=',')\n        for row in data_reader:\n            if len(row) > 0:\n                err_msg = f'The file {fname}: line {line_idx} is wrong!'\n                if len(header) == 0:\n                    header = copy.copy(row)\n                    ok = True\n                    try:\n                        customer_ID_col = row.index('customer_ID')\n                    except:\n                        ok = False\n                    if not ok:\n                        err_msg += ' The column \"customer_ID\" is not found!'\n                        raise ValueError(err_msg)\n                    for cat_ft in names_of_categorical_features:\n                        try:\n                            cat_idx = row.index(cat_ft)\n                        except:\n                            cat_idx = -1\n                        if cat_idx < 0:\n                            err_msg += f' The column \"{cat_ft}\" is not found!'\n                            ok = False\n                            break\n                    if not ok:\n                        raise ValueError(err_msg)\n                    for datetime_ft in names_of_datetime_features:\n                        try:\n                            datetime_idx = row.index(datetime_ft)\n                        except:\n                            datetime_idx = -1\n                        if datetime_idx < 0:\n                            err_msg += f' The column \"{datetime_ft}\" is not found!'\n                            ok = False\n                            break\n                    if not ok:\n                        raise ValueError(err_msg)\n                    all_ft_names = set()\n                    for col_name in header:\n                        if col_name.startswith('D_'):\n                            all_ft_names.add(col_name)\n                        elif col_name.startswith('S_'):\n                            all_ft_names.add(col_name)\n                        elif col_name.startswith('P_'):\n                            all_ft_names.add(col_name)\n                        elif col_name.startswith('B_'):\n                            all_ft_names.add(col_name)\n                        elif col_name.startswith('R_'):\n                            all_ft_names.add(col_name)\n                    if len(all_ft_names) <= 180:\n                        err_msg += ' Columns number = '\n                        err_msg += f'{len(all_ft_names)}'\n                        err_msg += ' is too small!'\n                        raise ValueError(err_msg)\n                    categorical_features = list(map(\n                        lambda ft2: row.index(ft2),\n                        filter(\n                            lambda ft1: ft1 in names_of_categorical_features,\n                            header\n                        )\n                    ))\n                    datetime_features = list(map(\n                        lambda ft2: row.index(ft2),\n                        filter(\n                            lambda ft1: ft1 in names_of_datetime_features,\n                            header\n                        )\n                    ))\n                    numerical_features = list(map(\n                        lambda ft2: row.index(ft2),\n                        filter(\n                            lambda ft1: ft1 in (all_ft_names - \\\n                                                names_of_datetime_features - \\\n                                                names_of_categorical_features),\n                            header\n                        )\n                    ))\n                else:\n                    if len(header) != len(row):\n                        raise ValueError(err_msg)\n                    ok = True\n                    new_sample = []\n                    for ft_idx in numerical_features:\n                        if len(row[ft_idx]) > 0:\n                            try:\n                                col_value = [float(row[ft_idx])]\n                            except:\n                                col_value = None\n                            if col_value is None:\n                                ok = False\n                                err_msg += f' Column {header[ft_idx]} has '\n                                err_msg += f'impossible value {row[ft_idx]}!'\n                                break\n                        else:\n                            col_value = [MISSING_VALUE]\n                        new_sample += col_value\n                    if not ok:\n                        raise ValueError(err_msg)\n                    for ft_idx in categorical_features:\n                        col_value = []\n                        if len(row[ft_idx]) > 0:\n                            ft_name = header[ft_idx]\n                            ft_key = len(new_sample)\n                            if ft_key not in cat_ft_values:\n                                err_msg += f' Categorical feature {ft_key} is unknown!'\n                                ok = False\n                                break\n                            ft_val = row[ft_idx]\n                            if not isinstance(ft_val, str):\n                                ft_val = str(ft_val)\n                            if ft_val not in cat_ft_values[ft_key]:\n                                ft_val_ = cat_ft_values[ft_key][0]\n                            else:\n                                ft_val_ = cat_ft_values[ft_key].index(ft_val)\n                            col_value.append(ft_val_)\n                        else:\n                            col_value.append(MISSING_VALUE)\n                        new_sample += col_value\n                    if not ok:\n                        raise ValueError(err_msg)\n                    for ft_idx in datetime_features:\n                        if len(row[ft_idx]) > 0:\n                            try:\n                                time_obj = time.strptime(row[ft_idx], '%Y-%m-%d')\n                                col_value = [time_obj.tm_mon, time_obj.tm_mday,\n                                             time_obj.tm_wday]\n                            except:\n                                col_value = None\n                            if col_value is None:\n                                ok = False\n                                err_msg += f' Column {header[ft_idx]} has '\n                                err_msg += f'impossible value {row[ft_idx]}!'\n                                break\n                            ft_key = len(new_sample)\n                            if ft_key not in cat_ft_values:\n                                err_msg += f' Categorical feature {ft_key} is unknown!'\n                                ok = False\n                                break\n                            if ft_val not in cat_ft_values[ft_key]:\n                                ft_val_ = cat_ft_values[ft_key][0]\n                            else:\n                                ft_val_ = cat_ft_values[ft_key].index(ft_val)\n                            col_value[0] = ft_val_\n                            ft_key += 1\n                            if ft_key not in cat_ft_values:\n                                err_msg += f' Categorical feature {ft_key} is unknown!'\n                                ok = False\n                                break\n                            if ft_val not in cat_ft_values[ft_key]:\n                                ft_val_ = cat_ft_values[ft_key][0]\n                            else:\n                                ft_val_ = cat_ft_values[ft_key].index(ft_val)\n                            col_value[1] = ft_val_\n                            ft_key += 1\n                            if ft_key not in cat_ft_values:\n                                err_msg += f' Categorical feature {ft_key} is unknown!'\n                                ok = False\n                                break\n                            if ft_val not in cat_ft_values[ft_key]:\n                                ft_val_ = cat_ft_values[ft_key][0]\n                            else:\n                                ft_val_ = cat_ft_values[ft_key].index(ft_val)\n                            col_value[2] = ft_val_\n                        else:\n                            col_value = [MISSING_VALUE, MISSING_VALUE, MISSING_VALUE]\n                        new_sample += col_value\n                    if not ok:\n                        raise ValueError(err_msg)\n                    new_customer_ID = row[customer_ID_col]\n                    if len(new_customer_ID) == 0:\n                        err_msg += ' Customer ID is empty!'\n                        raise ValueError(err_msg)\n                    if new_customer_ID == prev_customer_ID:\n                        samples.append(new_sample)\n                    else:\n                        if len(prev_customer_ID) > 0:\n                            if len(samples) == 0:\n                                err_msg += f' There are no samples for {prev_customer_ID}.'\n                                raise ValueError(err_msg)\n                            set_of_IDs.add(prev_customer_ID)\n                            samples = np.array(samples, dtype=np.float32)\n                            if samples.shape[1] != ft_size:\n                                err_msg += ' Feature vector size is incorrect! '\n                                err_msg += f'Expected {ft_size}, got {samples.shape[1]}.'\n                                raise ValueError(err_msg)\n                            yield (prev_customer_ID, samples)\n                        if new_customer_ID in set_of_IDs:\n                            err_msg += f' Customer {new_customer_ID} is duplicated!'\n                            raise ValueError(err_msg)\n                        del samples\n                        samples = [new_sample]\n                        prev_customer_ID = new_customer_ID\n                    sample_idx += 1\n                    del new_sample\n            if line_idx % 100000 == 0:\n                print(f'{line_idx} lines are processed...')\n                gc.collect()\n            line_idx += 1\n    if (line_idx - 1) % 100000 != 0:\n        print(f'{line_idx - 1} lines are processed...')\n    if len(prev_customer_ID) > 0:\n        if len(samples) == 0:\n            err_msg += f' There are no samples for {prev_customer_ID}.'\n            raise ValueError(err_msg)\n        set_of_IDs.add(prev_customer_ID)\n        samples = np.array(samples, dtype=np.float32)\n        if samples.shape[1] != ft_size:\n            err_msg += ' Feature vector size is incorrect! '\n            err_msg += f'Expected {ft_size}, got {samples.shape[1]}.'\n            raise ValueError(err_msg)\n        yield (prev_customer_ID, samples)\n    print(f'There are {len(set_of_IDs)} unique customers.')\n    print(f'Number of numerical features is {len(numerical_features)}.')\n    print(f'Number of categorical features is {len(categorical_features)}.')\n    print(f'Number of datetime features is {len(datetime_features) * 3}.')\n    del header, samples","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs_for_submission = dict()\ntest_predictions = dict()\ntest_gen = generate_test_inputs(\n    fname=os.path.join(dataset_dir, 'test_data.csv'),\n    cat_ft_values=cat_feature_vals,\n    ft_size=source_ft_vector_size\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nvisualization = False\nfor cur_ID, cur_data in test_gen:\n    assert len(cur_data.shape) == 2\n    assert cur_data.shape[1] == source_ft_vector_size\n    assert cur_ID not in inputs_for_submission\n    inputs_for_submission[cur_ID] = cur_data\n    if len(inputs_for_submission) >= 20000:\n        all_batch_IDs = sorted(list(inputs_for_submission.keys()))\n        max_inp_len = max(map(\n            lambda it: inputs_for_submission[it].shape[0], all_batch_IDs\n        ))\n        print(f'Number of customers in the batch is {len(all_batch_IDs)}.')\n        new_mask_for_nn = np.zeros(\n            (len(inputs_for_submission), max_inp_len),\n            dtype=np.bool_\n        )\n        new_sample_for_nn = np.zeros(\n            (len(inputs_for_submission), max_inp_len, processed_ft_vector_size),\n            dtype=np.float32\n        )\n        print(f'new_sample_for_nn.shape = {new_sample_for_nn.shape}')\n        print(f'new_mask_for_nn.shape = {new_mask_for_nn.shape}')\n        customer_history_lengths = []\n        for batch_idx, batch_ID in enumerate(all_batch_IDs):\n            customer_history = preprocessor.transform(inputs_for_submission[batch_ID])\n            customer_history_lengths.append(customer_history.shape[0])\n            for time_idx in range(customer_history.shape[0]):\n                new_mask_for_nn[batch_idx, time_idx] = True\n                new_sample_for_nn[batch_idx, time_idx] = customer_history[time_idx]\n        customer_history_lengths.sort()\n        assert max_inp_len == customer_history_lengths[-1], f'{max_inp_len} != {customer_history_lengths[-1]}'\n        print('Customer history length:')\n        print(f'  - minimal = {customer_history_lengths[0]};')\n        print(f'  - maximal = {customer_history_lengths[-1]};')\n        print(f'  - mean    = {np.mean(customer_history_lengths)};')\n        print(f'  - median  = {customer_history_lengths[len(customer_history_lengths) // 2]}.')\n        print('')\n        pred = nn.predict([new_sample_for_nn, new_mask_for_nn], batch_size=MINIBATCH_SIZE)\n        assert isinstance(pred, list), f'type(pred) = {type(pred)}'\n        assert len(pred) > 0, 'len(pred) == 0'\n        assert isinstance(pred[0], np.ndarray), f'type(pred[0]) = {type(pred[0])}'\n        assert isinstance(pred[1], np.ndarray), f'type(pred[1]) = {type(pred[1])}'\n        assert len(pred[1].shape) == 2, f'len(pred[1].shape) = {len(pred[1].shape)}'\n        probas = pred[0]\n        projections = pred[1]\n        if not visualization:\n            visualization = True\n            X_emb = reduce_dimensions_of_submission_data(projections)\n            y_emb = np.asarray(probas[:, 0] >= 0.5, dtype=np.int32)\n            show_projections(X_emb, y_emb, 4, 'for submission data')\n            del X_emb, y_emb\n        del new_sample_for_nn, new_mask_for_nn, pred\n        gc.collect()\n        new_mask_for_nn = np.zeros(\n            (len(inputs_for_submission), max_inp_len),\n            dtype=np.bool_\n        )\n        new_sample_for_nn = np.zeros(\n            (len(inputs_for_submission), max_inp_len, processed_ft_vector_size),\n            dtype=np.float32\n        )\n        for batch_idx, batch_ID in enumerate(all_batch_IDs):\n            customer_history = preprocessor.transform(inputs_for_submission[batch_ID])\n            for time_idx in range(customer_history.shape[0]):\n                new_mask_for_nn[batch_idx, time_idx] = True\n                new_sample_for_nn[batch_idx, time_idx] = customer_history[time_idx]\n        probas2 = nn2.predict([new_sample_for_nn, new_mask_for_nn], batch_size=MINIBATCH_SIZE)[0]\n        probas3 = nn3.predict([new_sample_for_nn, new_mask_for_nn], batch_size=MINIBATCH_SIZE)[0]\n        probas4 = nn4.predict([new_sample_for_nn, new_mask_for_nn], batch_size=MINIBATCH_SIZE)[0]\n        del new_sample_for_nn, new_mask_for_nn\n        gc.collect()\n        for batch_idx, batch_ID in enumerate(all_batch_IDs):\n            test_predictions[batch_ID] = probas[batch_idx, 0]\n            test_predictions[batch_ID] += probas2[batch_idx, 0]\n            test_predictions[batch_ID] += probas3[batch_idx, 0]\n            test_predictions[batch_ID] += probas4[batch_idx, 0]\n            test_predictions[batch_ID] /= 4.0\n        del inputs_for_submission\n        gc.collect()\n        inputs_for_submission = dict()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nif len(inputs_for_submission) > 0:\n    all_batch_IDs = sorted(list(inputs_for_submission.keys()))\n    max_inp_len = max(map(\n        lambda it: inputs_for_submission[it].shape[0], all_batch_IDs\n    ))\n    print(f'Number of customers in the batch is {len(all_batch_IDs)}.')\n    print(f'Maximal customer history is {max_inp_len}.')\n    new_mask_for_nn = np.zeros(\n        (len(inputs_for_submission), max_inp_len),\n        dtype=np.bool_\n    )\n    new_sample_for_nn = np.zeros(\n        (len(inputs_for_submission), max_inp_len, processed_ft_vector_size),\n        dtype=np.float32\n    )\n    print(f'new_sample_for_nn.shape = {new_sample_for_nn.shape}')\n    print(f'new_mask_for_nn.shape = {new_mask_for_nn.shape}')\n    customer_history_lengths = []\n    for batch_idx, batch_ID in enumerate(all_batch_IDs):\n        customer_history = preprocessor.transform(inputs_for_submission[batch_ID])\n        customer_history_lengths.append(customer_history.shape[0])\n        for time_idx in range(customer_history.shape[0]):\n            new_mask_for_nn[batch_idx, time_idx] = True\n            new_sample_for_nn[batch_idx, time_idx] = customer_history[time_idx]\n    customer_history_lengths.sort()\n    assert max_inp_len == customer_history_lengths[-1], f'{max_inp_len} != {customer_history_lengths[-1]}'\n    print('Customer history length:')\n    print(f'  - minimal = {customer_history_lengths[0]};')\n    print(f'  - maximal = {customer_history_lengths[-1]};')\n    print(f'  - mean    = {np.mean(customer_history_lengths)};')\n    print(f'  - median  = {customer_history_lengths[len(customer_history_lengths) // 2]}.')\n    probas = nn.predict([new_sample_for_nn, new_mask_for_nn], batch_size=MINIBATCH_SIZE)[0]\n    probas2 = nn2.predict([new_sample_for_nn, new_mask_for_nn], batch_size=MINIBATCH_SIZE)[0]\n    probas3 = nn3.predict([new_sample_for_nn, new_mask_for_nn], batch_size=MINIBATCH_SIZE)[0]\n    probas4 = nn4.predict([new_sample_for_nn, new_mask_for_nn], batch_size=MINIBATCH_SIZE)[0]\n    del new_sample_for_nn, new_mask_for_nn\n    gc.collect()\n    for batch_idx, batch_ID in enumerate(all_batch_IDs):\n        test_predictions[batch_ID] = probas[batch_idx, 0]\n        test_predictions[batch_ID] += probas2[batch_idx, 0]\n        test_predictions[batch_ID] += probas3[batch_idx, 0]\n        test_predictions[batch_ID] += probas4[batch_idx, 0]\n        test_predictions[batch_ID] /= 4.0\n    del inputs_for_submission","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_gen\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Number of customers for submission is {len(test_predictions)}.')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with codecs.open('submission.csv', mode='w', encoding='utf-8') as res_fp:\n    data_writer = csv.writer(res_fp, delimiter=',', quotechar='\"')\n    data_writer.writerow(['customer_ID', 'prediction'])\n    for cur_ID in sorted(list(test_predictions.keys())):\n        data_writer.writerow([cur_ID, f'{test_predictions[cur_ID]}'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}