{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Info\n\nThis Notebook was initially built off of [this Notebook](https://www.kaggle.com/code/ambrosm/amex-keras-quickstart-1-training) by AmbrosM\n\nI mainly changed the data, model structure, optimizer, and loss function\n\nThe loss function uses the oof predictions from an lgbm model to perform [knowledge distillation](https://www.kaggle.com/code/mathormad/knowledge-distillation-with-nn-rankgauss/notebook)\n\nThis kernel over fits alot as does the starter knowledge distillation notebook I built this notebook off of. Unfortunately I have been unable to solve this issue though I have gotten LB 0.794 using this notebook on another run. \n\nCV 0.8045 LB 0.791\n\nThe datasets I use as train datasets and soft targets are essentially the datasets from [this lgbm dart notebook](https://www.kaggle.com/code/ragnar123/amex-lgbm-dart-cv-0-7977) with standard scaling and some additional features","metadata":{}},{"cell_type":"code","source":"%%capture\n!pip install adabelief-tf --no-cache-dir ","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:16:20.868450Z","iopub.execute_input":"2022-07-26T20:16:20.868862Z","iopub.status.idle":"2022-07-26T20:16:33.144517Z","shell.execute_reply.started":"2022-07-26T20:16:20.868832Z","shell.execute_reply":"2022-07-26T20:16:33.143330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport dill as pickle   \nfrom matplotlib import pyplot as plt\nimport random\nimport datetime\nimport math\nimport gc\nimport os\nimport warnings\nimport seaborn as sns\nimport itertools\nimport multiprocessing\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\nfrom matplotlib.ticker import MaxNLocator\nfrom colorama import Fore, Back, Style\nfrom tqdm import tqdm\nimport h5py\n\nfrom sklearn.model_selection import StratifiedKFold,KFold\nfrom sklearn.preprocessing import StandardScaler, QuantileTransformer, OneHotEncoder, PowerTransformer\nfrom sklearn.metrics import roc_curve, roc_auc_score, average_precision_score\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.utils import class_weight \nfrom sklearn.utils.class_weight import compute_class_weight\n\nimport tensorflow_addons as tfa\nimport tensorflow as tf\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '1'\ntf.config.threading.set_inter_op_parallelism_threads(4)\nfrom adabelief_tf import AdaBeliefOptimizer\nfrom tensorflow.keras.models import Model, load_model,model_from_json\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, LearningRateScheduler, EarlyStopping\nfrom tensorflow.keras.layers import Dense, Input, InputLayer, Add, Concatenate, Dropout, BatchNormalization\nfrom tensorflow.keras.utils import plot_model\nfrom tensorflow.keras.losses import binary_crossentropy\nimport tensorflow.keras.backend as K","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-26T20:16:33.146444Z","iopub.execute_input":"2022-07-26T20:16:33.146769Z","iopub.status.idle":"2022-07-26T20:16:33.159089Z","shell.execute_reply.started":"2022-07-26T20:16:33.146739Z","shell.execute_reply":"2022-07-26T20:16:33.158399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true, y_pred, return_components=False) -> float:\n    \"\"\"Amex metric for ndarrays\"\"\"\n    def top_four_percent_captured(df) -> float:\n        \"\"\"Corresponds to the recall for a threshold of 4 %\"\"\"\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(df) -> float:\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(df) -> float:\n        \"\"\"Corresponds to 2 * AUC - 1\"\"\"\n        df2 = pd.DataFrame({'target': df.target, 'prediction': df.target})\n        df2.sort_values('prediction', ascending=False, inplace=True)\n        return weighted_gini(df) / weighted_gini(df2)\n\n    df = pd.DataFrame({'target': y_true.ravel(), 'prediction': y_pred.ravel()})\n    df.sort_values('prediction', ascending=False, inplace=True)\n    g = normalized_weighted_gini(df)\n    d = top_four_percent_captured(df)\n\n    if return_components: return g, d, 0.5 * (g + d)\n    del df\n    gc.collect()\n    return 0.5 * (g + d)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-26T20:16:33.160320Z","iopub.execute_input":"2022-07-26T20:16:33.160826Z","iopub.status.idle":"2022-07-26T20:16:33.192045Z","shell.execute_reply.started":"2022-07-26T20:16:33.160795Z","shell.execute_reply":"2022-07-26T20:16:33.191343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# The model\n\nI experimented with some models and this one did the best by far","metadata":{}},{"cell_type":"code","source":"def my_model(n_inputs):\n    \"\"\"Sequential neural network with a skip connection.\n    \n    Returns a compiled instance of tensorflow.keras.models.Model.\n    \"\"\"\n    activation = 'swish'\n    l1 = 1e-6\n    l2 = 1e-6\n    inputs = Input(shape=(n_inputs, ))\n    x0 = Dense(64, \n               kernel_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n               activity_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n              activation=activation,\n             )(inputs)\n    x1 = Dense(32, \n               kernel_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n               activity_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n          activation=activation,\n         )(x0)\n    x2 = Dense(16, \n               kernel_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n               activity_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n          activation=activation,\n         )(x1)\n    x3 = Dense(8, \n           kernel_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n           activity_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n      activation=activation,\n     )(x2)\n    x4 = Dense(4, \n           kernel_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n           activity_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n      activation=activation,\n     )(x3)\n    x5 = Dense(2, \n           kernel_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n           activity_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n      activation=activation,\n     )(x4)\n    x6 = Dense(1, \n           kernel_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n           activity_regularizer=tf.keras.regularizers.L1L2(l1=l1,l2=l2),\n      activation=activation,\n    )(x5)\n    x = Concatenate()([x0,x1,x2,x3,x4,x5,x6])\n    x = Dropout(0.15)(x)\n    x = Dense(1,\n              activation='sigmoid',\n             )(x)\n    del x0,x1,x2,x3,x4,x5,x6\n    gc.collect()\n    return Model(inputs, x)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:16:33.194028Z","iopub.execute_input":"2022-07-26T20:16:33.194608Z","iopub.status.idle":"2022-07-26T20:16:33.210691Z","shell.execute_reply.started":"2022-07-26T20:16:33.194575Z","shell.execute_reply":"2022-07-26T20:16:33.209736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Cross-validation\n\nWe use a standard cross-validation loop. In the loop, we scale the data and train a model. We use a StratifiedKFold because the data is imbalanced.\n\nI use multiple different seeds (SEEDS) for the cv folds because I want to use this model's oof predictions for ensembling later. Averaging across multiple folds makes the oof predictions more closely resemble the test predictions that the model outputs \n\nI run each seed in a separate version then in the 'Inference' section I ensemble them together","metadata":{}},{"cell_type":"code","source":"VERBOSE = 0\nCYCLES = 1\nEPOCHS = 400\nBATCH_SIZE = 4096\nFOLDS = 10\nSEED = 0\nCURRENT_FOLD =9\n\ndef fit_model(seed, fold):\n    train = pd.read_feather('../input/amex-dart-agg/test_agg_extra2_scaled.ftr').drop('customer_ID', axis = 1)\n    target = np.clip(pd.read_csv(f'../input/amex-best-submission/best_submission.csv').prediction,0,1)\n    idx_tr0, idx_va0 = list(KFold(n_splits=FOLDS, shuffle= True, random_state= SEED).split(target,target))[fold]\n    X_va0 = train.iloc[idx_va0]\n    X_tr = train.iloc[idx_tr0]\n    y_tr, y_va0 = target[idx_tr0], target[idx_va0]\n    del train, target\n    gc.collect()\n    lr = ReduceLROnPlateau(monitor=\"val_loss\", factor=0.6, \n                           patience=7, verbose=VERBOSE)\n    es = EarlyStopping(monitor=\"val_loss\",\n                       patience=30, \n                       verbose=1,\n                       mode=\"min\", \n                       restore_best_weights=True)\n    callbacks = [lr, es, tf.keras.callbacks.TerminateOnNaN()]\n    del lr,es\n    gc.collect()\n    model = my_model(X_tr.shape[1])\n    model.compile(optimizer=AdaBeliefOptimizer(learning_rate=0.001,\n                                               weight_decay = 1e-5,\n        ),\n        loss='binary_crossentropy',\n        )\n    gc.collect()\n    model.fit(X_tr, y_tr, \n            validation_data=(X_va0, y_va0),\n            epochs=EPOCHS,\n            verbose=VERBOSE,\n            batch_size=BATCH_SIZE,\n            shuffle=True,\n            callbacks=callbacks)\n    del X_tr, y_tr\n    gc.collect()\n    \n    def split(a, n):\n        k, m = divmod(len(a), n)\n        return (a[i*k+min(i, m):(i+1)*k+min(i+1, m)] for i in range(n))\n    split_ids = split([str(i) for i in X_va0.columns],2)\n    train = []\n    for ids in split_ids:\n        train.append(pd.read_feather('../input/amex-dart-agg/train_agg_extra2_scaled.ftr',columns = ids))\n    train = pd.concat(train,axis = 1)\n    del split_ids\n    gc.collect()\n    target = pd.read_csv('../input/amex-oof-predictions/lgbm-dart_extra2_oof_0.7996.csv').prediction      \n    idx_tr1, idx_va1 = list(KFold(n_splits=FOLDS, shuffle= True, random_state= SEED).split(target,target))[fold]\n    X_va1 = train.iloc[idx_va1]\n    X_tr = train.iloc[idx_tr1]       \n    y_tr, y_va1 = target[idx_tr1], target[idx_va1]\n    del train, target\n    model.fit(X_tr, y_tr, \n            validation_data=(X_va1, y_va1),\n            epochs=EPOCHS,\n            verbose=VERBOSE,\n            batch_size=BATCH_SIZE,\n            shuffle=True,\n            callbacks=callbacks)\n    del X_tr, y_tr, callbacks\n    gc.collect()\n    test_pred = np.zeros(924621)\n    oof_pred = np.zeros(458913)\n    oof_pred[idx_va1] = model.predict(X_va1).reshape( (len(X_va1), )) \n    test_pred[idx_va0] = model.predict(X_va0).reshape( (len(X_va0), )) \n    model.save(f\"model_fold{fold}_seed{seed}.h5\")\n    return oof_pred, test_pred\n\n\n\n\nnp.random.seed(0)\nrandom.seed(0)\ndef fit_train_models(current_fold = 0):\n    tf.random.set_seed(0)\n    print(f'KFOLD WITH SEED {SEED}:')\n    oof_pred, test_pred = fit_model(SEED, current_fold)\n    predictions= pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\n    predictions['prediction'] = test_pred\n    predictions.to_csv(f'keras-knowledgedistilled_sub_fold{current_fold}_seed{SEED}.csv')\n    predictions= pd.read_csv('../input/amex-default-prediction/train_labels.csv').drop('target',axis=1)\n    predictions['prediction'] = oof_pred\n    predictions.to_csv(f'keras-knowledgedistilled_oof_fold{current_fold}_seed{SEED}.csv')\n    gc.collect()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-26T20:16:33.212307Z","iopub.execute_input":"2022-07-26T20:16:33.212626Z","iopub.status.idle":"2022-07-26T20:16:33.238339Z","shell.execute_reply.started":"2022-07-26T20:16:33.212599Z","shell.execute_reply":"2022-07-26T20:16:33.237678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"fit_train_models(current_fold = CURRENT_FOLD)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:16:33.239379Z","iopub.execute_input":"2022-07-26T20:16:33.240119Z","iopub.status.idle":"2022-07-26T21:29:34.286024Z","shell.execute_reply.started":"2022-07-26T20:16:33.240085Z","shell.execute_reply":"2022-07-26T21:29:34.284905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #clear memory\n# del train, target\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:29:34.288130Z","iopub.execute_input":"2022-07-26T21:29:34.288515Z","iopub.status.idle":"2022-07-26T21:29:34.293489Z","shell.execute_reply.started":"2022-07-26T21:29:34.288484Z","shell.execute_reply":"2022-07-26T21:29:34.292500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference\n\nBecause my dataset is too big, I need to run the inference after training the model on a separate run. \n\nWe also need to split the test data into multiple parts because of memory.\n\nThe following code is the result of running the above training 5 times with seeds 0-5","metadata":{}},{"cell_type":"code","source":"# test= pd.read_feather('../input/amex-dart-agg/test_agg_extra2_scaled.ftr')\n# test.drop('customer_ID',axis=1,inplace = True)\n# y_pred_avg = np.zeros(len(test))\n# y_preds = list()\n\n# seeds = [0,1,2,3,4,5]\n\n# for seed, fold in tqdm([(i,j) for i,j in itertools.product(seeds, range(FOLDS))]):\n#     json_file = open(f\"../input/amex-keras-models/model_fold{fold}_seed{seed}.json\", 'r')\n#     model = json_file.read()\n#     json_file.close()\n#     del json_file\n#     model = model_from_json(model)\n#     model.load_weights(f\"../input/amex-keras-models/model_fold{fold}_seed{seed}.h5\")\n#     gc.collect()\n#     pred = model.predict(test)[:, 1]\n#     y_pred_avg += pred / FOLDS / len(seeds)\n#     y_preds.append(pred)\n#     del model\n#     gc.collect()\n\n# del test\n# gc.collect()\n\n# with open('y_preds.pkl', 'wb') as f:\n#     pickle.dump(y_preds, f)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:29:34.295017Z","iopub.execute_input":"2022-07-26T21:29:34.296001Z","iopub.status.idle":"2022-07-26T21:29:34.307738Z","shell.execute_reply.started":"2022-07-26T21:29:34.295941Z","shell.execute_reply":"2022-07-26T21:29:34.306956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_preds = pickle.load(open('../input/temporary/y_preds.pkl', 'rb'))\n\n# desired = pd.read_csv('../input/amex-submission-files/lgbm-dart_extra2_test_0.7996.csv').prediction\n\n# y_pred_max = list()\n# for i in np.array(y_preds).T:\n#     y_pred_max.append(np.max(i))\n# y_pred_max = np.array(y_pred_max)\n\n# y_pred_median = list()\n# for i in np.array(y_preds).T:\n#     y_pred_median.append(np.median(i))\n# y_pred_median = np.array(y_pred_median)\n\n# y_pred_avg = list()\n# for i in np.array(y_preds).T:\n#     y_pred_avg.append(np.mean(i))\n# y_pred_avg = np.array(y_pred_avg)\n\n# indices = np.argsort(y_pred_avg)\n# needed_distrib = np.sort(desired)\n# y_pred_matched = np.zeros(len(y_pred_avg))\n# for i in range(len(indices)):\n#     y_pred_matched[indices[i]] = needed_distrib[i]","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:29:34.308972Z","iopub.execute_input":"2022-07-26T21:29:34.309515Z","iopub.status.idle":"2022-07-26T21:29:34.322923Z","shell.execute_reply.started":"2022-07-26T21:29:34.309484Z","shell.execute_reply":"2022-07-26T21:29:34.321910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_pred_max.mean()\n# y_pred_matched.mean()\n# y_pred_avg.mean()\n# y_pred_median.mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:29:34.325923Z","iopub.execute_input":"2022-07-26T21:29:34.326299Z","iopub.status.idle":"2022-07-26T21:29:34.335420Z","shell.execute_reply.started":"2022-07-26T21:29:34.326239Z","shell.execute_reply":"2022-07-26T21:29:34.333946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\n# sub['prediction'] = y_pred_avg\n# sub.to_csv(f'keras-knowledge_sub_0.805_avg.csv', index=False)\n\n# sub = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\n# sub['prediction'] = y_pred_median\n# sub.to_csv(f'keras-knowledge_sub_0.805_median.csv', index=False)\n\n# sub = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\n# sub['prediction'] = y_pred_matched\n# sub.to_csv(f'keras-knowledge_sub_0.805_matched.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:29:34.337502Z","iopub.execute_input":"2022-07-26T21:29:34.337848Z","iopub.status.idle":"2022-07-26T21:29:34.349103Z","shell.execute_reply.started":"2022-07-26T21:29:34.337819Z","shell.execute_reply":"2022-07-26T21:29:34.348331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plt.figure(figsize=(30, 16))\n# plt.hist(y_pred_median, bins=np.linspace(0, 1, 50), alpha = 0.33, density=True, label = 'median')\n# plt.hist(y_pred_avg, bins=np.linspace(0, 1, 50), alpha = 0.33, density=True, label = 'mean')\n# plt.hist(desired, bins=np.linspace(0, 1, 50),alpha = 0.33, density=True, label = 'desired')\n# plt.title(\"Plausibility check\", fontsize=20)\n# plt.legend()\n# plt.xlabel('Prediction')\n# plt.ylabel('Density')\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:29:34.351233Z","iopub.execute_input":"2022-07-26T21:29:34.352945Z","iopub.status.idle":"2022-07-26T21:29:34.363395Z","shell.execute_reply.started":"2022-07-26T21:29:34.352909Z","shell.execute_reply":"2022-07-26T21:29:34.362739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# folds = 5\n# seeds = [0,1,2,3]\n# oof_predictions = np.zeros(458913)\n# test_predictions = np.zeros(924621)\n# target = pd.read_csv('../input/amex-default-prediction/train_labels.csv').target\n# for seed in seeds:\n#     oof_pred = np.zeros(458913)\n#     for fold in range(folds):\n# #         idx = list(KFold(n_splits=FOLDS, shuffle= True, random_state= SEED).split(target,target))[fold][1]\n#         pred = pd.read_csv(f'../input/amex-keras-models/keras-knowledgedistilled_oof_fold{fold}_seed{seed}.csv',usecols = ['prediction']).prediction\n#         oof_pred +=pred\n#         pred = pd.read_csv(f'../input/amex-keras-models/keras-knowledgedistilled_sub_fold{fold}_seed{seed}.csv',usecols = ['prediction']).prediction\n#         test_predictions +=pred / len(seeds)\n#     print(amex_metric(target,oof_pred))\n#     oof_predictions+=oof_pred/len(seeds)\n    \n# print(amex_metric(target,oof_predictions))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:29:34.364387Z","iopub.execute_input":"2022-07-26T21:29:34.365124Z","iopub.status.idle":"2022-07-26T21:29:34.374945Z","shell.execute_reply.started":"2022-07-26T21:29:34.365073Z","shell.execute_reply":"2022-07-26T21:29:34.374043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# oof_df = pd.read_csv('../input/amex-default-prediction/train_labels.csv').drop('target',axis = 1)\n# oof_df['prediction'] = oof_predictions \n# oof_df.to_csv('keras-knowledge_oof_0.8026.csv',index = False)\n# oof_df = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\n# oof_df['prediction'] = test_predictions\n# oof_df.to_csv('keras-knowledge_sub_0.8026.csv',index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:29:34.376118Z","iopub.execute_input":"2022-07-26T21:29:34.376663Z","iopub.status.idle":"2022-07-26T21:29:34.385422Z","shell.execute_reply.started":"2022-07-26T21:29:34.376628Z","shell.execute_reply":"2022-07-26T21:29:34.384472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# oof_df","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:29:34.386750Z","iopub.execute_input":"2022-07-26T21:29:34.387241Z","iopub.status.idle":"2022-07-26T21:29:34.399588Z","shell.execute_reply.started":"2022-07-26T21:29:34.387212Z","shell.execute_reply":"2022-07-26T21:29:34.398742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission\n\nthe result of running the above code cells multiple times with different seeds and averaging them together\n\nthis model has CV 0.805 and LB 0.791\n\nbest LB score on this run is the matched version","metadata":{}},{"cell_type":"code","source":"# sub = pd.read_csv('keras-knowledge_sub_0.805_matched.csv')\n# sub.to_csv('keras-knowledge_sub_0.805.csv',index = False)\n# sub","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:29:34.400828Z","iopub.execute_input":"2022-07-26T21:29:34.401437Z","iopub.status.idle":"2022-07-26T21:29:34.410206Z","shell.execute_reply.started":"2022-07-26T21:29:34.401405Z","shell.execute_reply":"2022-07-26T21:29:34.409544Z"},"trusted":true},"execution_count":null,"outputs":[]}]}