{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"! pip install -q imagesize\n! pip install -q tensorflow_addons\n! pip install -q imagesize\n! pip install -q https://github.com/pandas-profiling/pandas-profiling/archive/master.zip\n    \nprint('Modules Installed!')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-07T20:24:35.243068Z","iopub.execute_input":"2022-08-07T20:24:35.243380Z","iopub.status.idle":"2022-08-07T20:25:18.985628Z","shell.execute_reply.started":"2022-08-07T20:24:35.243304Z","shell.execute_reply":"2022-08-07T20:25:18.984617Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import imagesize, os, random\nimport seaborn as sns\nimport math\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport sklearn\nimport tensorflow as tf\nimport tensorflow.keras.backend as K\nimport tensorflow_addons as tfa\nimport tensorflow.keras.applications.efficientnet as efn\nfrom IPython.core.display import display\nfrom sklearn.model_selection import StratifiedKFold\nfrom tqdm.notebook import tqdm\nfrom pandas_profiling import ProfileReport\nfrom kaggle_datasets import KaggleDatasets\n\nprint('tf:', tf.__version__)\nprint('np:', np.__version__)\nprint('pd:', pd.__version__)\nprint('sklearn:', sklearn.__version__)\nprint('tfa:', tfa.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:25:18.987550Z","iopub.execute_input":"2022-08-07T20:25:18.987815Z","iopub.status.idle":"2022-08-07T20:25:26.238332Z","shell.execute_reply.started":"2022-08-07T20:25:18.987782Z","shell.execute_reply":"2022-08-07T20:25:26.237600Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    TRAIN = True\n    \n    # Variable v = verbose: \n    # v=0 for silent output, \n    # v=1 for interactive \n    # v=2 for commit\n    v = 1\n    \n    device = \"TPU\"  # GPU or TPU, Set to TPU when Training, or GPU when predicting\n\n    # Sets program's 'Random' Seed. Used for replicating randomized data sets\n    SEED = 123456789\n\n    # Sets the number of K folds for verification\n    K_FOLDS = 5\n\n    # Sets which specific fold numbers to train. Normally a list of all K_FOLDS.\n    FOLDS_TO_TRAIN = [0, 1, 2, 3, 4]\n\n    # Sets the image sizes to feed to the algorithm\n    IMG_SIZE = [224, 224] #224x224 is optimal for the EfficientNetB0 Model currently selected\n\n    # Sets the batch size and number of epochs\n    BATCH_SIZE = 32\n    EPOCHS = 16\n\n    # Sets loss and optimizer algorithms\n    LOSS_OPTIMIZER = 'Adam'\n\n    # Sets if we want to flip horizontally or vertically\n    HORIZ_FLIP = True\n    VERT_FLIP = False\n\n    # Sets the method to be used in the learning scheduler\n    scheduler = 'exp'  # Cosine\n\n    # Sets the list of columns in the csv file\n    tab_cols = ['Subject Focus', 'Eyes', 'Face', 'Near', 'Action', 'Accessory',\n                'Group', 'Collage', 'Human', 'Occlusion', 'Info', 'Blur']\n    \n    #  Sets the target column for the algorithm to predict\n    target_col = ['Pawpularity']","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:25:26.239480Z","iopub.execute_input":"2022-08-07T20:25:26.239710Z","iopub.status.idle":"2022-08-07T20:25:26.246431Z","shell.execute_reply.started":"2022-08-07T20:25:26.239684Z","shell.execute_reply":"2022-08-07T20:25:26.245478Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seeding(SEED):\n    np.random.seed(SEED)\n    random.seed(SEED)\n    os.environ['PYTHONHASHSEED'] = str(SEED)\n    os.environ['TF_CUDNN_DETERMINISTIC'] = str(SEED)\n    tf.random.set_seed(SEED)\n    print('Random Values Seeded')\n\nseeding(Config.SEED)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:25:26.248520Z","iopub.execute_input":"2022-08-07T20:25:26.249357Z","iopub.status.idle":"2022-08-07T20:25:26.263255Z","shell.execute_reply.started":"2022-08-07T20:25:26.249318Z","shell.execute_reply":"2022-08-07T20:25:26.262009Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if Config.device == \"TPU\":\n    print(\"Attempting to Connect to TPU\")\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        print('Connected to TPU ', tpu.master())\n    except ValueError:\n        print(\"Unable to connect to TPU\")\n        tpu = None\n\n    if tpu:\n        try:\n            print(\"Initializing TPU\")\n            tf.config.experimental_connect_to_cluster(tpu)\n            tf.tpu.experimental.initialize_tpu_system(tpu)\n            strategy = tf.distribute.experimental.TPUStrategy(tpu)\n            print(\"TPU Initialized Successfully..(hopefully)\")\n        except _:\n            print(\"TPU Initialization Failed :()\")\n    else:\n        Config.device = \"GPU\"\n\nif Config.device != \"TPU\":\n    print(\"Using Default TF Strategy. (CPU/GPU)\")\n    strategy = tf.distribute.get_strategy()\n\nif Config.device == \"GPU\":\n    print(\"GPUs Currently Available: \", len(tf.config.experimental.list_physical_devices('GPU')))\n\nAUTO = tf.data.experimental.AUTOTUNE\nnum_replicas = strategy.num_replicas_in_sync\nprint(f'Replica Sets: {num_replicas}')\n\nROOT = '/kaggle/input/petfinder-pawpularity-score'  # Root working directory\nG_ROOT = KaggleDatasets().get_gcs_path('petfinder-pawpularity-score')  # Root directory on Google Cloud Service. (Required for TPU to function)\nprint(G_ROOT)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:25:26.264357Z","iopub.execute_input":"2022-08-07T20:25:26.264800Z","iopub.status.idle":"2022-08-07T20:25:32.971483Z","shell.execute_reply.started":"2022-08-07T20:25:26.264775Z","shell.execute_reply":"2022-08-07T20:25:32.970315Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_imgsize(row):\n    width, height = imagesize.get(row['image_path'].replace(G_ROOT, ROOT))\n    row['width'] = width\n    row['height'] = height\n    return row\n\n# Train Data\ndf = pd.read_csv(ROOT + '/train.csv')\ndf['image_path'] = G_ROOT + '/train/' + df.Id + '.jpg'\ntqdm.pandas(desc='train')\ndf = df.progress_apply(get_imgsize, axis=1)\ndisplay(df.head(2))\n\n# Test Data\ntest_df = pd.read_csv(ROOT + '/test.csv')\ntest_df['image_path'] = G_ROOT + '/test/' + test_df.Id + '.jpg'\ntqdm.pandas(desc='test')\ntest_df = test_df.progress_apply(get_imgsize, axis=1)\n\ndisplay(test_df.head(2))\n\nprint('Training Files:', df.shape[0])\nprint('Test Files:', test_df.shape[0])\n\nnum_bins = int(np.floor(1 + np.log2(len(df))))\ndf[\"bins\"] = pd.cut(df[Config.target_col].values.reshape(-1), bins=num_bins, labels=False)\n\nskf = StratifiedKFold(n_splits=Config.K_FOLDS, shuffle=True, random_state=Config.SEED)\n\nfor fold, (train_idx, val_idx) in enumerate(skf.split(df, df[\"bins\"])):\n    df.loc[val_idx, 'fold'] = fold\n    \ndisplay(df.groupby(['fold', \"bins\"]).size())","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:25:32.972817Z","iopub.execute_input":"2022-08-07T20:25:32.973280Z","iopub.status.idle":"2022-08-07T20:26:50.293340Z","shell.execute_reply.started":"2022-08-07T20:25:32.973249Z","shell.execute_reply":"2022-08-07T20:26:50.292546Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_profile = ProfileReport(df, title=\"Train Data\")\ntest_profile = ProfileReport(test_df, title=\"Test Data\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:26:50.294766Z","iopub.execute_input":"2022-08-07T20:26:50.295302Z","iopub.status.idle":"2022-08-07T20:26:50.304042Z","shell.execute_reply.started":"2022-08-07T20:26:50.295263Z","shell.execute_reply":"2022-08-07T20:26:50.303477Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decoder_gen(with_labels=True, target_size=Config.IMG_SIZE, ext='jpg'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path)\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3)\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels=3)\n        else:\n            raise ValueError(\"Image extension not supported\")\n\n        img = tf.image.resize(img, target_size)\n        img = tf.cast(img, tf.float32) / 255.0\n        img = tf.reshape(img, [*target_size, 3])\n\n        return img\n\n    def decode_with_labels(path, label):\n        return decode(path), tf.cast(label, tf.float32)\n\n    return decode_with_labels if with_labels else decode\n\n\ndef datagen(paths, labels=None, batch_size=32, cache=True,\n                  decode_fn=None, repeat=True, shuffle=1024,\n                  cache_dir=\"\", drop_remainder=False):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n\n    if decode_fn is None:\n        decode_fn = decoder_gen(labels is not None)\n\n\n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n\n    ds = tf.data.Dataset.from_tensor_slices(slices)\n    ds = ds.map(decode_fn, num_parallel_calls=AUTO)\n    ds = ds.cache(cache_dir) if cache else ds\n    ds = ds.repeat() if repeat else ds\n    if shuffle:\n        ds = ds.shuffle(shuffle, seed=Config.SEED)\n        opt = tf.data.Options()\n        opt.experimental_deterministic = False\n        ds = ds.with_options(opt)\n    ds = ds.batch(batch_size, drop_remainder=drop_remainder)\n    ds = ds.prefetch(AUTO)\n    return ds\n\n# Method used to display a batch of pictures.\ndef display_batch(batch, size=2):\n    imgs, tars = batch\n    plt.figure(figsize=(size * 5, 5))\n    for img_idx in range(size):\n        plt.subplot(1, size, img_idx + 1)\n        plt.title(f'{Config.target_col[0]}: {tars[img_idx].numpy()[0]}', fontsize=15)\n        plt.imshow(imgs[img_idx, :, :, :])\n        plt.xticks([])\n        plt.yticks([])\n    plt.tight_layout()\n    plt.show()\n\n\ndef RMSE(y_true, y_pred):\n    loss = tf.math.sqrt(tf.math.reduce_mean(tf.math.square(tf.subtract(y_true, y_pred))))\n    return loss\n\n# Model Creation Method. Called to create a new model.\ndef create_model(DIM=Config.IMG_SIZE[0], compile_model=True, include_top=False):\n    \n    base = efn.EfficientNetB0(input_shape=(DIM, DIM, 3),include_top=include_top,weights='imagenet',)\n    inp = base.inputs\n    out = base.output\n    out = tf.keras.layers.GlobalAveragePooling2D()(out)\n    out = tf.keras.layers.Dense(64, activation='selu')(out)\n    out = tf.keras.layers.Dense(1)(out)\n    model = tf.keras.Model(inputs=inp, outputs=out)\n    if compile_model:\n        # Selected Optimizer\n        opt = tf.keras.optimizers.Adam(learning_rate=0.001)\n        # Selected Loss Algorithm\n        loss = RMSE\n        # Selected Metric\n        rmse = tf.keras.metrics.RootMeanSquaredError(name='rmse')\n        model.compile(optimizer=opt, loss=loss, metrics=[rmse])\n        \n    return model\n\n\n# Dynamically schedules the learning rate to prioritize the middle EPOCHs where training is more valuable\ndef get_lr_callback(batch_size=8):\n    lr_start = 0.000005\n    lr_max = 0.00000125 * num_replicas * batch_size\n    lr_min = 0.000001\n    lr_ramp_ep = 5\n    lr_sus_ep = 0\n    lr_decay = 0.8\n\n    def lrfn(epoch):\n        if epoch < lr_ramp_ep:\n            lr = (lr_max - lr_start) / lr_ramp_ep * epoch + lr_start\n\n        elif epoch < lr_ramp_ep + lr_sus_ep:\n            lr = lr_max\n\n        elif Config.scheduler == 'exp':\n            lr = (lr_max - lr_min) * lr_decay ** (epoch - lr_ramp_ep - lr_sus_ep) + lr_min\n\n        elif Config.scheduler == 'cosine':\n            decay_total_epochs = Config.EPOCHS - lr_ramp_ep - lr_sus_ep + 3\n            decay_epoch_index = epoch - lr_ramp_ep - lr_sus_ep\n            phase = math.pi * decay_epoch_index / decay_total_epochs\n            cosine_decay = 0.5 * (1 + math.cos(phase))\n            lr = (lr_max - lr_min) * cosine_decay + lr_min\n        return lr\n\n    lr_callback = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=False)\n    return lr_callback\n\n\n_ = get_lr_callback(Config.BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:26:50.305291Z","iopub.execute_input":"2022-08-07T20:26:50.305516Z","iopub.status.idle":"2022-08-07T20:26:50.649804Z","shell.execute_reply.started":"2022-08-07T20:26:50.305492Z","shell.execute_reply":"2022-08-07T20:26:50.648935Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fold = 0 # Sets the first fold to fold 0\nfold_df = df.query('fold==@fold')[:1000] \npaths = fold_df.image_path.tolist()  # Sets paths to each path in the csv dataset\nlabels = fold_df[Config.target_col].values  # Sets labels to all pawpularity values\n\n# Pick a batch of animals at random to display. This ensures that we are loading and rebuilding the images correctly\nds = datagen(paths, labels, cache=True, batch_size=Config.BATCH_SIZE * num_replicas,\n                   repeat=True, shuffle=True)\nds = ds.unbatch().batch(20)\nbatch = next(iter(ds))\ndisplay_batch(batch, 5); # Calls display_batch to display the batch of images","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:26:50.651479Z","iopub.execute_input":"2022-08-07T20:26:50.651792Z","iopub.status.idle":"2022-08-07T20:27:04.111938Z","shell.execute_reply.started":"2022-08-07T20:26:50.651755Z","shell.execute_reply":"2022-08-07T20:27:04.111288Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Initialize Prediction/Validation Arrays\nout = [];\nout_tar = [];\nout_val = [];\nout_ids = [];\nout_folds = [];\nout_pred = [];\npreds = np.zeros((test_df.shape[0], 1))\n\n# Repeats for each fold in the number of K_Folds set in the Config class\nfor fold in np.arange(Config.K_FOLDS):\n    if fold not in Config.FOLDS_TO_TRAIN:\n        continue\n    if Config.device == 'TPU':\n        if tpu: tf.tpu.experimental.initialize_tpu_system(tpu)\n\n    # Create train and validation dataframes\n    t_df = df.query(\"fold!=@fold\")\n    v_df = df.query(\"fold==@fold\")\n\n    # Create train and validation subsets\n    t_paths = t_df.image_path.values;\n    t_labels = t_df[Config.target_col].values.astype(np.float32)\n    valid_paths = v_df.image_path.values;\n    valid_labels = v_df[Config.target_col].values.astype(np.float32)\n    test_paths = test_df.image_path.values\n\n    # Here we shuffle around the images\n    index = np.arange(len(t_paths))\n    np.random.shuffle(index)\n    t_paths = t_paths[index]\n    t_labels = t_labels[index]\n\n    print('*' * 100);\n    print('** Fold #: ', fold)\n    print('** Image Size: (%i, %i) | Batch Size: %i' %\n          (Config.IMG_SIZE[0], Config.IMG_SIZE[1], Config.BATCH_SIZE * num_replicas))\n    train_images = len(t_paths)\n    val_images = len(valid_paths)\n\n    print('*** Training Images: %i | Validation Images: %i' % (train_images, val_images))\n\n    # Begin Creation of Model\n    K.clear_session()\n    with strategy.scope():\n        model = create_model(DIM=Config.IMG_SIZE[0], compile_model=True)\n\n    # Generate the datasets\n    train_ds = datagen(t_paths, t_labels, cache=True, batch_size=Config.BATCH_SIZE * num_replicas,\n                             repeat=True, shuffle=True)\n    val_ds = datagen(valid_paths, valid_labels, cache=True, batch_size=Config.BATCH_SIZE * num_replicas,\n                           repeat=False, shuffle=False)\n\n    print('*' * 100)\n    # Save the model if it is the best for the current fold. Fold number in the name\n    sv = tf.keras.callbacks.ModelCheckpoint('fold-%i.h5' % fold, monitor='val_rmse', verbose=0, save_best_only=True,\n        save_weights_only=False, mode='min', save_freq='epoch')\n    callbacks = [sv, get_lr_callback(Config.BATCH_SIZE)]\n\n    # Start Training the Model\n    print('Starting Training with model.fit() ...')\n    history = model.fit(train_ds,epochs=Config.EPOCHS,callbacks=callbacks,steps_per_epoch=len(t_paths) / Config.BATCH_SIZE // num_replicas,\n        validation_data=val_ds,verbose=0)\n\n    # Load previously saved model weights\n    model.load_weights('fold-%i.h5' % fold)\n\n    # Complete out-of-fold predictions\n    print('Generating predictions...')\n    ds_validation = datagen(valid_paths, labels=None, cache=False, batch_size=Config.BATCH_SIZE * num_replicas * 2,\n                             repeat=True, shuffle=False)\n\n    ct_valid = len(valid_paths);\n    STEPS = 1 * ct_valid / Config.BATCH_SIZE / 2 / num_replicas\n    pred = model.predict(ds_validation, steps=STEPS, verbose=0)[:1 * ct_valid, ]\n    out_pred.append(np.mean(pred.reshape((ct_valid, -1, 1), order='F'), axis=-1))\n\n    # Append out of fold targets and IDs\n    out_tar.append(v_df[Config.target_col].values[:len(v_df)])\n    out_folds.append(np.ones_like(out_tar[-1], dtype='int8') * fold)\n    out_ids.append(v_df.Id.values)\n\n    # Complete prediction using TTA\n    print('Predicting Test with TTA...')\n    ds_test = datagen(test_paths, labels=None, cache=False,\n                            batch_size=(Config.BATCH_SIZE * 2 if len(test_df) > 8 else 1) * num_replicas,\n                            repeat=True, shuffle=False)\n\n    test_count = len(test_paths);\n    STEPS = 1 if len(test_df) <= 8 else (1 * test_count / Config.BATCH_SIZE / 2 / num_replicas)\n    pred = model.predict(ds_test, steps=STEPS, verbose=0)[:1 * test_count, ]\n    preds[:test_count, :] += np.mean(pred.reshape((test_count, -1, 1), order='F'), axis=-1) / Config.K_FOLDS\n\n    # Print out loss results\n    y_true = out_tar[-1];\n    y_pred = out_pred[-1]\n    rmse = RMSE(y_true.astype(np.float32), y_pred).numpy()\n    out_val.append(np.min(history.history['val_rmse']))\n    print('*** Fold #: %i Out of Fold Root Mean Sqare Error Test 1 = %.3f, Test 2 = %.3f' % (fold, out_val[-1], rmse))\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:27:04.113845Z","iopub.execute_input":"2022-08-07T20:27:04.114248Z","iopub.status.idle":"2022-08-07T21:02:32.815192Z","shell.execute_reply.started":"2022-08-07T20:27:04.114217Z","shell.execute_reply":"2022-08-07T21:02:32.814194Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate and display the Overall RMSE\nout = np.concatenate(out_pred);\ntrue = np.concatenate(out_tar);\nids = np.concatenate(out_ids);\nfolds = np.concatenate(out_folds)\nrmse = RMSE(true.astype(np.float32), out)\nprint('Out of fold root mean square deviation = %.3f' % rmse)\n\n# Save the Out of fold data to csv file\ncolumns = ['Id', 'fold', 'true', 'pred']\ndf_out = pd.DataFrame(np.concatenate([ids[:, None], folds[:, 0:1], true, out], axis=1), columns=columns)\ndf_out.to_csv('out_of_fold.csv', index=False)\ndf_out.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:02:32.817032Z","iopub.execute_input":"2022-08-07T21:02:32.817724Z","iopub.status.idle":"2022-08-07T21:02:32.875430Z","shell.execute_reply.started":"2022-08-07T21:02:32.817678Z","shell.execute_reply":"2022-08-07T21:02:32.874534Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NOT YET IMPLEMENTED :((\n# save model locally from tpu using Tensorflow's \"SavedModel\" format\n# save_locally = tf.saved_model.SaveOptions(experimental_io_device='/job:localhost')\n# model.save('./model', options=save_locally)\n# ! zip -r saved_model.zip ./model","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:02:32.876854Z","iopub.execute_input":"2022-08-07T21:02:32.877358Z","iopub.status.idle":"2022-08-07T21:02:32.881251Z","shell.execute_reply.started":"2022-08-07T21:02:32.877317Z","shell.execute_reply":"2022-08-07T21:02:32.880676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(style='dark')\n\n# Plot first chart\nplt.figure(figsize=(20, 10))\nplt.subplot(1, 2, 1)\nsns.kdeplot(x=t_df[Config.target_col[0]], color='g', shade=True);\nsns.kdeplot(x=df_out.pred.values, color='r', shade=True);\nplt.grid('ON')\nplt.xlabel(Config.target_col[0]);\nplt.ylabel('Density');\nplt.title('Kernal Density Estimation')\nplt.legend(['train', 'test'])\n\n# Second chart\nplt.subplot(1, 2, 2)\nsns.histplot(x=t_df[Config.target_col[0]], color='g');\nsns.histplot(x=df_out.pred.values, color='r');\nplt.grid('ON')\nplt.xlabel(Config.target_col[0]);\nplt.ylabel('Density');\nplt.title('Histogram')\nplt.legend(['train', 'test'])\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:02:32.882323Z","iopub.execute_input":"2022-08-07T21:02:32.882928Z","iopub.status.idle":"2022-08-07T21:02:34.015947Z","shell.execute_reply.started":"2022-08-07T21:02:32.882900Z","shell.execute_reply":"2022-08-07T21:02:34.015022Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.DataFrame({'Id': test_df.Id,\n                        'Pawpularity': preds.reshape(-1)})\nsub_df = pd.read_csv(ROOT + '/sample_submission.csv')\ndel sub_df['Pawpularity']\nsub_df = sub_df.merge(pred_df, on='Id', how='left')\nsub_df.to_csv('test.csv', index=False)\nsub_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:02:34.017184Z","iopub.execute_input":"2022-08-07T21:02:34.017467Z","iopub.status.idle":"2022-08-07T21:02:34.054375Z","shell.execute_reply.started":"2022-08-07T21:02:34.017408Z","shell.execute_reply":"2022-08-07T21:02:34.053825Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_profile)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:02:34.055509Z","iopub.execute_input":"2022-08-07T21:02:34.055849Z","iopub.status.idle":"2022-08-07T21:03:20.892105Z","shell.execute_reply.started":"2022-08-07T21:02:34.055823Z","shell.execute_reply":"2022-08-07T21:03:20.891092Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(test_profile)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:03:20.893474Z","iopub.execute_input":"2022-08-07T21:03:20.893690Z","iopub.status.idle":"2022-08-07T21:03:35.086209Z","shell.execute_reply.started":"2022-08-07T21:03:20.893664Z","shell.execute_reply":"2022-08-07T21:03:35.085307Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]}]}