{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install -q pydot graphviz","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nimport re\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\n\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Hyperparameter"},{"metadata":{"trusted":true},"cell_type":"code","source":"EPOCHS = 10\n\nTRAINING_STEPS = 1000\nVALIDATION_STEPS = 100\nEVALUATION_STEPS = 100\n\nSHUFFLE_BUFFER = 200\nBATCH_SIZE = 64\n\nIMAGE_DIM = 256\n\nATOMS_TO_COUNT = [\n    \"C\",\n    \"H\",\n    \"O\",\n    \"S\",\n    \"N\",\n    \"Br\",\n    \"F\",\n    \"Cl\",\n    \"P\",\n    \"Si\",\n    \"B\",\n    \"I\"\n]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"INPUT_PATH = \"../input/bms-molecular-translation\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"raw_train_df = pd.read_csv(os.path.join(INPUT_PATH, \"train_labels.csv\"))\nraw_train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"raw_train_df['ImagePath'] = raw_train_df.image_id.map(lambda x: os.path.join(x[0], x[1], x[2], x + \".png\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"raw_train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"raw_train_df.InChI.map(lambda x: x.split(\"=\")[1].split(\"/\")).map(len).value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"raw_train_df[raw_train_df.InChI.map(lambda x: x.split(\"=\")[1].split(\"/\")).map(len)==11]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"raw_train_df.InChI[774948].split(\"=\")[1].split(\"/\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"raw_train_df.InChI.map(lambda x: x.split(\"=\")[1].split(\"/\")[0]).value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"raw_train_df.InChI.map(lambda x: x.split(\"=\")[1].split(\"/\")[1]).value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"re.sub(r\"([A-Z][a-z]*)(?=[A-Z]|$)\",r\"\\g<1>1\",\"C15H18BrN5O2SSS\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_molecule_names = raw_train_df.InChI.map(lambda x: x.split(\"=\")[1].split(\"/\")[1])\ntrain_molecule_names = train_molecule_names.map(lambda x: re.sub(r\"([A-Z][a-z]*)(?=[A-Z]|$)\",r\"\\g<1>1\",x))\ntrain_molecule_names = train_molecule_names.map(lambda x: list(filter(None, re.split(r\"([A-Z]+[a-z\\d]+)\", x))))\ntrain_molecule_names = train_molecule_names.map(lambda x: map(lambda y: list(filter(None, re.split(r\"(\\D+)\", y))),x))\ntrain_molecule_names = train_molecule_names.map(dict)\ntrain_molecule_names = train_molecule_names.tolist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"names_df = pd.DataFrame(train_molecule_names)\nnames_df = names_df.fillna(0)\nnames_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"names_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"processed_train_df = raw_train_df.join(names_df.astype(int))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"processed_train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del raw_train_df\ndel names_df\ndel train_molecule_names","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Normalization"},{"metadata":{"trusted":true},"cell_type":"code","source":"label_max = processed_train_df.describe()[ATOMS_TO_COUNT].T['max'].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"processed_train_df[ATOMS_TO_COUNT] = processed_train_df[ATOMS_TO_COUNT]/label_max","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"processed_train_df.describe().T","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Train Test Split"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df, val_df = train_test_split(processed_train_df, random_state=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def parse_image(file_path, labels, data_dir):\n    img = tf.io.read_file(os.path.join(INPUT_PATH, data_dir.numpy().decode('utf-8'), file_path.numpy().decode('utf-8')))\n    img = tf.image.decode_jpeg(img, channels=1)\n    img = tf.image.convert_image_dtype(img, tf.float32)\n    img = tf.image.resize(img, (IMAGE_DIM, IMAGE_DIM))\n    return img, labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_dataset = tf.data.Dataset.from_tensor_slices((train_df.ImagePath, train_df[ATOMS_TO_COUNT]))\ntrain_dataset = train_dataset.map(\n    lambda file_path, labels: tf.py_function(parse_image, [file_path, labels, \"train\"], [tf.float32, tf.float64]),\n    num_parallel_calls=tf.data.experimental.AUTOTUNE\n)\ntrain_dataset = train_dataset.shuffle(SHUFFLE_BUFFER)\ntrain_dataset = train_dataset.batch(BATCH_SIZE)\ntrain_dataset = train_dataset.prefetch(tf.data.experimental.AUTOTUNE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"list(train_dataset.take(1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"val_dataset = tf.data.Dataset.from_tensor_slices((val_df.ImagePath, val_df[ATOMS_TO_COUNT]))\nval_dataset = val_dataset.map(\n    lambda file_path, labels: tf.py_function(parse_image, [file_path, labels, \"train\"], [tf.float32, tf.float64]),\n    num_parallel_calls=tf.data.experimental.AUTOTUNE\n)\nval_dataset = val_dataset.shuffle(SHUFFLE_BUFFER)\nval_dataset = val_dataset.batch(BATCH_SIZE)\nval_dataset = val_dataset.prefetch(tf.data.experimental.AUTOTUNE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images, labels = list(train_dataset.take(1))[0]\nplt.figure(figsize=(40,20))\nfor i, image in enumerate(images, 1):\n    plt.subplot(4,8,i)\n    plt.imshow(image)\n    plt.axis(\"off\")\n    plt.title(np.array(labels)[i-1]*label_max)\n    if i==32:\n        break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels.shape, images.shape","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Model"},{"metadata":{"trusted":true},"cell_type":"code","source":"feat_model = tf.keras.Sequential([\n    tf.keras.layers.Conv2D(32, 3, input_shape=(256, 256, 1)),\n    tf.keras.layers.MaxPool2D(2,2),\n    tf.keras.layers.Dropout(0.2),\n    tf.keras.layers.Conv2D(64, 3),\n    tf.keras.layers.MaxPool2D(2,2),\n    tf.keras.layers.Dropout(0.2),\n    tf.keras.layers.Conv2D(128, 3),\n    tf.keras.layers.MaxPool2D(2,2),\n    tf.keras.layers.Dropout(0.2),\n    tf.keras.layers.Conv2D(256, 3),\n    tf.keras.layers.MaxPool2D(2,2),\n    tf.keras.layers.Dropout(0.2),\n    tf.keras.layers.Conv2D(512, 3),\n    tf.keras.layers.MaxPool2D(2,2),\n    tf.keras.layers.Dropout(0.2),\n    tf.keras.layers.Conv2D(1024, 3),\n    tf.keras.layers.Flatten()\n], name=\"FeatureModel\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"feat_model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_regressor(symb):\n    return tf.keras.Sequential([\n        tf.keras.layers.Dense(1024, activation='relu'),\n        tf.keras.layers.Dropout(0.2),\n        tf.keras.layers.Dense(256, activation='relu'),\n        tf.keras.layers.Dropout(0.2),\n        tf.keras.layers.Dense(64, activation='relu'),\n        tf.keras.layers.Dense(1)\n    ], name=f\"{symb}_Regressor\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_input = tf.keras.Input(shape=(256,256,1), name=\"InputImage\")\n\nfeatures = feat_model(image_input)\n\noutputs = []\nfor symb in ATOMS_TO_COUNT:\n    output = get_regressor(symb)(features)\n    outputs.append(output)\n\noutputs = tf.concat(outputs, axis=1)\n\nmodel = tf.keras.Model(image_input, outputs)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tf.keras.utils.plot_model(model, show_shapes=True, expand_nested=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(\n    optimizer=\"adam\",\n    loss=\"mae\",\n    metrics=[\"mse\", \"mae\"]\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.rint(model.predict(images) * label_max)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(\n    'model.h5',\n    save_best_only=True\n)\n\nearly_stop_callback = tf.keras.callbacks.EarlyStopping(patience=3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history = model.fit(\n    train_dataset,\n    validation_data=val_dataset,\n    epochs=EPOCHS,\n    steps_per_epoch=TRAINING_STEPS,\n    validation_steps=VALIDATION_STEPS,\n    callbacks=[checkpoint_callback, early_stop_callback]\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = tf.keras.models.load_model(\"model.h5\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history.epoch, history.history['mse'], label=\"Train MSE\")\nplt.plot(history.epoch, history.history['val_mse'], '--', label=\"Validation MSE\")\nplt.title(\"Mean Squared Error\")\nplt.legend()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history.epoch, history.history['mae'], label=\"Train MAE\")\nplt.plot(history.epoch, history.history['val_mae'], '--', label=\"Validation MAE\")\nplt.title(\"Mean Absolute Error\")\nplt.legend()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.evaluate(train_dataset, steps=EVALUATION_STEPS)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.evaluate(val_dataset, steps=EVALUATION_STEPS)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for images, labels in val_dataset.take(1):\n    plt.figure(figsize=(40,20))\n    preds = model.predict(images)*label_max\n    preds[preds<0] = 0\n    preds = np.rint(preds)\n    labels = labels.numpy()*label_max\n    for i, image in enumerate(images, 1):\n        plt.subplot(4,8,i)\n        plt.imshow(image)\n        plt.axis(\"off\")\n        plt.title(np.array([labels[i-1], preds[i-1]]))\n        if i==32:\n            break","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Prediction"},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df = pd.read_csv(os.path.join(INPUT_PATH, \"sample_submission.csv\"))\ntest_df['ImagePath'] = test_df.image_id.map(lambda x: os.path.join(x[0], x[1], x[2], x + \".png\"))\ntest_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_dataset = tf.data.Dataset.from_tensor_slices(test_df.ImagePath)\ntest_dataset = test_dataset.map(\n    lambda file_path: tf.py_function(parse_image, [file_path, \"\", \"test\"], [tf.float32]),\n    num_parallel_calls=tf.data.experimental.AUTOTUNE\n)\ntest_dataset = test_dataset.batch(BATCH_SIZE)\ntest_dataset = test_dataset.prefetch(tf.data.experimental.AUTOTUNE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for images in test_dataset.take(1):\n    plt.figure(figsize=(40,20))\n    labels = model.predict(images)*label_max\n    labels[labels<0] = 0\n    labels = np.rint(labels)\n    for i, image in enumerate(images[0], 1):\n        plt.subplot(4,8,i)\n        plt.imshow(image)\n        plt.axis(\"off\")\n        plt.title(np.array(labels)[i-1])\n        if i==32:\n            break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = model.predict(test_dataset, verbose=1)*label_max","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = np.rint(preds)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds[preds<0] = 0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_labels = pd.DataFrame(preds, columns=ATOMS_TO_COUNT).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = pd.DataFrame(list(\n    map(\n        lambda y: \"InChI=1S/\"+y,\n        map(\n            lambda x: \"\".join([f\"{c}{x[c] if x[c]>1 else ''}\" for c in x if x[c]]),\n            test_labels.to_dict(orient=\"records\")\n        )\n    )\n), columns=[\"InChI\"]).join(test_df.image_id)[['image_id', 'InChI']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":" ","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}