{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.9","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":22422,"databundleVersionId":2153105,"sourceType":"competition"},{"sourceId":2010628,"sourceType":"datasetVersion","datasetId":1203186},{"sourceId":56429026,"sourceType":"kernelVersion"}],"dockerImageVersionId":30068,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1 style=\"text-align: font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: navy;\" id=\"imports\">0&nbsp;&nbsp;IMPORTS</h1>","metadata":{}},{"cell_type":"code","source":"# FOR KERAS UTILS PLOT\n!pip install -q pydot\n!pip install -q pydotplus\n!apt-get install -q graphviz\nfrom Levenshtein import distance\n\nprint(\"\\n... IMPORTS STARTING ...\\n\")\nprint(\"\\n\\tVERSION INFORMATION\")\n\n# Machine Learning and Data Science Imports\nimport tensorflow as tf; print(f\"\\t\\t– TENSORFLOW VERSION: {tf.__version__}\");\nimport pandas as pd; pd.options.mode.chained_assignment = None;\nimport numpy as np; print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\");\n\n# Built In Imports\nfrom kaggle_datasets import KaggleDatasets\nfrom collections import Counter\nfrom datetime import datetime\nfrom glob import glob\nimport warnings\nimport requests\nimport imageio\nimport IPython\nimport urllib\nimport zipfile\nimport pickle\nimport random\nimport shutil\nimport string\nimport math\nimport tqdm\nimport time\nimport gzip\nimport ast\nimport io\nimport os\nimport gc\nimport re\n\n# Visualization Imports\nfrom matplotlib.colors import ListedColormap\nimport matplotlib.patches as patches\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\nimport plotly.express as px\nimport seaborn as sns\nfrom PIL import Image\nimport matplotlib; print(f\"\\t\\t– MATPLOTLIB VERSION: {matplotlib.__version__}\");\nimport plotly\nimport PIL\nimport cv2\n\n\nprint(\"\\n\\n... IMPORTS COMPLETE ...\\n\")\n\n\nprint(\"\\n\\n... TPU SETUP STARTING ...\\n\")\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() # TPU detection\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\n    print(\"Running on TPU:\", tpu.master())\nexcept ValueError: # no TPU found, detect GPUs\n    strategy = tf.distribute.get_strategy() # for GPU or multi-GPU machines\n    print(\"\\n... USING GPU ...\\n\")\n    # Stop Tensorflow From Eating All The Memory\n    gpus = tf.config.experimental.list_physical_devices('GPU')\n    if gpus:\n        try:\n            # Currently, memory growth needs to be the same across GPUs\n            for gpu in gpus:\n                tf.config.experimental.set_memory_growth(gpu, True)\n            logical_gpus = tf.config.experimental.list_logical_devices('GPU')\n            print(len(gpus), \"... Physical GPUs,\", len(logical_gpus), \"Logical GPUs ...\\n\")\n        except RuntimeError as e:\n            # Memory growth must be set before GPUs have been initialized\n            print(e)\n        \nN_REPLICAS = strategy.num_replicas_in_sync\nprint(f\"... Number Of Accelerators: {N_REPLICAS} ...\\n\")\nprint(\"\\n\\n... TPU SETUP COMPLETE COMPLETE ...\\n\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-23T11:43:21.140427Z","iopub.execute_input":"2023-11-23T11:43:21.140841Z","iopub.status.idle":"2023-11-23T11:43:38.539452Z","shell.execute_reply.started":"2023-11-23T11:43:21.140806Z","shell.execute_reply":"2023-11-23T11:43:38.538289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"text-align: font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: navy; background-color: #ffffff;\" id=\"background_information\">1&nbsp;&nbsp;BACKGROUND INFORMATION</h1>","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"text-align: font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: navy; background-color: #ffffff;\" id=\"setup\">2&nbsp;&nbsp;SETUP</h1>","metadata":{}},{"cell_type":"code","source":"ALL_ELEMENTS = [\"H\", \" He\", \" Li\", \" Be\", \" B\", \" C\", \" N\", \" O\", \" F\", \" Ne\", \" Na\", \" Mg\", \" Al\", \" Si\", \" P\", \n                \"S\", \" Cl\", \" Ar\", \" K\", \" Ca\", \" Sc\", \" Ti\", \" V\", \" Cr\", \" Mn\", \" Fe\", \" Co\", \" Ni\", \" Cu\", \" Zn\", \n                \"Ga\", \" Ge\", \" As\", \" Se\", \" Br\", \" Kr\", \" Rb\", \" Sr\", \" Y\", \" Zr\", \" Nb\", \" Mo\", \" Tc\", \" Ru\", \n                \"Rh\", \" Pd\", \" Ag\", \" Cd\", \" In\", \" Sn\", \" Sb\", \" Te\", \" I\", \" Xe\", \" Cs\", \" Ba\", \" La\", \" Ce\", \n                \"Pr\", \" Nd\", \" Pm\", \" Sm\", \" Eu\", \" Gd\", \" Tb\", \" Dy\", \" Ho\", \" Er\", \" Tm\", \" Yb\", \" Lu\", \" Hf\", \n                \"Ta\", \" W\", \" Re\", \" Os\", \" Ir\", \" Pt\", \" Au\", \" Hg\", \" Tl\", \" Pb\", \" Bi\", \" Po\", \" At\", \" Rn\", \n                \"Fr\", \" Ra\", \" Ac\", \" Th\", \" Pa\", \" U\", \" Np\", \" Pu\", \" Am\", \" Cm\", \" Bk\", \" Cf\", \" Es\", \" Fm\", \n                \"Md\", \" No\", \" Lr\", \" Rf\", \" Db\", \" Sg\", \" Bh\", \" Hs\", \" Mt\", \" Ds\", \" Rg\", \" Cn\", \" Uut\", \n                \"Fl\", \" Uup\", \" Lv\", \" Uus\", \" Uuo\"]\n\n# In Order\nTRAIN_ELEMENTS = ['C', 'H', 'B', 'Br', 'Cl', 'F', 'I', 'N', 'O', 'P', 'S', 'Si']\n\n# Prefixes and How Common\n#      -- ORDERING --> {c}{h/None}{b/None}{t/None}{m/None}{s/None}{i/None}{h/None}{t/None}{m/None}\nPREFIX_ORDERING = \"chbtmsihtm\"\n\n# Define the root and data directories\nROOT_DIR = \"/kaggle/input\"\n\n# try:\n#     DATA_DIR = KaggleDatasets().get_gcs_path(\"bms-molecular-translation\")\n#     print(\"USING TPU (GCS) DATA DIRECTORY\")\n# except:\n#     print(\"NOT USING TPU\")\nDATA_DIR = os.path.join(ROOT_DIR, \"bms-molecular-translation\")\n\nTRAIN_DIR = os.path.join(DATA_DIR, \"train\")\nTEST_DIR = os.path.join(DATA_DIR, \"test\")\n\nTRAIN_CSV_PATH = os.path.join(DATA_DIR, \"train_labels.csv\")\nSS_CSV_PATH = os.path.join(DATA_DIR, \"sample_submission.csv\")\n\ntrain_df = pd.read_csv(TRAIN_CSV_PATH)\ntrain_df[\"img_path\"] = train_df.image_id.apply(lambda x: os.path.join(TRAIN_DIR, x[0], x[1], x[2], x+\".png\"))\nprint(\"\\n... TRAIN DATAFRAME W/ PATHS ...\\n\")\ndisplay(train_df)\n\nss_df = pd.read_csv(SS_CSV_PATH)\nprint(\"\\n... SUBMISSION DATAFRAME ...\\n\")\ndisplay(ss_df)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T11:44:20.397762Z","iopub.execute_input":"2023-11-23T11:44:20.398148Z","iopub.status.idle":"2023-11-23T11:44:38.713942Z","shell.execute_reply.started":"2023-11-23T11:44:20.398114Z","shell.execute_reply":"2023-11-23T11:44:38.712813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"text-align: font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: navy; background-color: #ffffff;\" id=\"helper_functions\">3&nbsp;&nbsp;HELPER FUNCTIONS</h1>","metadata":{}},{"cell_type":"code","source":"def flatten_l_o_l(nested_list):\n    return [item for sublist in nested_list for item in sublist]\n\ndef get_sublayer(split_list, prefix=\"c\"):\n    if not split_list:\n        return -1\n    else:\n        return split_list.pop(0)[1:]\n    \n    \ndef get_sublayer_hard(split_list, prefix=\"t\"):\n    if not split_list:\n        pass\n    else:\n        inchi_idx_list = [i for i,inchi_str in enumerate(split_list) if prefix in inchi_str]\n        if inchi_idx_list:\n            return split_list.pop(inchi_idx_list[0])[1:]\n        else:\n            pass\n        \n        \n# TRAIN --> ['S', 'Br', 'H', 'P', 'O', 'B', 'Si', 'N', 'C', 'I', 'F', 'Cl']\n# The chemical formula is represented according to Hill convention, that is, \n# - beginning with carbon atoms\n# - then hydrogens\n# - then all other elements in alphabetical order\ndef identify_relevant_elements(df):\n    tmp_chem_list = df[\"inchi_chemical_formula\"].apply(lambda x: re.split(r'(\\d+)', x))\n    tmp_chem_list = [x for x in flatten_l_o_l(tmp_chem_list) if (not x.isnumeric() and x!=\"\")]\n    return list(set(flatten_l_o_l([re.findall('[A-Z][^A-Z]*',x) for x in tmp_chem_list])))\n\n\ndef chem_form_list_to_columns(chem_form_list, all_possible_elem_list):\n    all_possible_elems = {e:0 for e in all_possible_elem_list}\n    for e, val in chem_form_list:\n        if val==\"\":\n            val=1\n        else:\n            val=int(val)\n        all_possible_elems[e]=val\n    return [all_possible_elems[e] for e in all_possible_elem_list]\n\n\ndef tf_load_image(path, img_size=(224,224), tile_to_3_channel=True, for_tpu=False):\n    img = decode_img(tf.io.read_file(path), img_size, n_channels=1, for_tpu=for_tpu)\n    if tile_to_3_channel:\n        return tf.tile(img, tf.constant((1, 1, 3), dtype=tf.int32))\n    else:\n        return img\n\ndef decode_img(img, img_size=(224,224), n_channels=1, for_tpu=False):\n    \"\"\"TBD\"\"\"\n    \n    # convert the compressed string to a 3D uint8 tensor\n    img = tf.image.decode_png(img, channels=n_channels)\n\n    # resize the image to the desired size\n    if for_tpu:\n        return tf.cast(tf.image.resize(img, img_size), tf.float32)\n    else:\n        return tf.cast(tf.image.resize(img, img_size), tf.uint8)\n    \n    \ndef convert_pred_batch_to_str_batch(pred_batch):\n    return [\"\".join([\n        f\"{e}{int(c)}\" if c>1 else f\"{e}\" \\\n        for (c, e) in [\n            *zip(pred[np.nonzero(pred)], np.array(TRAIN_ELEMENTS)[np.nonzero(pred)])\n        ]]) for pred in np.round(pred_batch)]","metadata":{"execution":{"iopub.status.busy":"2023-11-23T11:44:38.736075Z","iopub.execute_input":"2023-11-23T11:44:38.736467Z","iopub.status.idle":"2023-11-23T11:44:38.755097Z","shell.execute_reply.started":"2023-11-23T11:44:38.736428Z","shell.execute_reply":"2023-11-23T11:44:38.754413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"prefixes\"] = train_df.InChI.apply(lambda x: [xx[0] for xx in x.split(\"/\")[2:]])\nprefix_combos = train_df[\"prefixes\"].apply(lambda x: \"\".join(x)).value_counts()\ndisplay(prefix_combos)\n\n# Housekeeping Stuff\ntrain_df[\"inchi_split\"] = train_df.InChI.str.split(\"/\")\ntrain_df[\"inchi_root\"] = train_df.inchi_split.apply(lambda x: x.pop(0))\n\n# Guaranteed Stuff\ntrain_df[\"inchi_chemical_formula\"] = train_df.inchi_split.apply(lambda x: x.pop(0))\ntrain_df[\"inchi_atom_connections_/c\"] = train_df.inchi_split.apply(lambda x: get_sublayer(x, \"c\"))\n\n# Optional Stuff - Note there are no included /q and /p layers\ntrain_df[\"inchi_hydrogen_atoms_/h\"] = train_df.inchi_split.apply(lambda x: get_sublayer(x, \"h\"))\ntrain_df[\"inchi_charge_/q\"] = None\ntrain_df[\"inchi_proton_/p\"] = None\n\ntrain_df[\"inchi_db_and_c_/b\"] = train_df.inchi_split.apply(lambda x: get_sublayer_hard(x, \"b\"))\ntrain_df[\"inchi_tetrahedral_/t\"] = train_df.inchi_split.apply(lambda x: get_sublayer_hard(x, \"t\"))\ntrain_df[\"inchi_tetrahedral_/m\"] = train_df.inchi_split.apply(lambda x: get_sublayer_hard(x, \"m\"))\ntrain_df[\"inchi_stereo_info_/s\"] = train_df.inchi_split.apply(lambda x: get_sublayer_hard(x, \"s\"))\n\ntrain_df[\"inchi_isotopic_/s\"] = train_df.inchi_split.apply(lambda x: get_sublayer_hard(x, \"i\"))\ntrain_df[\"inchi_isotopic_/h\"] = train_df.inchi_split.apply(lambda x: get_sublayer_hard(x, \"h\"))\n\ntrain_df[\"inchi_tetrahedral_secondary_/t\"] = train_df.inchi_split.apply(lambda x: get_sublayer_hard(x, \"t\"))\ntrain_df[\"inchi_tetrahedral_secondary_/m\"] = train_df.inchi_split.apply(lambda x: get_sublayer_hard(x, \"m\"))\n\ntrain_df = train_df.drop(columns=[\"inchi_split\"])\n\n# Non NAN Counts\ndisplay(train_df.count())\ndisplay(train_df.head())\n\ntrain_df[\"chemical_formula_list\"] = train_df[\"inchi_chemical_formula\"].apply(lambda x: re.findall(r'([A-Z][a-z]*)([0-9]*)', x))\ntrain_df[\"chemical_formula_list\"] = train_df[\"chemical_formula_list\"].apply(lambda x: chem_form_list_to_columns(x, TRAIN_ELEMENTS))\ntrain_df = pd.concat([train_df, pd.DataFrame(train_df[\"chemical_formula_list\"].to_list(), columns=[f\"{e}_count\" for e in TRAIN_ELEMENTS])], axis=1)\ntrain_df = train_df.drop(columns=[\"chemical_formula_list\"])\n\ndisplay(train_df)\ndisplay(train_df[[\"inchi_chemical_formula\"]+[f\"{e}_count\" for e in TRAIN_ELEMENTS]])","metadata":{"execution":{"iopub.status.busy":"2023-11-23T11:44:42.878753Z","iopub.execute_input":"2023-11-23T11:44:42.879134Z","iopub.status.idle":"2023-11-23T11:46:11.650447Z","shell.execute_reply.started":"2023-11-23T11:44:42.879097Z","shell.execute_reply":"2023-11-23T11:46:11.64943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"text-align: font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: navy; background-color: #ffffff;\" id=\"examine_identicals\">4&nbsp;&nbsp;INVESTIGATION OF IDENTICAL MOLECULAR STRUCTURES</h1>","metadata":{}},{"cell_type":"code","source":"repeated_formulas = [k for k,v in train_df[\"inchi_chemical_formula\"].value_counts().items() if v>1]\ndemo_idx = 0\ndemo_formula = repeated_formulas[demo_idx]\ntrain_df_formula_subset = train_df[train_df[\"inchi_chemical_formula\"]==demo_formula].reset_index(drop=True)\nrow_to_display = 4\n\nplt.figure(figsize=(18,5*row_to_display))\nplt.suptitle(f\"{demo_formula}\", fontsize=20)\nfor i in range(min(4*row_to_display, len(train_df_formula_subset))):\n    plt.subplot(5,4,i+1)\n    plt.imshow(np.asarray(Image.open(train_df_formula_subset.img_path[i])), cmap=\"gray\")\n    plt.title(\"{}\".format('\\n'.join(train_df_formula_subset.InChI[i].split('/')[2:])), fontweight=\"bold\", fontsize=9)\n    \nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-23T11:50:45.339099Z","iopub.execute_input":"2023-11-23T11:50:45.339534Z","iopub.status.idle":"2023-11-23T11:50:49.756476Z","shell.execute_reply.started":"2023-11-23T11:50:45.339492Z","shell.execute_reply":"2023-11-23T11:50:49.7556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"repeated_formulas = [k for k,v in train_df[\"inchi_chemical_formula\"].value_counts().items() if v>1]\ndemo_idx = 10\ndemo_formula = repeated_formulas[demo_idx]\ntrain_df_formula_subset = train_df[train_df[\"inchi_chemical_formula\"]==demo_formula].reset_index(drop=True)\nrow_to_display = 4\n\nplt.figure(figsize=(18,5*row_to_display))\nplt.suptitle(f\"{demo_formula}\", fontsize=20)\nfor i in range(min(4*row_to_display, len(train_df_formula_subset))):\n    plt.subplot(5,4,i+1)\n    plt.imshow(np.asarray(Image.open(train_df_formula_subset.img_path[i])), cmap=\"gray\")\n    plt.title(\"{}\".format('\\n'.join(train_df_formula_subset.InChI[i].split('/')[2:])), fontweight=\"bold\", fontsize=9)\n    \nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-23T11:51:10.441157Z","iopub.execute_input":"2023-11-23T11:51:10.441529Z","iopub.status.idle":"2023-11-23T11:51:14.552859Z","shell.execute_reply.started":"2023-11-23T11:51:10.441493Z","shell.execute_reply":"2023-11-23T11:51:14.552075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"text-align: font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: navy; background-color: #ffffff;\" id=\"baseline_model\">5&nbsp;&nbsp;BASELINE MODEL</h1>","metadata":{}},{"cell_type":"code","source":"train_df.image_id=train_df.image_id.astype(str)\ntrain_df.img_path=train_df.img_path.astype(str)\ntrain_df.inchi_chemical_formula=train_df.inchi_chemical_formula.astype(str)\n_ = [train_df[c].fillna(0, inplace=True) for c in train_df.columns if c.endswith(\"_count\")]\n\nVAL_SPLIT = 0.05\nVAL_INDICES = random.sample(range(len(train_df)), round(len(train_df)*VAL_SPLIT))\n\nval_df = train_df.iloc[VAL_INDICES]\ntrain_df = train_df[~train_df.index.isin(VAL_INDICES)]\n\nfeature_columns = [\"image_id\", \"img_path\"]\nlabel_columns = [f\"{e}_count\" for e in TRAIN_ELEMENTS]\n\ntrain_feature_ds = tf.data.Dataset.zip(tuple([tf.data.Dataset.from_tensor_slices(train_df[x].values) for x in feature_columns]))\ntrain_label_ds = tf.data.Dataset.from_tensor_slices(train_df[label_columns].values)\ntrain_ds = tf.data.Dataset.zip((train_feature_ds,train_label_ds))\n\nval_feature_ds = tf.data.Dataset.zip(tuple([tf.data.Dataset.from_tensor_slices(val_df[x].values) for x in feature_columns]))\nval_label_ds = tf.data.Dataset.from_tensor_slices(val_df[label_columns].values)\nval_ds = tf.data.Dataset.zip((val_feature_ds,val_label_ds))\n\ndel train_label_ds, train_feature_ds; gc.collect(); gc.collect();\ndel val_label_ds, val_feature_ds; gc.collect(); gc.collect();\n\nprint(train_ds)\nprint(val_ds)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T11:52:11.160321Z","iopub.execute_input":"2023-11-23T11:52:11.160712Z","iopub.status.idle":"2023-11-23T11:52:20.660839Z","shell.execute_reply.started":"2023-11-23T11:52:11.160662Z","shell.execute_reply":"2023-11-23T11:52:20.659993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MODEL_CKPT_DIR = \"/kaggle/working/eb0_ckpts\"\n\n# BATCH_SIZE=BATCH_SIZE_PER_REPLICA*N_REPLICAS\nBATCH_SIZE = 64\nAUTOTUNE   = tf.data.AUTOTUNE\nSHUFF_BUFF = 256\n\n# Drop Image ID for now and open the image and stack y values as a single vector\n# if N_REPLICAS==8: # TPU PATH\n#     SAVE_LOCAL = tf.saved_model.SaveOptions(experimental_io_device='/job:localhost')\n#     train_ds = train_ds.map(lambda x,y: ((tf_load_image(x[1], for_tpu=True)), tf.cast(y, tf.float32)))\n#     val_ds = val_ds.map(lambda x,y: ((tf_load_image(x[1], for_tpu=True)), tf.cast(y, tf.float32)))\n# else:\n# TRAIN_CACHE_DIR = \"/kaggle/train_cache\"\n# VAL_CACHE_DIR = \"/kaggle/val_cache\"\n# if not os.path.isdir(TRAIN_CACHE_DIR):\n#     os.makedirs(TRAIN_CACHE_DIR, exist_ok=True)\n# if not os.path.isdir(VAL_CACHE_DIR):\n#     os.makedirs(VAL_CACHE_DIR, exist_ok=True)\nSAVE_LOCAL=None\ntrain_ds = train_ds.map(lambda x,y: ((tf_load_image(x[1])), tf.cast(y, tf.int32)))#.cache(TRAIN_CACHE_DIR)\nval_ds = val_ds.map(lambda x,y: ((tf_load_image(x[1])), tf.cast(y, tf.int32)))#.cache(VAL_CACHE_DIR)\n\ntrain_ds = train_ds.shuffle(SHUFF_BUFF) \\\n                   .batch(BATCH_SIZE) \\\n                   .prefetch(AUTOTUNE)\n\nval_ds = val_ds.batch(BATCH_SIZE) \\\n               .prefetch(AUTOTUNE)\n\nprint(train_ds, val_ds)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T11:52:57.877967Z","iopub.execute_input":"2023-11-23T11:52:57.878332Z","iopub.status.idle":"2023-11-23T11:52:57.964293Z","shell.execute_reply.started":"2023-11-23T11:52:57.8783Z","shell.execute_reply":"2023-11-23T11:52:57.963368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_EPOCHS=4\nLR_START = 0.0005\nLR_MAX = 0.0015\nLR_MIN = 0.00075\nLR_RAMPUP_EPOCHS = 2\nLR_SUSTAIN_EPOCHS = 1\nLR_EXP_DECAY = 0.4\n\ndef lrfn(epoch):\n    if epoch < LR_RAMPUP_EPOCHS:\n        lr = (LR_MAX - LR_START) / LR_RAMPUP_EPOCHS * epoch + LR_START\n    elif epoch < LR_RAMPUP_EPOCHS + LR_SUSTAIN_EPOCHS:\n        lr = LR_MAX\n    else:\n        lr = (LR_MAX - LR_MIN) * LR_EXP_DECAY**(epoch - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS) + LR_MIN\n    return lr\n\n# VIEW SCHEDULE\nrng = [i for i in range(N_EPOCHS)]\ny = [lrfn(x) for x in rng]\n\nplt.figure(figsize=(10,4))\nplt.plot(rng, y)\nplt.title(\"CUSTOM LR SCHEDULE\", fontweight=\"bold\")\nplt.show()\n\nprint(f\"Learning rate schedule: {y[0]:.3g} to {max(y):.3g} to {y[-1]:.3g}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-23T11:53:02.359344Z","iopub.execute_input":"2023-11-23T11:53:02.359716Z","iopub.status.idle":"2023-11-23T11:53:02.517003Z","shell.execute_reply.started":"2023-11-23T11:53:02.359668Z","shell.execute_reply":"2023-11-23T11:53:02.516237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_backbone(efficientnet_name=\"efficientnet_b0\", input_shape=(224,224,3), include_top=False, weights=\"imagenet\", pooling=\"avg\"):\n    if \"b0\" in efficientnet_name:\n        eb = tf.keras.applications.EfficientNetB0(\n            include_top=include_top, weights=weights, pooling=pooling, input_shape=input_shape\n            )\n    elif \"b1\" in efficientnet_name:\n        eb = tf.keras.applications.EfficientNetB1(\n            include_top=include_top, weights=weights, pooling=pooling, input_shape=input_shape\n            )\n    elif \"b2\" in efficientnet_name:\n        eb = tf.keras.applications.EfficientNetB2(\n            include_top=include_top, weights=weights, pooling=pooling, input_shape=input_shape\n            )\n    elif \"b3\" in efficientnet_name:\n        eb = tf.keras.applications.EfficientNetB3(\n            include_top=include_top, weights=weights, pooling=pooling, input_shape=input_shape\n            )\n    elif \"b4\" in efficientnet_name:\n        eb = tf.keras.applications.EfficientNetB4(\n            include_top=include_top, weights=weights, pooling=pooling, input_shape=input_shape\n            )\n    elif \"b5\" in efficientnet_name:\n        eb = tf.keras.applications.EfficientNetB5(\n            include_top=include_top, weights=weights, pooling=pooling, input_shape=input_shape\n            )\n    elif \"b6\" in efficientnet_name:\n        eb = tf.keras.applications.EfficientNetB6(\n            include_top=include_top, weights=weights, pooling=pooling, input_shape=input_shape\n            )\n    elif \"b7\" in efficientnet_name:\n        eb = tf.keras.applications.EfficientNetB7(\n            include_top=include_top, weights=weights, pooling=pooling, input_shape=input_shape\n            )\n    else:\n        raise ValueError(\"Invalid EfficientNet Name!!!\")\n    return eb\n\n\ndef add_head_to_bb(bb, n_outputs=19, dropout=0.05, head_layer_nodes=(512,)):\n    x = tf.keras.layers.BatchNormalization()(bb.output)\n    x = tf.keras.layers.Dropout(dropout)(x)\n    \n    for n_nodes in head_layer_nodes:\n        x = tf.keras.layers.Dense(n_nodes, activation=\"relu\")(x)\n        x = tf.keras.layers.BatchNormalization()(x)\n        x = tf.keras.layers.Dropout(dropout/2)(x)\n    \n    output = tf.keras.layers.Dense(n_outputs, activation=\"linear\")(x)\n    return tf.keras.Model(inputs=bb.inputs, outputs=output)\n\n# tf.keras.backend.clear_session()\n# eb0 = get_backbone(\"b0\")\n# eb0 = add_head_to_bb(eb0, len(label_columns), dropout=0.4)\n# eb0.compile(optimizer=\"adam\", loss=tf.keras.losses.Huber(), metrics=[\"acc\", \"mae\"])\n\n# WE SKIP LOADING FRESH FOR DEMO PURPOSES\n# tf.keras.backend.clear_session()\neb0 = tf.keras.models.load_model(\"../input/moleculartranslationbaselinemodel/eb0_ckpts/ckpt-0001-2.0007.ckpt\")\ntf.keras.utils.plot_model(eb0, show_shapes=True, show_dtype=True, dpi=55)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T11:53:05.578433Z","iopub.execute_input":"2023-11-23T11:53:05.578858Z","iopub.status.idle":"2023-11-23T11:53:19.954382Z","shell.execute_reply.started":"2023-11-23T11:53:05.578817Z","shell.execute_reply":"2023-11-23T11:53:19.952885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# history = eb0.fit(\n#     train_ds.take(20), # DEMO PURPOSES\n#     validation_data=val_ds.take(10), # LEAVE AS FULL VAL SET FOR DEMO TO SEE PERFORMANCE ON VAL *~7-10 minutes~*\n#     callbacks=[\n#         tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=3, verbose=1, restore_best_weights=True),\n#         tf.keras.callbacks.ModelCheckpoint(filepath=os.path.join(MODEL_CKPT_DIR, \"ckpt-{epoch:04d}-{val_loss:.4f}.ckpt\"), verbose=1),\n#         tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=1)\n#     ], \n#     epochs=1, # DEMO PURPOSES\n# )\n# eb0.save(os.path.join(MODEL_CKPT_DIR, \"final\"))\n# eb0.evaluate(val_ds.take(10))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"repeated_formulas = [k for k,v in train_df[\"inchi_chemical_formula\"].value_counts().items() if v>1]\ndemo_idx = 2\ndemo_formula = repeated_formulas[demo_idx]\ntrain_df_formula_subset = train_df[train_df[\"inchi_chemical_formula\"]==demo_formula].reset_index(drop=True)\n\nbatch_to_infer_on = tf.stack([tf_load_image(path) for path in train_df_formula_subset.img_path.values[:16]])\nprint(f\"\\n... GT FORMULA = {demo_formula} ...\\n\")\npreds = eb0(batch_to_infer_on)\n\nprint(f\"\\nRANDOM STRING HAS A LEVENSHTEIN DISTANCE: {distance(demo_formula, 'C1H1O1N10Br4')}\\n\" \\\n      f\"\\t--> DEMO={demo_formula}\\n\" \\\n      f\"\\t--> RANDOM=C1HNO10Br4\\n\\n\")\nfor i, y_pred in enumerate(convert_pred_batch_to_str_batch(preds)):\n    print(f\"PRED {i+1} HAS A LEVENSHTEIN DISTANCE OF {distance(y_pred, demo_formula)}\\n\" \\\n          f\"\\t-->DEMO={demo_formula}\\n\\t-->PRED={y_pred}\\n\")","metadata":{"execution":{"iopub.status.busy":"2023-11-23T11:55:17.841347Z","iopub.execute_input":"2023-11-23T11:55:17.84178Z","iopub.status.idle":"2023-11-23T11:55:19.907952Z","shell.execute_reply.started":"2023-11-23T11:55:17.841738Z","shell.execute_reply":"2023-11-23T11:55:19.907061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Capture the image ids (paths) for the test set\nss_image_paths =  ss_df.image_id.apply(lambda x: os.path.join(TEST_DIR, x[0], x[1], x[2], x+\".png\")).values\n\ntest_feature_ds = tf.data.Dataset.from_tensor_slices(ss_image_paths)\ntest_ds = test_feature_ds.map(tf_load_image)\ntest_ds = test_ds.batch(BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T11:58:23.514076Z","iopub.execute_input":"2023-11-23T11:58:23.514471Z","iopub.status.idle":"2023-11-23T11:58:31.086978Z","shell.execute_reply.started":"2023-11-23T11:58:23.514437Z","shell.execute_reply":"2023-11-23T11:58:31.086022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\ndef tf_load_image(image_path):\n    # Read the image file\n    image = tf.io.read_file(image_path)\n    \n    # Decode the PNG image (you might need to adjust this based on your image format)\n    image = tf.image.decode_png(image, channels=3)\n    \n    # Normalize pixel values to the range [0, 1]\n    image = tf.image.convert_image_dtype(image, tf.float32)\n    \n    # Resize the image to your desired dimensions (e.g., 224x224)\n    image = tf.image.resize(image, [224, 224])\n    \n    return image\n\n# Test the tf_load_image function\ntest_image_path = ss_image_paths[0]  # Assuming there's at least one image path\ntest_loaded_image = tf_load_image(test_image_path)\n\n# Print the shape of the loaded image\nprint(\"Loaded image shape:\", test_loaded_image.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-23T11:59:10.08651Z","iopub.execute_input":"2023-11-23T11:59:10.086907Z","iopub.status.idle":"2023-11-23T11:59:10.110623Z","shell.execute_reply.started":"2023-11-23T11:59:10.086874Z","shell.execute_reply":"2023-11-23T11:59:10.109727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.columns","metadata":{"execution":{"iopub.status.busy":"2023-11-23T11:16:07.187936Z","iopub.execute_input":"2023-11-23T11:16:07.188314Z","iopub.status.idle":"2023-11-23T11:16:07.194728Z","shell.execute_reply.started":"2023-11-23T11:16:07.188282Z","shell.execute_reply":"2023-11-23T11:16:07.19376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.head())","metadata":{"execution":{"iopub.status.busy":"2023-11-23T11:26:00.301944Z","iopub.execute_input":"2023-11-23T11:26:00.302357Z","iopub.status.idle":"2023-11-23T11:26:00.310095Z","shell.execute_reply.started":"2023-11-23T11:26:00.302325Z","shell.execute_reply":"2023-11-23T11:26:00.309174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_path = \"/kaggle/input/bms-molecular-translation/extra_approved_InChIs.csv\"  # Replace with the actual path\ntest_df = pd.read_csv(test_data_path)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:04:11.204438Z","iopub.execute_input":"2023-11-23T12:04:11.204974Z","iopub.status.idle":"2023-11-23T12:04:27.674727Z","shell.execute_reply.started":"2023-11-23T12:04:11.204927Z","shell.execute_reply":"2023-11-23T12:04:27.673881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.columns","metadata":{"execution":{"iopub.status.busy":"2023-11-23T11:16:45.27274Z","iopub.execute_input":"2023-11-23T11:16:45.2731Z","iopub.status.idle":"2023-11-23T11:16:45.278969Z","shell.execute_reply.started":"2023-11-23T11:16:45.273068Z","shell.execute_reply":"2023-11-23T11:16:45.278104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate over the submission dataframe and generate predictions for chemical formula\nchem_formula_preds = np.zeros((len(ss_image_paths,))).astype(str)\nfor i, tf_batch in enumerate(tqdm(test_ds)):\n    chem_formula_preds[i*BATCH_SIZE:(i+1)*BATCH_SIZE] = convert_pred_batch_to_str_batch(eb0(tf_batch))\n    if i%1000==0:\n        tf.keras.backend.clear_session()\n        eb0 = tf.keras.models.load_model(\"../input/moleculartranslationbaselinemodel/eb0_ckpts/ckpt-0001-2.0007.ckpt\")\n        gc.collect(); gc.collect();\n        \nwith open('/kaggle/working/preds.pickle', 'wb') as handle:\n    pickle.dump(chem_formula_preds, handle)\n\n\n# with open('../input/molecular-translation-eda-smart-baseline/preds.pickle', 'rb') as f:\n#     chem_formula_preds = pickle.load(f)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:07:10.786458Z","iopub.execute_input":"2023-11-23T12:07:10.786895Z","iopub.status.idle":"2023-11-23T12:12:30.511695Z","shell.execute_reply.started":"2023-11-23T12:07:10.786856Z","shell.execute_reply":"2023-11-23T12:12:30.510106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(chem_formula_preds)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chem_formula_preds = [str(x) for x in chem_formula_preds]\nss_df[\"InChI_build\"] = [f\"InChI=1S/{x}/c\"+x+\"/c\" for x in chem_formula_preds]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"common_C = train_df[\"inchi_atom_connections_/c\"].value_counts(dropna=False).index[0]\ncommon_H = train_df[\"inchi_hydrogen_atoms_/h\"].value_counts(dropna=False).index[0]\ncommon_B = train_df[\"inchi_db_and_c_/b\"].value_counts(dropna=False).index[0]\ncommon_T1 = train_df[\"inchi_tetrahedral_/t\"].value_counts(dropna=False).index[0]\ncommon_M1 = train_df[\"inchi_tetrahedral_/m\"].value_counts(dropna=False).index[0]\ncommon_S1 = train_df[\"inchi_stereo_info_/s\"].value_counts(dropna=False).index[0]\ncommon_S2 = train_df[\"inchi_isotopic_/s\"].value_counts(dropna=False).index[0]\ncommon_H = train_df[\"inchi_isotopic_/h\"].value_counts(dropna=False).index[0]\ncommon_T2 = train_df[\"inchi_tetrahedral_secondary_/t\"].value_counts(dropna=False).index[0]\ncommon_M2 = train_df[\"inchi_tetrahedral_secondary_/m\"].value_counts(dropna=False).index[0]\n\nfor x in [common_C, common_H, common_B, common_T1, common_M1, common_S1, common_S2, common_H, common_T2, common_M2,]: print(x)","metadata":{"execution":{"iopub.status.busy":"2023-11-23T10:37:18.012343Z","iopub.execute_input":"2023-11-23T10:37:18.012629Z","iopub.status.idle":"2023-11-23T10:37:24.169787Z","shell.execute_reply.started":"2023-11-23T10:37:18.0126Z","shell.execute_reply":"2023-11-23T10:37:24.168831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss_df[\"InChI\"] = ss_df[\"InChI_build\"]+common_C\nss_df = ss_df.drop(columns=[\"InChI_build\"])\nss_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inchi_submissions = []\n# for pred in chem_formula_preds:\n#     inchi_submissions.append(train_df.iloc[np.argmin((train_df[[f\"{e}_count\" for e in TRAIN_ELEMENTS]]-pred).abs().sum(axis=1))].InChI)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss_df.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**knn **","metadata":{}},{"cell_type":"code","source":"from tqdm.notebook import tqdm\n","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:17:20.215117Z","iopub.execute_input":"2023-11-23T12:17:20.215469Z","iopub.status.idle":"2023-11-23T12:17:20.220078Z","shell.execute_reply.started":"2023-11-23T12:17:20.215439Z","shell.execute_reply":"2023-11-23T12:17:20.21881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\nlabels = pd.read_csv('/kaggle/input/bms-molecular-translation/train_labels.csv')\nss = pd.read_csv('/kaggle/input/bms-molecular-translation/sample_submission.csv', index_col=0)\n\nlabels['path'] = labels['image_id'].apply(\n    lambda x: \"/kaggle/input/moleculartranslationbaselinemodel/model.png\".format(\n        x[0], x[1], x[2], x))\nlabels.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:24:12.517563Z","iopub.execute_input":"2023-11-23T12:24:12.517995Z","iopub.status.idle":"2023-11-23T12:24:20.41652Z","shell.execute_reply.started":"2023-11-23T12:24:12.517953Z","shell.execute_reply":"2023-11-23T12:24:20.415535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(labels.path[55])\nimg = plt.imread(labels.path[55])\n","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:28:34.705152Z","iopub.execute_input":"2023-11-23T12:28:34.70555Z","iopub.status.idle":"2023-11-23T12:28:35.100647Z","shell.execute_reply.started":"2023-11-23T12:28:34.705517Z","shell.execute_reply":"2023-11-23T12:28:35.099596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = plt.imread(labels.path[55])\nimg = clean_noise(img)\n\n# Convert to grayscale\nimg_gray = cv2.cvtColor(img, cv2.COLOR_RGBA2GRAY)\n\npimg = cv2.erode(img_gray, np.ones((3,3)), iterations=2)\ncimg = cv2.cvtColor(img_gray, cv2.COLOR_GRAY2RGB)\n\nplt.subplot(2, 2, 1)\nplt.title('Eroded Example')\nplt.imshow(pimg)\n\nplt.subplot(2, 2, 2)\nR = cv2.cornerHarris(np.float32(pimg), 7, 3, 0.04)\nR_dilate = cv2.dilate(R, np.ones((3,3)))\nR_nms = R >= R_dilate\nR_th = R > R.max() * 0.1\nR_final = R_th * R_nms\n[y, x] = np.nonzero(R_final)\n\nfor x, y in zip(x, y):\n    cimg = cv2.circle(cimg, (x, y), 3, (255, 0, 0))\n\nplt.title('Detected Vertices')\nplt.imshow(cimg)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:33:08.97453Z","iopub.execute_input":"2023-11-23T12:33:08.974978Z","iopub.status.idle":"2023-11-23T12:33:39.39851Z","shell.execute_reply.started":"2023-11-23T12:33:08.974945Z","shell.execute_reply":"2023-11-23T12:33:39.397488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sliding_window(image, stepSize, windowSize):\n    for y in range(0, image.shape[0], stepSize):\n        for x in range(0, image.shape[1], stepSize):\n            yield (x, y, image[y:y + windowSize[1], x:x + windowSize[0]])\n\n        \ndef clean_noise(image,stepSize=2,windowSize=(3,3)):\n    \n    \"\"\"\n     \n    Parameters\n    ----------\n    image : np.array\n        an image of a chemical compound\n    stepSize : int\n        The number of pixels the sliding window moves each step\n    windowSize : tuple-(int,int)\n        The size of the sliding window \n\n    \"\"\"\n    for y in range(0, image.shape[0], stepSize):\n        for x in range(0, image.shape[1], stepSize):\n            if (windowSize[0]**2 - image[y:y + windowSize[1], x:x + windowSize[0]].flatten().sum()) == 1:\n                image[y:y + windowSize[1], x:x + windowSize[0]] = 1\n    return image\n\n\ndef threshold_neighbours(distance_matrix,index,threshold):\n    \"\"\"\n     \n    Parameters\n    ----------\n    distance_matrix : np.array\n        The distnace matrix of each detected vertex with every other\n    index : int\n        The index of our target vertex\n    threshold : float\n        The maximum distance allowed before considering a vertex to close to our target i.e a neighbour\n\n    \"\"\"\n    return np.where(distance_matrix[index,:] <threshold)[0][1:]\n\ndef cluster_vertices(x,y,threshold=5,P=2,clustering_type=1):\n    \n    \"\"\"\n     \n    Parameters\n    ----------\n    x : np.array\n        The list of x coordinates of vertices\n    y : np.array\n        The list of y coordinates of vertices\n    threshold : float\n        The maximum distance allowed before considering a vertex to close to our target i.e a neighbour\n    P : float\n         Which Minkowski p-norm to use when calculating distance between vertices\n\n    \"\"\"\n    if clustering_type == 1:\n        #Calculate Distance Matrix\n        dm = distance_matrix(np.stack([x,y]).T,np.stack([x,y]).T,p=P)\n        removed = set()\n        #iterate an accumulate the vertices which are to close or overlaping others \n        for row in np.arange(0,len(x)):\n            if row not in removed:\n                removed = removed | set(threshold_neighbours(dm,row,threshold))\n\n        xx=np.take(x,list(set(np.arange(0,len(x)))-(removed)))\n        yy=np.take(y,list(set(np.arange(0,len(x)))-(removed)))\n        #return the clusterd vertecis\n        return xx,yy\n\ndef clean_vertices(img,x_nodes,y_nodes):\n    xx,yy = [],[]\n    for x,y in zip(x_nodes,y_nodes):\n        if img[yy,xx] < 1:\n            xx.append(x)\n            yy.append(y)\n    return np.array(xx),np.array(yy)\n\n\ndef tag_vertices(pimg, blockSize=7, apertureSize=13, harrisAlpha=0.04, cluster_threshold=7, nms_threshold=0.1, minkowski_p=2):\n    # Convert the image to grayscale\n    pimg_gray = cv2.cvtColor(pimg, cv2.COLOR_RGB2GRAY)\n    \n    # Ensure the image values are in the correct range (0 to 255)\n    pimg_gray = (pimg_gray * 255).astype(np.uint8)\n\n    # Detect Harris Corners According To Given Parameters\n    R = cv2.cornerHarris(np.float32(pimg_gray), blockSize, apertureSize, harrisAlpha)\n\n    # Preform Non Maxima Suprresion On Resulting R Matrix to Eliminate Weak Candidates \n    R_dilate = cv2.dilate(R, np.ones((3,3)))\n    R_nms = R >= R_dilate  # NMS\n    R_th = R > R.max() * nms_threshold  # threshold\n    R_final = R_th * R_nms  # threshold and NMS\n\n    [y, x] = np.nonzero(R_final)\n    xx, yy = cluster_vertices(x, y, cluster_threshold, minkowski_p)\n    return xx, yy\n\n# Rest of your code remains unchanged\n\n            \n    ","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-11-23T12:36:07.834292Z","iopub.execute_input":"2023-11-23T12:36:07.834735Z","iopub.status.idle":"2023-11-23T12:36:07.857824Z","shell.execute_reply.started":"2023-11-23T12:36:07.834671Z","shell.execute_reply":"2023-11-23T12:36:07.856876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape of X:\", X.shape)\nprint(\"Shape of y:\", y.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:49:14.931588Z","iopub.execute_input":"2023-11-23T12:49:14.931992Z","iopub.status.idle":"2023-11-23T12:49:14.937521Z","shell.execute_reply.started":"2023-11-23T12:49:14.931956Z","shell.execute_reply":"2023-11-23T12:49:14.936285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import svm, neighbors, linear_model\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\n\n# Generate random data\nnp.random.seed(42)  # for reproducibility\n\n# Assuming each data point has 100 features\nnum_samples = 1000\nnum_features = 100\n\n# Generate random features\nX = np.random.rand(num_samples, num_features)\n\n# Generate random labels (0 or 1)\ny = np.random.randint(2, size=num_samples)\n\n# Step 1: Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Step 2: Train classifiers\nsvm_classifier = svm.SVC()\nknn_classifier = neighbors.KNeighborsClassifier()\nlogistic_regression_classifier = linear_model.LogisticRegression()\n\nsvm_classifier.fit(X_train, y_train)\nknn_classifier.fit(X_train, y_train)\nlogistic_regression_classifier.fit(X_train, y_train)\n\n# Step 3: Evaluate classifiers on the testing set\nsvm_predictions = svm_classifier.predict(X_test)\nknn_predictions = knn_classifier.predict(X_test)\nlogistic_regression_predictions = logistic_regression_classifier.predict(X_test)\n\n# Step 4: Compare performance metrics\nsvm_accuracy = accuracy_score(y_test, svm_predictions)\nsvm_precision = precision_score(y_test, svm_predictions)\nsvm_recall = recall_score(y_test, svm_predictions)\nsvm_f1 = f1_score(y_test, svm_predictions)\n\nknn_accuracy = accuracy_score(y_test, knn_predictions)\nknn_precision = precision_score(y_test, knn_predictions)\nknn_recall = recall_score(y_test, knn_predictions)\nknn_f1 = f1_score(y_test, knn_predictions)\n\nlogistic_regression_accuracy = accuracy_score(y_test, logistic_regression_predictions)\nlogistic_regression_precision = precision_score(y_test, logistic_regression_predictions)\nlogistic_regression_recall = recall_score(y_test, logistic_regression_predictions)\nlogistic_regression_f1 = f1_score(y_test, logistic_regression_predictions)\n\n# Print or use the metrics as needed\nprint(\"SVM Accuracy:\", svm_accuracy)\nprint(\"SVM Precision:\", svm_precision)\nprint(\"SVM Recall:\", svm_recall)\nprint(\"SVM F1 Score:\", svm_f1)\n\nprint(\"KNN Accuracy:\", knn_accuracy)\nprint(\"KNN Precision:\", knn_precision)\nprint(\"KNN Recall:\", knn_recall)\nprint(\"KNN F1 Score:\", knn_f1)\n\nprint(\"Logistic Regression Accuracy:\", logistic_regression_accuracy)\nprint(\"Logistic Regression Precision:\", logistic_regression_precision)\nprint(\"Logistic Regression Recall:\", logistic_regression_recall)\nprint(\"Logistic Regression F1 Score:\", logistic_regression_f1)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-23T12:59:06.790629Z","iopub.execute_input":"2023-11-23T12:59:06.791018Z","iopub.status.idle":"2023-11-23T12:59:09.827856Z","shell.execute_reply.started":"2023-11-23T12:59:06.790986Z","shell.execute_reply":"2023-11-23T12:59:09.826944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import svm, neighbors, linear_model\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\n\n# Generate random data\nnp.random.seed(42)  # for reproducibility\n\n# Assuming each data point has 100 features\nnum_samples = 1000\nnum_features = 100\n\n# Generate random features\nX = np.random.rand(num_samples, num_features)\n\n# Generate random labels for demonstration purpose\ny = np.random.choice(['ch', 'chtms', 'chb', 'chbtms', 'cht', 'chi', 'chtmsi', 'c', 'chbi', 'chbt',\n                      'chih', 'chtmsih', 'chbtmsi', 'chtmsit', 'chtmsitm', 'cb', 'chitms', 'chbih',\n                      'ctms', 'chti', 'chtims', 'chib', 'chbitms', 'chbtmsitm', 'chbtmsih', 'chtih',\n                      'chtitms', 'chtit', 'ct'], size=num_samples)\n\n# Step 1: Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Step 2: Train classifiers\nsvm_classifier = svm.SVC()\nknn_classifier = neighbors.KNeighborsClassifier()\nlogistic_regression_classifier = linear_model.LogisticRegression()\n\n# Dictionary to store results\nresults_dict = {'Classifier': [], 'Accuracy': [], 'Precision': [], 'Recall': [], 'F1 Score': []}\n\n# Loop through classifiers\nclassifiers = {'SVM': svm_classifier, 'KNN': knn_classifier, 'Logistic Regression': logistic_regression_classifier}\n\nfor classifier_name, classifier in classifiers.items():\n    # Train the classifier\n    classifier.fit(X_train, y_train)\n\n    # Predictions\n    predictions = classifier.predict(X_test)\n\n    # Calculate metrics\n    accuracy = accuracy_score(y_test, predictions)\n    precision = precision_score(y_test, predictions, average='weighted')\n    recall = recall_score(y_test, predictions, average='weighted')\n    f1 = f1_score(y_test, predictions, average='weighted')\n\n    # Store results\n    results_dict['Classifier'].append(classifier_name)\n    results_dict['Accuracy'].append(accuracy)\n    results_dict['Precision'].append(precision)\n    results_dict['Recall'].append(recall)\n    results_dict['F1 Score'].append(f1)\n\n# Display results in a DataFrame\nresults_df = pd.DataFrame(results_dict)\nprint(results_df)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-23T13:09:29.775158Z","iopub.execute_input":"2023-11-23T13:09:29.775566Z","iopub.status.idle":"2023-11-23T13:09:30.529996Z","shell.execute_reply.started":"2023-11-23T13:09:29.775526Z","shell.execute_reply":"2023-11-23T13:09:30.528985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display results in a DataFrame\nresults_df = pd.DataFrame(results_dict)\nprint(results_df)\n\n# Plotting the bar chart\nfig, axes = plt.subplots(nrows=2, ncols=2, figsize=(12, 10))\naxes = axes.flatten()\n\nmetrics = ['Accuracy', 'Precision', 'Recall', 'F1 Score']\ncolors = ['blue', 'green', 'orange', 'red']\n\nfor i, metric in enumerate(metrics):\n    results_df.plot(kind='bar', x='Classifier', y=metric, ax=axes[i], color=colors[i], legend=False)\n    axes[i].set_title(f'{metric} by Classifier')\n    axes[i].set_ylabel(metric)\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-23T13:10:40.826279Z","iopub.execute_input":"2023-11-23T13:10:40.82674Z","iopub.status.idle":"2023-11-23T13:10:41.416192Z","shell.execute_reply.started":"2023-11-23T13:10:40.826697Z","shell.execute_reply":"2023-11-23T13:10:41.415101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.cluster import KMeans\nimport matplotlib.pyplot as plt\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\n\n# Assuming X is your feature matrix with shape (num_samples, num_features)\n\n# Generate random labels for demonstration purpose\ny = np.random.choice(['ch', 'chtms', 'chb', 'chbtms', 'cht', 'chi', 'chtmsi', 'c', 'chbi', 'chbt',\n                      'chih', 'chtmsih', 'chbtmsi', 'chtmsit', 'chtmsitm', 'cb', 'chitms', 'chbih',\n                      'ctms', 'chti', 'chtims', 'chib', 'chbitms', 'chbtmsitm', 'chbtmsih', 'chtih',\n                      'chtitms', 'chtit', 'ct'], size=num_samples)\n\n# Standardize the features\nX_standardized = StandardScaler().fit_transform(X)\n\n# Apply PCA for dimensionality reduction\npca = PCA(n_components=2)\nX_pca = pca.fit_transform(X_standardized)\n\n# Get unique chemical classes\nunique_classes = np.unique(y)\n\n# Create subplots for each chemical class\nfig, axes = plt.subplots(nrows=len(unique_classes), ncols=1, figsize=(8, 4 * len(unique_classes)))\n\nfor i, chemical_class in enumerate(unique_classes):\n    # Select samples for the current chemical class\n    class_indices = (y == chemical_class)\n    X_class = X_pca[class_indices]\n\n    # Apply k-means clustering\n    kmeans = KMeans(n_clusters=3, random_state=42)  # You can adjust the number of clusters\n    y_kmeans = kmeans.fit_predict(X_class)\n\n    # Plot the clustering results\n    axes[i].scatter(X_class[:, 0], X_class[:, 1], c=y_kmeans, cmap='viridis', label=chemical_class)\n    axes[i].scatter(kmeans.cluster_centers_[:, 0], kmeans.cluster_centers_[:, 1], marker='x', s=100, c='red', label='Cluster Centers')\n    axes[i].set_title(f'Clustering of {chemical_class}')\n    axes[i].set_xlabel('Principal Component 1')\n    axes[i].set_ylabel('Principal Component 2')\n    axes[i].legend()\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-23T13:14:27.673655Z","iopub.execute_input":"2023-11-23T13:14:27.674036Z","iopub.status.idle":"2023-11-23T13:14:34.24346Z","shell.execute_reply.started":"2023-11-23T13:14:27.674003Z","shell.execute_reply":"2023-11-23T13:14:34.242671Z"},"trusted":true},"execution_count":null,"outputs":[]}]}