{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Load Libraries**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nfrom tqdm.auto import tqdm\ntqdm.pandas()\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport Levenshtein\nimport cv2\nimport re\nimport torch\nimport sys\nfrom keras.applications.densenet import preprocess_input\nimport torch\n!pip install wordcloud\nfrom torch.utils.data import DataLoader, Dataset\nimport random\nfrom torchvision import transforms\nfrom PIL import Image\nfrom wordcloud import WordCloud\n!ls /kaggle/input/\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-19T19:36:18.196396Z","iopub.execute_input":"2023-10-19T19:36:18.196759Z","iopub.status.idle":"2023-10-19T19:36:37.841707Z","shell.execute_reply.started":"2023-10-19T19:36:18.196729Z","shell.execute_reply":"2023-10-19T19:36:37.840625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data Preparation**","metadata":{}},{"cell_type":"code","source":"#Reading the content\ntrain = pd.read_csv('/kaggle/input/bms-molecular-translation/train_labels.csv')\nprint(f'train.shape: {train.shape}')\ntrain.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-10-19T19:36:41.918143Z","iopub.execute_input":"2023-10-19T19:36:41.919204Z","iopub.status.idle":"2023-10-19T19:36:49.998818Z","shell.execute_reply.started":"2023-10-19T19:36:41.919158Z","shell.execute_reply":"2023-10-19T19:36:49.998009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = \"/kaggle/input/bms-molecular-translation/train/{}/{}/{}/{}.png\"\nget_image_path = lambda image_id_details :train_path.format(image_id_details[0], image_id_details[1], image_id_details[2], image_id_details) \ntrain['image_path']=train['image_id'].apply(get_image_path)\nprint(train.loc[100].InChI)\nprint(train.loc[100])","metadata":{"execution":{"iopub.status.busy":"2023-10-19T19:36:55.016473Z","iopub.execute_input":"2023-10-19T19:36:55.016798Z","iopub.status.idle":"2023-10-19T19:36:56.815392Z","shell.execute_reply.started":"2023-10-19T19:36:55.016773Z","shell.execute_reply":"2023-10-19T19:36:56.814536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Preparing INCHI Formulas for Extracting Features**","metadata":{}},{"cell_type":"code","source":"\n# Split the 'InChI' column into a list of components\ntrain['InChI_Columns'] = train['InChI'].progress_apply(lambda x: x.split('/'))\n\n# Calculate the length of the 'InChI_Columns' for each row\ntrain['InChI_Len'] = train['InChI_Columns'].progress_apply(len)\n\nInChI_new_df = train['InChI_Columns'].progress_apply(pd.Series)\n\n# Concatenate 'InChI_df' with the original 'train' DataFrame\ntrain = pd.concat([train, InChI_new_df.add_prefix('InChI_Column')], axis=1)\n\n# Extract additional components\ntrain['molecular_formula'] = train['InChI'].progress_apply(lambda x: x.split('/')[1])\ntrain['connectivity_layer'] = train['InChI'].progress_apply(lambda x: x.split('/')[2])\n\ndef extract_stereochemical_layer(InChI):\n    components = InChI.split('/')\n    if len(components) > 3:\n        return components[3]\n    else:\n        return 'No Stereochemical Layer'\n\ntrain['Stereochemical Layer'] = train['InChI'].progress_apply(extract_stereochemical_layer)\n\n# Created a new column 'InChI_New' by combining the extracted components\ntrain['InChI_New'] = train['molecular_formula'] + \"/\" + train['connectivity_layer'] + \"/\" + train['Stereochemical Layer']\n","metadata":{"execution":{"iopub.status.busy":"2023-10-19T19:37:02.117906Z","iopub.execute_input":"2023-10-19T19:37:02.118267Z","iopub.status.idle":"2023-10-19T19:43:36.046393Z","shell.execute_reply.started":"2023-10-19T19:37:02.118241Z","shell.execute_reply":"2023-10-19T19:43:36.045345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-10-11T16:18:36.853914Z","iopub.execute_input":"2023-10-11T16:18:36.854232Z","iopub.status.idle":"2023-10-11T16:18:36.875577Z","shell.execute_reply.started":"2023-10-11T16:18:36.854206Z","shell.execute_reply":"2023-10-11T16:18:36.874595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Check for Duplicate Rows**","metadata":{}},{"cell_type":"code","source":"# Convert list columns to tuples\ntrain['InChI_Columns'] = train['InChI_Columns'].apply(tuple)\n\n# Use the duplicated() method to check for duplicate rows based on all columns\nduplicates = train.duplicated(keep='first')\n\n# Filter the DataFrame to show only the duplicate rows\nduplicate_rows = train[duplicates]\n\n# Display the duplicate rows\nprint(\"Duplicate Rows:\")\nprint(duplicate_rows)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T16:18:55.254357Z","iopub.execute_input":"2023-10-11T16:18:55.254713Z","iopub.status.idle":"2023-10-11T16:19:13.162754Z","shell.execute_reply.started":"2023-10-11T16:18:55.254680Z","shell.execute_reply":"2023-10-11T16:19:13.161802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Check for Missing Values**","metadata":{}},{"cell_type":"code","source":"# Check for missing data\nmissing_data = train.isnull().sum()\nprint(missing_data)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T16:19:13.164502Z","iopub.execute_input":"2023-10-11T16:19:13.165401Z","iopub.status.idle":"2023-10-11T16:19:15.423170Z","shell.execute_reply.started":"2023-10-11T16:19:13.165368Z","shell.execute_reply":"2023-10-11T16:19:15.419707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Exploratory Data Analysis**","metadata":{}},{"cell_type":"code","source":"# Create a sample subset for faster exploration\nsample_df = train.sample(frac=1.0, random_state=42)\n\n# Function to perform EDA for a column\ndef perform_eda(column_name):\n    print(f\"Exploratory Data Analysis for '{column_name}':\")\n    \n    # Summary statistics\n    stats = sample_df[column_name].describe()\n    print(stats)\n    \n    # Check unique values and their counts\n    unique_values = sample_df[column_name].nunique()\n    print(f\"Number of unique values: {unique_values}\")\n    \n    if unique_values <= 20:\n        # Count plot for categorical columns with <= 20 unique values\n        plt.figure(figsize=(10, 5))\n        sns.countplot(data=sample_df, x=column_name, order=sample_df[column_name].value_counts().index)\n        plt.xticks(rotation=90)\n        plt.xlabel(column_name)\n        plt.ylabel('Count')\n        plt.title(f'Distribution of {column_name}')\n        plt.show()\n    else:\n        # Display the first few rows of the column for inspection (for text data)\n        print(f\"Sample values in '{column_name}':\\n\")\n        print(sample_df[column_name].head())\n    \n    print(\"\\n\" + \"=\" * 50 + \"\\n\")\n\n# Perform EDA for the specified columns\ncolumns_to_eda = ['molecular_formula', 'connectivity_layer', 'Stereochemical Layer'] + [f'InChI_Column{i}' for i in range(11)]\nfor column_name in columns_to_eda:\n    perform_eda(column_name)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T16:19:15.424689Z","iopub.execute_input":"2023-10-11T16:19:15.425037Z","iopub.status.idle":"2023-10-11T16:19:58.178212Z","shell.execute_reply.started":"2023-10-11T16:19:15.425005Z","shell.execute_reply":"2023-10-11T16:19:58.177177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Fill Inchi Missing values with Mode**","metadata":{}},{"cell_type":"code","source":"# Columns to fill NaN values with mode\ninchi_columns_to_fill = [f'InChI_Column{i}' for i in range(11)]\n\n# Initialize the concatenated InChI string\nconcatenated_inchi = ''\n\n# Iterate through the specified InChI columns\nfor column_name in inchi_columns_to_fill:\n    # Find the mode value for the current InChI column\n    mode_value = train[column_name].mode()[0]\n    \n    # Fill NaN values in the current InChI column with the mode value\n    train[column_name].fillna(mode_value, inplace=True)\n    \n    # Append the mode value to the concatenated InChI string\n    if mode_value != 'nan':\n        if not concatenated_inchi:\n            concatenated_inchi += mode_value\n        else:\n            concatenated_inchi += '/' + mode_value\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T16:19:58.180472Z","iopub.execute_input":"2023-10-11T16:19:58.181351Z","iopub.status.idle":"2023-10-11T16:20:04.662331Z","shell.execute_reply.started":"2023-10-11T16:19:58.181316Z","shell.execute_reply":"2023-10-11T16:20:04.661367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data = train.isnull().sum()\nprint(missing_data)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T16:20:04.663602Z","iopub.execute_input":"2023-10-11T16:20:04.664219Z","iopub.status.idle":"2023-10-11T16:20:06.857251Z","shell.execute_reply.started":"2023-10-11T16:20:04.664186Z","shell.execute_reply":"2023-10-11T16:20:06.856227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Visualizations using Graph, WordCloud and Heat Map**","metadata":{}},{"cell_type":"code","source":"from wordcloud import WordCloud\n\n# Function to perform advanced EDA for a column\ndef perform_advanced_eda(train, column_name):\n    print(f\"Advanced Exploratory Data Analysis for '{column_name}':\")\n    \n    if column_name == 'InChI':\n        # Special handling for 'InChI' column\n        if train[column_name].dtype == 'object':\n            # Sample word cloud for 'InChI' text column\n            plt.figure(figsize=(12, 6))\n            wordcloud = WordCloud(width=800, height=400, background_color='white').generate(' '.join(train[column_name]))\n            plt.imshow(wordcloud, interpolation='bilinear')\n            plt.axis('off')\n            plt.title(f'Word Cloud for {column_name}')\n            plt.show()\n    elif train[column_name].dtype == 'object':\n        # Sample word cloud for other text columns\n        plt.figure(figsize=(12, 6))\n        wordcloud = WordCloud(width=800, height=400, background_color='white').generate(' '.join(train[column_name]))\n        plt.imshow(wordcloud, interpolation='bilinear')\n        plt.axis('off')\n        plt.title(f'Word Cloud for {column_name}')\n        plt.show()\n    else:\n        # Numerical data distribution visualization\n        plt.figure(figsize=(12, 6))\n        sns.histplot(train[column_name], bins=20, kde=True)\n        plt.xlabel(column_name)\n        plt.ylabel('Frequency')\n        plt.title(f'Distribution of {column_name}')\n        plt.show()\n\n# Perform advanced EDA for the specified columns, including 'InChI'\ncolumns_to_eda = ['molecular_formula', 'connectivity_layer', 'Stereochemical Layer'] + [f'InChI_Column{i}' for i in range(11)]\nfor column_name in columns_to_eda:\n    perform_advanced_eda(train, column_name)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T16:20:06.858569Z","iopub.execute_input":"2023-10-11T16:20:06.859455Z","iopub.status.idle":"2023-10-11T16:24:03.818928Z","shell.execute_reply.started":"2023-10-11T16:20:06.859421Z","shell.execute_reply":"2023-10-11T16:24:03.817952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to calculate Jaccard similarity between two sets\ndef jaccard_similarity(set1, set2):\n    intersection = len(set1.intersection(set2))\n    union = len(set1) + len(set2) - intersection\n    return intersection / union\n\n# Create a similarity matrix\ncolumns_to_compare = [f'InChI_Column{i}' for i in range(11)] \nsimilarity_matrix = pd.DataFrame(index=columns_to_compare, columns=columns_to_compare)\n\nfor col1 in columns_to_compare:\n    for col2 in columns_to_compare:\n        similarity = jaccard_similarity(set(train[col1]), set(train[col2]))\n        similarity_matrix.loc[col1, col2] = similarity\n\n# Create a heatmap\nplt.figure(figsize=(10, 6))\nsns.heatmap(similarity_matrix.astype(float), annot=True, cmap=\"YlGnBu\")\nplt.title(\"Jaccard Similarity Heatmap\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T16:24:03.820355Z","iopub.execute_input":"2023-10-11T16:24:03.820886Z","iopub.status.idle":"2023-10-11T16:25:24.704543Z","shell.execute_reply.started":"2023-10-11T16:24:03.820852Z","shell.execute_reply":"2023-10-11T16:25:24.703661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"atom=['Cl', 'C','Br','B','Si','S' 'H', 'P', 'O', 'N', 'I', 'F']\natom_revised=['G', 'C','R','B','L','S' 'H', 'P', 'O', 'N', 'I', 'F']\n\n# Define a function to revise elements in 'molecular_formula'\ndef revise_formula_element(formula):\n    revised_formula = formula\n    if \"Cl\" in formula:\n        revised_formula = revised_formula.replace(\"Cl\", \"W\")\n    if \"Br\" in formula:\n        revised_formula = revised_formula.replace(\"Br\", \"R\")\n    if \"Si\" in formula:\n        revised_formula = revised_formula.replace(\"Si\", \"L\")\n    return revised_formula\n\n# Apply the 'revise_formula_element' function to the 'molecular_formula' column\ntrain['molecular_formula'] = train['molecular_formula'].apply(revise_formula_element)\n ","metadata":{"execution":{"iopub.status.busy":"2023-10-11T16:42:30.497373Z","iopub.execute_input":"2023-10-11T16:42:30.498021Z","iopub.status.idle":"2023-10-11T16:42:31.193347Z","shell.execute_reply.started":"2023-10-11T16:42:30.497990Z","shell.execute_reply":"2023-10-11T16:42:31.192398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Define a function to extract the number of carbon atoms from 'molecular_formula'\ndef extract_carbon_count(formula):\n    nums = {\"0\", \"1\", \"2\", \"3\", \"4\", \"5\", \"6\", \"7\", \"8\", \"9\"}\n    result = \"\"\n    for i in range(1, len(formula)):\n        if formula[i] in nums:\n            result += formula[i]\n        else:\n            if result:\n                return int(result) \n            else:\n                return 0  # Return 0 if there are no digits\n    return int(result)  # Return the result if digits are found\n\n# Apply the 'extract_carbon_count' function to create a new 'C' column\ntrain['Total Carbon'] = train['molecular_formula'].apply(extract_carbon_count)\n\n# Display the modified DataFrame\ntrain.head()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T16:42:34.916299Z","iopub.execute_input":"2023-10-11T16:42:34.917307Z","iopub.status.idle":"2023-10-11T16:42:37.235875Z","shell.execute_reply.started":"2023-10-11T16:42:34.917264Z","shell.execute_reply":"2023-10-11T16:42:37.234889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Frequency of Counts of Total Carbon**","metadata":{}},{"cell_type":"code","source":"carbon_counts = train['Total Carbon']\n\n# Count the frequency of each unique carbon count\ncarbon_count_freq = carbon_counts.value_counts().sort_index()\n\n# Create a bar plot\nplt.figure(figsize=(10, 6))\nplt.bar(carbon_count_freq.index, carbon_count_freq.values, color='blue')\nplt.xlabel('Total Carbon Count')\nplt.ylabel('Frequency')\nplt.title('Distribution of Total Carbon Counts')\nplt.xticks(rotation=45)\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T16:42:41.845735Z","iopub.execute_input":"2023-10-11T16:42:41.846728Z","iopub.status.idle":"2023-10-11T16:42:42.213554Z","shell.execute_reply.started":"2023-10-11T16:42:41.846678Z","shell.execute_reply":"2023-10-11T16:42:42.212653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Deep Analysis of Image Resolution**","metadata":{}},{"cell_type":"code","source":"\ndef analyze_image_shapes_and_aspect_ratios(image_paths, limit=3000):\n    h_shapes = []\n    w_shapes = []\n    aspect_ratios = []\n\n    for idx, image_path in enumerate(image_paths[:limit]):\n        image = cv2.imread(image_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        h_shapes.append(image.shape[0])\n        w_shapes.append(image.shape[1])\n        aspect_ratios.append(1.0 * (image.shape[1] / image.shape[0]))\n\n    plt.figure(figsize=(12, 12))\n    plt.subplots_adjust(top=0.5, bottom=0.01, hspace=1, wspace=0.4)\n\n    plt.subplot(2, 2, 1)\n    plt.hist(np.array(h_shapes) * np.array(w_shapes), bins=50)\n    plt.xticks(rotation=45)\n    plt.title(\"Area Image Distribution\", fontsize=14)\n\n    plt.subplot(2, 2, 2)\n    plt.hist(h_shapes, bins=50)\n    plt.title(\"Height Image Distribution\", fontsize=14)\n\n    plt.subplot(2, 2, 3)\n    plt.hist(w_shapes, bins=50)\n    plt.title(\"Width Image Distribution\", fontsize=14)\n\n    plt.subplot(2, 2, 4)\n    plt.hist(aspect_ratios, bins=50)\n    plt.title(\"Aspect Ratio Distribution\", fontsize=14)\n\n    plt.show()\n\n\nimage_paths = train['image_path'].values\nanalyze_image_shapes_and_aspect_ratios(image_paths, limit=3000)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T16:47:22.029854Z","iopub.execute_input":"2023-10-11T16:47:22.030236Z","iopub.status.idle":"2023-10-11T16:47:51.075666Z","shell.execute_reply.started":"2023-10-11T16:47:22.030208Z","shell.execute_reply":"2023-10-11T16:47:51.074779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****","metadata":{}},{"cell_type":"markdown","source":"**Denoise and Preprocessing**","metadata":{}},{"cell_type":"code","source":"\ndef remove_dots_and_noise(image):\n    # Threshold the grayscale image to create a binary mask\n    _, binary_mask = cv2.threshold(image, 127, 255, cv2.THRESH_BINARY_INV)\n\n    # Find connected components and filter small dotted regions\n    nlabels, labels, stats, centroids = cv2.connectedComponentsWithStats(binary_mask, None, None, None, 8, cv2.CV_32S)\n\n    sizes = stats[1:, -1]  # Get CC_STAT_AREA component\n    img2 = np.zeros_like(labels, np.uint8)\n\n    for i in range(0, nlabels - 1):\n        if sizes[i] >= 2:   # Filter small dotted regions\n            img2[labels == i + 1] = 255\n\n    # Invert the binary mask\n    result_image = cv2.bitwise_not(img2)\n\n    return result_image\n\ndef load_and_preprocess_image(path):\n    image = Image.open(path)\n    image = np.array(image)\n\n    if len(image.shape) == 3:\n        # Convert to grayscale if the image has 3 channels\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n\n    preprocessed_image = remove_dots_and_noise(image)\n\n    return preprocessed_image\n\n# Limit the number of images to process (e.g., to the first 1000)\nlimit = 1000  # Process 1000 images\ntrain_subset = train[:limit].copy()\ntrain_subset['X'] = train_subset['image_path'].apply(load_and_preprocess_image)\n\n# Randomly select and display 10 images from the 1000 processed images\nsample_indices = random.sample(range(limit), 10)\nfor i in sample_indices:\n    sample_image = train_subset['X'].iloc[i]\n\n    # Load the original image separately for comparison\n    original_image = cv2.imread(train['image_path'].iloc[i])\n\n    plt.figure(figsize=(18, 6))\n    plt.subplot(1, 3, 1)\n    plt.imshow(original_image, cmap='gray')\n    plt.title('Original Image', fontsize=14)\n    plt.axis('off')\n\n    plt.subplot(1, 3, 2)\n    plt.imshow(sample_image, cmap='gray')\n    plt.title('Preprocessed Image', fontsize=14)\n    plt.axis('off')\n\n    denoised_image = remove_dots_and_noise(sample_image)\n    plt.subplot(1, 3, 3)\n    plt.imshow(denoised_image, cmap='gray')\n    plt.title('Denoised Image', fontsize=14)\n    plt.axis('off')\n\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T16:50:09.604384Z","iopub.execute_input":"2023-10-11T16:50:09.605204Z","iopub.status.idle":"2023-10-11T16:50:18.583248Z","shell.execute_reply.started":"2023-10-11T16:50:09.605164Z","shell.execute_reply":"2023-10-11T16:50:18.582370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  image path to test\nimage_path = '/kaggle/input/bms-molecular-translation/test/0/0/0/00000d2a601c.png'\n\n# Load and preprocess the image\npreprocessed_image = load_and_preprocess_image(image_path)\n\n# Display the original and preprocessed images\noriginal_image = cv2.imread(image_path)\n\nplt.figure(figsize=(12, 6))\nplt.subplot(1, 2, 1)\nplt.imshow(original_image, cmap='gray')\nplt.title('Original Image', fontsize=14)\nplt.axis('off')\n\nplt.subplot(1, 2, 2)\nplt.imshow(preprocessed_image, cmap='gray')\nplt.title('Preprocessed Image', fontsize=14)\nplt.axis('off')\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T16:50:23.540427Z","iopub.execute_input":"2023-10-11T16:50:23.541471Z","iopub.status.idle":"2023-10-11T16:50:23.795957Z","shell.execute_reply.started":"2023-10-11T16:50:23.541431Z","shell.execute_reply":"2023-10-11T16:50:23.795082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = np.array(train_subset.X.values.tolist())\nX.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-11T16:50:28.924043Z","iopub.execute_input":"2023-10-11T16:50:28.925181Z","iopub.status.idle":"2023-10-11T16:50:28.936372Z","shell.execute_reply.started":"2023-10-11T16:50:28.925141Z","shell.execute_reply":"2023-10-11T16:50:28.935585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Model Building**","metadata":{}},{"cell_type":"markdown","source":"**Extracting features using DenseNet121**","metadata":{}},{"cell_type":"code","source":"from keras.applications.densenet import DenseNet121, preprocess_input\nfrom keras.models import Model\n\ndef get_features(train):\n    # Load the DenseNet121 model\n    model = DenseNet121()\n    \n    # Re-structure the model to remove the final classification layer\n    model = Model(inputs=model.inputs, outputs=model.layers[-2].output)\n    \n    # Create an empty dictionary to store features\n    features = dict()\n    \n    # Iterate through image paths in the DataFrame\n    for idx, name in enumerate(train_subset['image_path'].values[:1000]):\n        filename = name\n        \n        try:\n            # Load and preprocess the image using PIL\n            image = Image.open(filename)\n            image = image.resize((224, 224))\n            image = np.array(image)\n            \n            # Check if the image is grayscale (single channel), and if so, convert it to RGB\n            if len(image.shape) == 2:\n                image = np.stack((image,) * 3, axis=-1)\n            \n            image = image.reshape((1, image.shape[0], image.shape[1], image.shape[2]))\n            image = preprocess_input(image)\n            \n            # Extract features from the image using the model\n            feature = model.predict(image, verbose=0)\n            \n            # Store the feature in the dictionary with the corresponding image_id\n            features[train['image_id'][idx]] = feature\n        except Exception as e:\n            print(f\"Error processing image {filename}: {e}\") \n        \n    return features\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T07:07:42.467178Z","iopub.execute_input":"2023-10-11T07:07:42.467613Z","iopub.status.idle":"2023-10-11T07:07:42.477881Z","shell.execute_reply.started":"2023-10-11T07:07:42.467581Z","shell.execute_reply":"2023-10-11T07:07:42.476462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features =get_features(train)\nprint('Extracted Features: %d' % len(features))","metadata":{"execution":{"iopub.status.busy":"2023-10-11T07:07:48.415208Z","iopub.execute_input":"2023-10-11T07:07:48.415753Z","iopub.status.idle":"2023-10-11T07:09:29.147807Z","shell.execute_reply.started":"2023-10-11T07:07:48.415658Z","shell.execute_reply":"2023-10-11T07:09:29.146731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_text_mappings(train, num_samples=1000):\n    text_mapping = {}\n    for idx, text in enumerate(train['InChI'].values[:num_samples]):\n        image_id = train['image_id'][idx]\n        text_mapping[image_id] = text\n    return text_mapping\n\ndef create_vocabulary(text_mappings):\n    vocabulary = set(text_mappings.values())\n    return vocabulary\n\n# Load text mappings from the DataFrame\ntext_mappings = load_text_mappings(train, num_samples=1000)\n\n# Create a vocabulary from the loaded text mappings\nvocabulary = create_vocabulary(text_mappings)\n\n# Print the number of loaded text mappings and vocabulary size\nprint(f'Loaded: {len(text_mappings)} text mappings')\nprint(f'Vocabulary Size: {len(vocabulary)}')\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T07:09:38.308889Z","iopub.execute_input":"2023-10-11T07:09:38.309265Z","iopub.status.idle":"2023-10-11T07:09:38.322155Z","shell.execute_reply.started":"2023-10-11T07:09:38.309237Z","shell.execute_reply":"2023-10-11T07:09:38.321088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**** Calculate Mean Levenshtein distance****","metadata":{}},{"cell_type":"code","source":"def calculate_score(y_true, y_pred):\n    scores = []\n    for true, pred in zip(y_true, y_pred):\n        score = Levenshtein.distance(true, pred)\n        scores.append(score)\n    mean_score = np.mean(scores)\n    return mean_score","metadata":{"execution":{"iopub.status.busy":"2023-10-11T17:22:31.167661Z","iopub.execute_input":"2023-10-11T17:22:31.168082Z","iopub.status.idle":"2023-10-11T17:22:31.173517Z","shell.execute_reply.started":"2023-10-11T17:22:31.168055Z","shell.execute_reply":"2023-10-11T17:22:31.172358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = train['InChI'].values\ny_pred = [concatenated_inchi]  * len(train)\nscore = calculate_score(y_true, y_pred)\nprint(score)","metadata":{"execution":{"iopub.status.busy":"2023-10-11T17:22:34.182362Z","iopub.execute_input":"2023-10-11T17:22:34.183075Z","iopub.status.idle":"2023-10-11T17:22:39.748392Z","shell.execute_reply.started":"2023-10-11T17:22:34.183043Z","shell.execute_reply":"2023-10-11T17:22:39.747379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\n\nvocabulary = {word: index for index, word in enumerate(text_mappings.values())}\n\n# Step 1: Tokenize the InChI strings\ntokenizer = Tokenizer(num_words=len(vocabulary), oov_token='<OOV>')  # Initialize the tokenizer\ntokenizer.fit_on_texts(text_mappings.values())  # Fit the tokenizer on your text data\ntokenized_sequences = tokenizer.texts_to_sequences(text_mappings.values())  # Tokenize the text data\n\n# Step 2: Map tokens to integer IDs using the vocabulary\ninteger_sequences = [[vocabulary.get(token, 0) for token in sequence] for sequence in tokenized_sequences]\n\n# Step 3: Determine the maximum sequence length\nmax_sequence_length = 100  # Define your desired maximum sequence length\n\n# Step 4: Pad or truncate the sequences to the maximum sequence length\npadded_sequences = pad_sequences(integer_sequences, maxlen=max_sequence_length, padding='post', truncating='post')\n\n# Now, 'padded_sequences' contains your tokenized sequences, padded or truncated to the maximum sequence length.\n\n# Step 5: One-hot encode the tokenized sequences as your target data (y_train)\nnum_samples = len(padded_sequences)\nvocab_size = len(vocabulary)\n\ny_train = np.zeros((num_samples, max_sequence_length, vocab_size))\nfor i, sequence in enumerate(padded_sequences):\n    for t, token_id in enumerate(sequence):\n        if token_id != 0:  # Skip padding tokens (assuming 0 is the padding token)\n            y_train[i, t, token_id] = 1  # Set the corresponding entry to 1\n\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T07:09:53.451565Z","iopub.execute_input":"2023-10-11T07:09:53.452815Z","iopub.status.idle":"2023-10-11T07:09:53.883882Z","shell.execute_reply.started":"2023-10-11T07:09:53.452752Z","shell.execute_reply":"2023-10-11T07:09:53.882795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = np.array([features[image_id] for image_id in train['image_id'].values[:num_samples]])\nprint(\"Shape of X_train:\", X_train.shape)\nprint(\"Shape of y_train:\",y_train.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-11T07:18:36.570167Z","iopub.execute_input":"2023-10-11T07:18:36.570612Z","iopub.status.idle":"2023-10-11T07:18:36.581102Z","shell.execute_reply.started":"2023-10-11T07:18:36.570577Z","shell.execute_reply":"2023-10-11T07:18:36.579769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Spliting Dataset**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n# Split the data into training, validation, and test sets\nX_train, X_temp, y_train, y_temp = train_test_split(X_train, y_train[:52], test_size=0.2, random_state=42)\nX_val, X_test, y_val, y_test = train_test_split(X_temp, y_temp, test_size=0.5, random_state=42)\n\n\n# Reshape the data to the desired number of time steps\nX_train = X_train[:, :seq_length, :]\nX_val = X_val[:, :seq_length, :]\nX_test = X_test[:, :seq_length, :]\ny_train = y_train[:, :seq_length, :]\ny_val = y_val[:, :seq_length, :]\ny_test = y_test[:, :seq_length, :]\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T07:40:41.146892Z","iopub.execute_input":"2023-10-11T07:40:41.147255Z","iopub.status.idle":"2023-10-11T07:40:41.177239Z","shell.execute_reply.started":"2023-10-11T07:40:41.147228Z","shell.execute_reply":"2023-10-11T07:40:41.176216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import LSTM, Dense\n\nnum_samples = min(X_train.shape[0], y_train.shape[0])\nX_train = X_train[:num_samples]\ny_train = y_train[:num_samples]\n\n# Compile the model with the MSE loss function\nmodel = Sequential()\nmodel.add(LSTM(256, return_sequences=False))\nmodel.add(Dense(1))\nmodel.compile(loss='mse', optimizer='adam')\n\n# Train the model\nmodel.fit(X_train, y_train[:, :, 0], epochs=50, batch_size=32)\nmodel.summary()\n\n# Make predictions on your test data to generate y_pred\ny_pred = model.predict(X_test)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T07:40:51.990187Z","iopub.execute_input":"2023-10-11T07:40:51.990559Z","iopub.status.idle":"2023-10-11T07:40:57.263376Z","shell.execute_reply.started":"2023-10-11T07:40:51.990530Z","shell.execute_reply":"2023-10-11T07:40:57.262279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_samples_to_visualize = 5 \nfor i in range(num_samples_to_visualize):\n    true_values = y_test[i, :, 0]\n    predicted_values = y_pred[i]\n    plt.plot(true_values, label=f'Sample {i + 1} - True', marker='o')\n    plt.plot(predicted_values, label=f'Sample {i + 1} - Predicted', linestyle='--', marker='x')\n\nplt.xlabel('Time Step')\nplt.ylabel('Value')\nplt.title('LSTM Model Performance for Multiple Samples')\nplt.legend()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T07:44:38.520471Z","iopub.execute_input":"2023-10-11T07:44:38.520880Z","iopub.status.idle":"2023-10-11T07:44:38.939909Z","shell.execute_reply.started":"2023-10-11T07:44:38.520849Z","shell.execute_reply":"2023-10-11T07:44:38.938898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"error = y_test[:, :, 0] - y_pred\nplt.figure(figsize=(10, 6))\nplt.hist(error, bins=30, density=True, alpha=0.7, label='Error Distribution')\nplt.xlabel('Error')\nplt.ylabel('Density')\nplt.title('Prediction Error Distribution')\nplt.legend()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T07:45:54.395551Z","iopub.execute_input":"2023-10-11T07:45:54.396735Z","iopub.status.idle":"2023-10-11T07:46:00.328968Z","shell.execute_reply.started":"2023-10-11T07:45:54.396664Z","shell.execute_reply":"2023-10-11T07:46:00.327916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Mean Score LV**","metadata":{}},{"cell_type":"code","source":"y_true = train['InChI'].values\nscore = calculate_score(y_true, y_pred)\nprint(score)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-11T07:41:46.436502Z","iopub.execute_input":"2023-10-11T07:41:46.436886Z","iopub.status.idle":"2023-10-11T07:41:46.450846Z","shell.execute_reply.started":"2023-10-11T07:41:46.436856Z","shell.execute_reply":"2023-10-11T07:41:46.449559Z"},"trusted":true},"execution_count":null,"outputs":[]}]}