{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-02T06:20:15.747891Z","iopub.execute_input":"2024-07-02T06:20:15.748336Z","iopub.status.idle":"2024-07-02T06:20:15.756454Z","shell.execute_reply.started":"2024-07-02T06:20:15.748303Z","shell.execute_reply":"2024-07-02T06:20:15.754956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:20:15.758723Z","iopub.execute_input":"2024-07-02T06:20:15.759211Z","iopub.status.idle":"2024-07-02T06:20:15.768324Z","shell.execute_reply.started":"2024-07-02T06:20:15.759167Z","shell.execute_reply":"2024-07-02T06:20:15.767181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install rdkit\n'''\nRDKit, an open-source cheminformatics tool, is used for generating ECFP features.\nIt facilitates the creation of hashed bit vectors, streamlining the process\n'''","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:20:15.769939Z","iopub.execute_input":"2024-07-02T06:20:15.770627Z","iopub.status.idle":"2024-07-02T06:20:28.315334Z","shell.execute_reply.started":"2024-07-02T06:20:15.770585Z","shell.execute_reply":"2024-07-02T06:20:28.313822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install duckdb\n\n'''\nWe have a large training set, and we can consider the parquet files as\ndatabases by using duckdb. We'll use this approach to create a smaller\ndataset for demonstration purposes.\n'''","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:20:28.318133Z","iopub.execute_input":"2024-07-02T06:20:28.318510Z","iopub.status.idle":"2024-07-02T06:20:40.637454Z","shell.execute_reply.started":"2024-07-02T06:20:28.318476Z","shell.execute_reply":"2024-07-02T06:20:40.635841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nWe utilize duckdb to scan through the large training sets. To begin,\nwe'll sample an equal number of positive and negative samples. \nThe query selects 3 lakh samples where \"binds\" equals both 0 and 1.\n'''","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:20:40.647064Z","iopub.execute_input":"2024-07-02T06:20:40.647528Z","iopub.status.idle":"2024-07-02T06:20:40.655050Z","shell.execute_reply.started":"2024-07-02T06:20:40.647482Z","shell.execute_reply":"2024-07-02T06:20:40.653954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import duckdb\nimport pandas as pd\n\ntrain_path = '/kaggle/input/leash-BELKA/train.parquet'\ntest_path = '/kaggle/input/leash-BELKA/test.parquet'","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:20:40.659379Z","iopub.execute_input":"2024-07-02T06:20:40.659840Z","iopub.status.idle":"2024-07-02T06:20:40.667560Z","shell.execute_reply.started":"2024-07-02T06:20:40.659801Z","shell.execute_reply":"2024-07-02T06:20:40.666183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"con = duckdb.connect()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:20:40.669239Z","iopub.execute_input":"2024-07-02T06:20:40.669732Z","iopub.status.idle":"2024-07-02T06:20:40.694244Z","shell.execute_reply.started":"2024-07-02T06:20:40.669694Z","shell.execute_reply":"2024-07-02T06:20:40.692884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = con.query(f\"\"\"(SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 0\n                        ORDER BY random()\n                        LIMIT 30000)\n                        UNION ALL\n                        (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 1\n                        ORDER BY random()\n                        LIMIT 30000)\"\"\").df()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:20:40.696041Z","iopub.execute_input":"2024-07-02T06:20:40.696516Z","iopub.status.idle":"2024-07-02T06:21:30.850824Z","shell.execute_reply.started":"2024-07-02T06:20:40.696477Z","shell.execute_reply":"2024-07-02T06:21:30.849139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"con.close()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:30.853039Z","iopub.execute_input":"2024-07-02T06:21:30.853599Z","iopub.status.idle":"2024-07-02T06:21:30.870362Z","shell.execute_reply.started":"2024-07-02T06:21:30.853553Z","shell.execute_reply":"2024-07-02T06:21:30.868763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:30.872383Z","iopub.execute_input":"2024-07-02T06:21:30.872757Z","iopub.status.idle":"2024-07-02T06:21:30.886869Z","shell.execute_reply.started":"2024-07-02T06:21:30.872725Z","shell.execute_reply":"2024-07-02T06:21:30.885761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:30.888364Z","iopub.execute_input":"2024-07-02T06:21:30.888845Z","iopub.status.idle":"2024-07-02T06:21:30.909466Z","shell.execute_reply.started":"2024-07-02T06:21:30.888806Z","shell.execute_reply":"2024-07-02T06:21:30.908398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:30.911147Z","iopub.execute_input":"2024-07-02T06:21:30.911576Z","iopub.status.idle":"2024-07-02T06:21:30.934747Z","shell.execute_reply.started":"2024-07-02T06:21:30.911537Z","shell.execute_reply":"2024-07-02T06:21:30.933398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:30.936203Z","iopub.execute_input":"2024-07-02T06:21:30.936617Z","iopub.status.idle":"2024-07-02T06:21:30.946742Z","shell.execute_reply.started":"2024-07-02T06:21:30.936584Z","shell.execute_reply":"2024-07-02T06:21:30.945518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### ","metadata":{}},{"cell_type":"code","source":"''' \nPRINTS THE TOP 3 LONGEST COMPOUNDS AND THE LAST 3 SMALLEST COMPOUNDS\n'''\n\ndef print_top_and_bottom_lengths(df, column_name):\n    if column_name not in df.columns:\n        print(f\"Column '{column_name}' does not exist in the DataFrame.\")\n        return\n\n    df['length'] = df[column_name].apply(len)\n    sorted_df = df.sort_values(by='length', ascending=False)\n    top_5 = sorted_df.head(5)\n    bottom_5 = sorted_df.tail(5)\n\n    print(\"Top 5 elements with the highest length:\")\n    print(top_5[[column_name, 'length']])\n\n    print(\"\\nLast 5 elements with the smallest length:\")\n    print(bottom_5[[column_name, 'length']])","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:30.948388Z","iopub.execute_input":"2024-07-02T06:21:30.948829Z","iopub.status.idle":"2024-07-02T06:21:30.960079Z","shell.execute_reply.started":"2024-07-02T06:21:30.948789Z","shell.execute_reply":"2024-07-02T06:21:30.958877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nFUNCTION THAT PLOTS THE TOP 3 HIGHEST AND THE LAST 3 LOWEST FREQUENCY \nCOMPOUNDS \n'''\n\ndef plot_top_and_bottom_values(df, col_name):\n    value_counts = df[col_name].value_counts()\n    top_3 = value_counts.head(3)\n    last_3 = value_counts.tail(3)\n    \n    combined = pd.concat([top_3, last_3]).reset_index()\n    combined.columns = [col_name, 'counts']\n\n    combined[col_name] = combined[col_name].str.upper()\n\n    plt.figure(figsize=(14, 7))\n    sns.barplot(x=col_name, y='counts', data=combined, palette='viridis')\n\n    plt.title(f'Value Counts of {col_name.upper()} (Top 3 and Last 3)')\n    plt.xlabel(f'{col_name.upper()}')\n    plt.ylabel('COUNTS')\n\n    plt.xticks(rotation=45)\n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:30.961828Z","iopub.execute_input":"2024-07-02T06:21:30.962333Z","iopub.status.idle":"2024-07-02T06:21:30.973324Z","shell.execute_reply.started":"2024-07-02T06:21:30.962293Z","shell.execute_reply":"2024-07-02T06:21:30.971873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:30.974784Z","iopub.execute_input":"2024-07-02T06:21:30.975157Z","iopub.status.idle":"2024-07-02T06:21:30.994225Z","shell.execute_reply.started":"2024-07-02T06:21:30.975119Z","shell.execute_reply":"2024-07-02T06:21:30.992801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### buildingblock1_smiles","metadata":{}},{"cell_type":"code","source":"from rdkit import Chem\nfrom rdkit.Chem import AllChem\nfrom rdkit.Chem import Draw\nfrom PIL import Image\nfrom IPython.display import display\n\nsamples = df[\"buildingblock1_smiles\"].sample(7)\nfor sample in samples:\n    molecule_object = Chem.MolFromSmiles(sample)\n    img = Draw.MolToImage(molecule_object)\n    print(\"Sample:\",sample)\n    print()\n    print(\"Molecule Object:\",molecule_object)\n    print()\n    display(img)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:30.995829Z","iopub.execute_input":"2024-07-02T06:21:30.996237Z","iopub.status.idle":"2024-07-02T06:21:31.153736Z","shell.execute_reply.started":"2024-07-02T06:21:30.996200Z","shell.execute_reply":"2024-07-02T06:21:31.152491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"buildingblock1_smiles\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:31.159732Z","iopub.execute_input":"2024-07-02T06:21:31.160151Z","iopub.status.idle":"2024-07-02T06:21:31.171688Z","shell.execute_reply.started":"2024-07-02T06:21:31.160097Z","shell.execute_reply":"2024-07-02T06:21:31.170539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nplot_top_and_bottom_values(df, \"buildingblock1_smiles\")","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:31.173285Z","iopub.execute_input":"2024-07-02T06:21:31.173814Z","iopub.status.idle":"2024-07-02T06:21:31.627719Z","shell.execute_reply.started":"2024-07-02T06:21:31.173763Z","shell.execute_reply":"2024-07-02T06:21:31.626516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print_top_and_bottom_lengths(df , \"buildingblock1_smiles\")","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:31.629171Z","iopub.execute_input":"2024-07-02T06:21:31.629542Z","iopub.status.idle":"2024-07-02T06:21:31.644947Z","shell.execute_reply.started":"2024-07-02T06:21:31.629512Z","shell.execute_reply":"2024-07-02T06:21:31.643665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### buildingblock2_smiles","metadata":{}},{"cell_type":"code","source":"samples = df[\"buildingblock2_smiles\"].sample(7)\nfor sample in samples:\n    molecule_object = Chem.MolFromSmiles(sample)\n    img = Draw.MolToImage(molecule_object)\n    print(\"Sample:\",sample)\n    print()\n    print(\"Molecule Object:\",molecule_object)\n    print()\n    display(img)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:31.646588Z","iopub.execute_input":"2024-07-02T06:21:31.646930Z","iopub.status.idle":"2024-07-02T06:21:31.791411Z","shell.execute_reply.started":"2024-07-02T06:21:31.646900Z","shell.execute_reply":"2024-07-02T06:21:31.790130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"buildingblock2_smiles\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:31.793051Z","iopub.execute_input":"2024-07-02T06:21:31.793423Z","iopub.status.idle":"2024-07-02T06:21:31.805456Z","shell.execute_reply.started":"2024-07-02T06:21:31.793393Z","shell.execute_reply":"2024-07-02T06:21:31.804323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_top_and_bottom_values(df, \"buildingblock2_smiles\")","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:31.806940Z","iopub.execute_input":"2024-07-02T06:21:31.807850Z","iopub.status.idle":"2024-07-02T06:21:32.158356Z","shell.execute_reply.started":"2024-07-02T06:21:31.807809Z","shell.execute_reply":"2024-07-02T06:21:32.157312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print_top_and_bottom_lengths(df , \"buildingblock2_smiles\")","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:32.159566Z","iopub.execute_input":"2024-07-02T06:21:32.159900Z","iopub.status.idle":"2024-07-02T06:21:32.174151Z","shell.execute_reply.started":"2024-07-02T06:21:32.159872Z","shell.execute_reply":"2024-07-02T06:21:32.172616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### buildingblock3_smiles","metadata":{}},{"cell_type":"code","source":"samples = df[\"buildingblock3_smiles\"].sample(7)\nfor sample in samples:\n    molecule_object = Chem.MolFromSmiles(sample)\n    img = Draw.MolToImage(molecule_object)\n    print(\"Sample:\",sample)\n    print()\n    print(\"Molecule Object:\",molecule_object)\n    print()\n    display(img)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:32.175506Z","iopub.execute_input":"2024-07-02T06:21:32.175852Z","iopub.status.idle":"2024-07-02T06:21:32.318278Z","shell.execute_reply.started":"2024-07-02T06:21:32.175815Z","shell.execute_reply":"2024-07-02T06:21:32.317151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"buildingblock3_smiles\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:32.319976Z","iopub.execute_input":"2024-07-02T06:21:32.320346Z","iopub.status.idle":"2024-07-02T06:21:32.330439Z","shell.execute_reply.started":"2024-07-02T06:21:32.320315Z","shell.execute_reply":"2024-07-02T06:21:32.329008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_top_and_bottom_values(df, \"buildingblock3_smiles\")","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:32.332193Z","iopub.execute_input":"2024-07-02T06:21:32.332600Z","iopub.status.idle":"2024-07-02T06:21:32.680342Z","shell.execute_reply.started":"2024-07-02T06:21:32.332545Z","shell.execute_reply":"2024-07-02T06:21:32.678802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print_top_and_bottom_lengths(df , \"buildingblock3_smiles\")","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:32.681918Z","iopub.execute_input":"2024-07-02T06:21:32.682423Z","iopub.status.idle":"2024-07-02T06:21:32.698367Z","shell.execute_reply.started":"2024-07-02T06:21:32.682382Z","shell.execute_reply":"2024-07-02T06:21:32.696959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:32.700076Z","iopub.execute_input":"2024-07-02T06:21:32.700575Z","iopub.status.idle":"2024-07-02T06:21:32.719376Z","shell.execute_reply.started":"2024-07-02T06:21:32.700536Z","shell.execute_reply":"2024-07-02T06:21:32.718034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### molecule_smiles","metadata":{}},{"cell_type":"code","source":"samples = df[\"molecule_smiles\"].sample(7)\nfor sample in samples:\n    molecule_object = Chem.MolFromSmiles(sample)\n    img = Draw.MolToImage(molecule_object)\n    print(\"Sample:\",sample)\n    print()\n    print(\"Molecule Object:\",molecule_object)\n    print()\n    display(img)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:32.720780Z","iopub.execute_input":"2024-07-02T06:21:32.721254Z","iopub.status.idle":"2024-07-02T06:21:32.890888Z","shell.execute_reply.started":"2024-07-02T06:21:32.721215Z","shell.execute_reply":"2024-07-02T06:21:32.889778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"molecule_smiles\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:32.892261Z","iopub.execute_input":"2024-07-02T06:21:32.892597Z","iopub.status.idle":"2024-07-02T06:21:32.904486Z","shell.execute_reply.started":"2024-07-02T06:21:32.892568Z","shell.execute_reply":"2024-07-02T06:21:32.903219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_top_and_bottom_values(df, \"molecule_smiles\")","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:32.905756Z","iopub.execute_input":"2024-07-02T06:21:32.906260Z","iopub.status.idle":"2024-07-02T06:21:33.452431Z","shell.execute_reply.started":"2024-07-02T06:21:32.906218Z","shell.execute_reply":"2024-07-02T06:21:33.451275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print_top_and_bottom_lengths(df , \"molecule_smiles\")","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:33.453786Z","iopub.execute_input":"2024-07-02T06:21:33.454159Z","iopub.status.idle":"2024-07-02T06:21:33.467509Z","shell.execute_reply.started":"2024-07-02T06:21:33.454122Z","shell.execute_reply":"2024-07-02T06:21:33.465985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### protein_names","metadata":{}},{"cell_type":"code","source":"df[\"protein_name\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:33.469021Z","iopub.execute_input":"2024-07-02T06:21:33.469413Z","iopub.status.idle":"2024-07-02T06:21:33.479607Z","shell.execute_reply.started":"2024-07-02T06:21:33.469381Z","shell.execute_reply":"2024-07-02T06:21:33.478228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.rcParams.update({'font.size': 14, 'font.weight': 'bold'})\nprotein_counts = df[\"protein_name\"].value_counts()\nplt.figure(figsize=(10, 8))\nprotein_counts.plot(kind=\"pie\", autopct=\"%1.1f%%\", startangle=90)\nplt.axis(\"equal\")\nplt.title(\"Protein Name Distribution\")\nplt.ylabel(\"\")  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:33.481227Z","iopub.execute_input":"2024-07-02T06:21:33.481662Z","iopub.status.idle":"2024-07-02T06:21:33.680734Z","shell.execute_reply.started":"2024-07-02T06:21:33.481624Z","shell.execute_reply":"2024-07-02T06:21:33.679219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Binds","metadata":{}},{"cell_type":"code","source":"df[\"binds\"].unique()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:33.683154Z","iopub.execute_input":"2024-07-02T06:21:33.683623Z","iopub.status.idle":"2024-07-02T06:21:33.692536Z","shell.execute_reply.started":"2024-07-02T06:21:33.683585Z","shell.execute_reply":"2024-07-02T06:21:33.691160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.countplot(data=df, x='binds')\nplt.title('Count of Binds')\nplt.xlabel('Binds')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:33.694498Z","iopub.execute_input":"2024-07-02T06:21:33.694835Z","iopub.status.idle":"2024-07-02T06:21:33.937546Z","shell.execute_reply.started":"2024-07-02T06:21:33.694807Z","shell.execute_reply":"2024-07-02T06:21:33.936155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.preprocessing import OneHotEncoder","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:33.939036Z","iopub.execute_input":"2024-07-02T06:21:33.939497Z","iopub.status.idle":"2024-07-02T06:21:33.945744Z","shell.execute_reply.started":"2024-07-02T06:21:33.939458Z","shell.execute_reply":"2024-07-02T06:21:33.944425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CONVERT SMILES TO RDkit MOLECULES\ndf['molecule'] = df['molecule_smiles'].apply(Chem.MolFromSmiles)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:33.947321Z","iopub.execute_input":"2024-07-02T06:21:33.947781Z","iopub.status.idle":"2024-07-02T06:21:34.149548Z","shell.execute_reply.started":"2024-07-02T06:21:33.947740Z","shell.execute_reply":"2024-07-02T06:21:34.148276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate ECFPs\ndef generate_ecfp(molecule, radius=2, bits=1024):\n    if molecule is None:\n        return None\n\n    fpgen = AllChem.GetMorganGenerator(radius=2)\n    fp1 = fpgen.GetFingerprint(molecule,customAtomInvariants=[1]*molecule.GetNumAtoms())\n    \n    return fp1","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.151066Z","iopub.execute_input":"2024-07-02T06:21:34.151533Z","iopub.status.idle":"2024-07-02T06:21:34.160751Z","shell.execute_reply.started":"2024-07-02T06:21:34.151486Z","shell.execute_reply":"2024-07-02T06:21:34.159481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n### ECFP (Extended-Connectivity Fingerprint) is a type of molecular fingerprint used in cheminformatics to represent the structure of a molecule in a way that is suitable for computational analysis. ECFPs are particularly useful for tasks such as similarity searching, clustering, and machine learning applications in drug discovery and materials science.","metadata":{}},{"cell_type":"code","source":"# from rdkit.Chem import rdMolDescriptors\n\n# def generate_ecfp_new(molecule, radius=2, bits=1024):\n#     if molecule is None:\n#         return None\n\n#     # Create a MorganGenerator\n#     morgan_gen = rdMolDescriptors.MorganGenerator(radius=radius, nBits=bits)\n    \n#     # Generate the fingerprint\n#     fp = morgan_gen.GetFingerprint(molecule)\n    \n#     # Convert to bit vector\n#     return list(fp.GetNonzeroElements().keys())","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.163037Z","iopub.execute_input":"2024-07-02T06:21:34.163512Z","iopub.status.idle":"2024-07-02T06:21:34.173219Z","shell.execute_reply.started":"2024-07-02T06:21:34.163470Z","shell.execute_reply":"2024-07-02T06:21:34.171952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit.Chem import Draw\nfrom PIL import Image\nfrom IPython.display import display\n\nsamples = df[\"molecule\"].sample(7)\nfor sample in samples:\n    img = Draw.MolToImage(sample)\n    print(sample)\n    display(img)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.174711Z","iopub.execute_input":"2024-07-02T06:21:34.175147Z","iopub.status.idle":"2024-07-02T06:21:34.355694Z","shell.execute_reply.started":"2024-07-02T06:21:34.175091Z","shell.execute_reply":"2024-07-02T06:21:34.354445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.options.mode.chained_assignment = None  # default='warn'\ndf['ecfp'] = df['molecule'].apply(generate_ecfp)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.357422Z","iopub.execute_input":"2024-07-02T06:21:34.357846Z","iopub.status.idle":"2024-07-02T06:21:34.446080Z","shell.execute_reply.started":"2024-07-02T06:21:34.357806Z","shell.execute_reply":"2024-07-02T06:21:34.444853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['ecfp'].shape","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.447279Z","iopub.execute_input":"2024-07-02T06:21:34.447589Z","iopub.status.idle":"2024-07-02T06:21:34.457602Z","shell.execute_reply.started":"2024-07-02T06:21:34.447563Z","shell.execute_reply":"2024-07-02T06:21:34.455972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# One-hot encode the protein_name\nonehot_encoder = OneHotEncoder(sparse_output=False)\nprotein_numpy_arr = onehot_encoder.fit_transform(df['protein_name'].values.reshape(-1, 1))","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.459035Z","iopub.execute_input":"2024-07-02T06:21:34.459424Z","iopub.status.idle":"2024-07-02T06:21:34.467764Z","shell.execute_reply.started":"2024-07-02T06:21:34.459395Z","shell.execute_reply":"2024-07-02T06:21:34.466587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(protein_numpy_arr)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.469483Z","iopub.execute_input":"2024-07-02T06:21:34.469836Z","iopub.status.idle":"2024-07-02T06:21:34.482420Z","shell.execute_reply.started":"2024-07-02T06:21:34.469807Z","shell.execute_reply":"2024-07-02T06:21:34.481195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"protein_numpy_arr[:5]","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.483794Z","iopub.execute_input":"2024-07-02T06:21:34.484401Z","iopub.status.idle":"2024-07-02T06:21:34.495301Z","shell.execute_reply.started":"2024-07-02T06:21:34.484361Z","shell.execute_reply":"2024-07-02T06:21:34.494005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.497004Z","iopub.execute_input":"2024-07-02T06:21:34.497432Z","iopub.status.idle":"2024-07-02T06:21:34.507187Z","shell.execute_reply.started":"2024-07-02T06:21:34.497375Z","shell.execute_reply":"2024-07-02T06:21:34.505988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.508698Z","iopub.execute_input":"2024-07-02T06:21:34.509216Z","iopub.status.idle":"2024-07-02T06:21:34.534535Z","shell.execute_reply.started":"2024-07-02T06:21:34.509176Z","shell.execute_reply":"2024-07-02T06:21:34.533204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # FIX THIS <rdkit.DataStructs.cDataStructs.ExplicitBitVect to NUMPY CONVERSION\n\n# import rdkit\n\n# samples = df[\"ecfp\"].sample(5)\n# for sample in samples:\n#     print(sample)\n#     bitvect_array = np.zeros((100,), dtype=np.int8)\n#     print(bitvect_array)\n#     output = rdkit.DataStructs.cDataStructs.ConvertToNumpyArray(sample, bitvect_array)\n#     print(output)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.535922Z","iopub.execute_input":"2024-07-02T06:21:34.536281Z","iopub.status.idle":"2024-07-02T06:21:34.544210Z","shell.execute_reply.started":"2024-07-02T06:21:34.536250Z","shell.execute_reply":"2024-07-02T06:21:34.542849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from rdkit import Chem\n# from rdkit.Chem import rdMolDescriptors\n# from rdkit.Chem import DataStructs\n\n# import numpy as np\n\n# mol = Chem.MolFromSmiles(\"CN1C=NC2=C1C(=O)N(C(=O)N2C)C\")\n\n# fp_vec = rdMolDescriptors.GetMorganFingerprintAsBitVect(\n#     mol,\n#     radius=2,\n#     nBits=2048,\n#     useFeatures=True,\n# )\n\n# print(mol)\n\n# print(fp_vec)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.545544Z","iopub.execute_input":"2024-07-02T06:21:34.545961Z","iopub.status.idle":"2024-07-02T06:21:34.554344Z","shell.execute_reply.started":"2024-07-02T06:21:34.545930Z","shell.execute_reply":"2024-07-02T06:21:34.553028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ecfp_protein_concatenated = []\n\n# for index, row in df.iterrows():\n#     ecfp = row[\"ecfp\"]\n#     protein = protein_onehot[index]\n                             \n#     print(ecfp)\n#     print(protein)\n                             \n#     print(\"****************\")","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.555610Z","iopub.execute_input":"2024-07-02T06:21:34.555933Z","iopub.status.idle":"2024-07-02T06:21:34.564283Z","shell.execute_reply.started":"2024-07-02T06:21:34.555902Z","shell.execute_reply":"2024-07-02T06:21:34.563149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"ecfp\"].shape","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.566161Z","iopub.execute_input":"2024-07-02T06:21:34.566563Z","iopub.status.idle":"2024-07-02T06:21:34.582538Z","shell.execute_reply.started":"2024-07-02T06:21:34.566525Z","shell.execute_reply":"2024-07-02T06:21:34.581428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(df[\"ecfp\"])","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.591721Z","iopub.execute_input":"2024-07-02T06:21:34.592162Z","iopub.status.idle":"2024-07-02T06:21:34.600405Z","shell.execute_reply.started":"2024-07-02T06:21:34.592119Z","shell.execute_reply":"2024-07-02T06:21:34.599135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['ecfp'].sample(5)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.602213Z","iopub.execute_input":"2024-07-02T06:21:34.602712Z","iopub.status.idle":"2024-07-02T06:21:34.614963Z","shell.execute_reply.started":"2024-07-02T06:21:34.602673Z","shell.execute_reply":"2024-07-02T06:21:34.613727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ecfp_numpy_arr = df[\"ecfp\"].to_numpy()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.616849Z","iopub.execute_input":"2024-07-02T06:21:34.617250Z","iopub.status.idle":"2024-07-02T06:21:34.626274Z","shell.execute_reply.started":"2024-07-02T06:21:34.617217Z","shell.execute_reply":"2024-07-02T06:21:34.624962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ecfp_numpy_arr[:5]","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.627924Z","iopub.execute_input":"2024-07-02T06:21:34.628360Z","iopub.status.idle":"2024-07-02T06:21:34.639774Z","shell.execute_reply.started":"2024-07-02T06:21:34.628320Z","shell.execute_reply":"2024-07-02T06:21:34.638719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit import DataStructs\n\necfp_lists = [DataStructs.cDataStructs.BitVectToText(bv) for bv in df[\"ecfp\"]]\n\n# Convert lists of integers to numpy array\necfp_array = np.array([list(map(int, fp)) for fp in ecfp_lists])\n\nprint(ecfp_array.shape)  # This will show you the dimensions of your array","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.641329Z","iopub.execute_input":"2024-07-02T06:21:34.641665Z","iopub.status.idle":"2024-07-02T06:21:34.908447Z","shell.execute_reply.started":"2024-07-02T06:21:34.641637Z","shell.execute_reply":"2024-07-02T06:21:34.907124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ecfp_array[:5]","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.909957Z","iopub.execute_input":"2024-07-02T06:21:34.910329Z","iopub.status.idle":"2024-07-02T06:21:34.918582Z","shell.execute_reply.started":"2024-07-02T06:21:34.910299Z","shell.execute_reply":"2024-07-02T06:21:34.917384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(type(ecfp_numpy_arr))\nprint(ecfp_numpy_arr.shape)\nprint(type(protein_numpy_arr))\nprint(protein_numpy_arr.shape)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.919966Z","iopub.execute_input":"2024-07-02T06:21:34.920329Z","iopub.status.idle":"2024-07-02T06:21:34.929387Z","shell.execute_reply.started":"2024-07-02T06:21:34.920300Z","shell.execute_reply":"2024-07-02T06:21:34.928273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ecfp_numpy_arr = ecfp_numpy_arr.reshape(ecfp_numpy_arr.shape[0] , 1)\n# Combine ECFPs and one-hot encoded protein_name\nX = np.concatenate((ecfp_array , protein_numpy_arr) , axis = 1)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.930859Z","iopub.execute_input":"2024-07-02T06:21:34.931316Z","iopub.status.idle":"2024-07-02T06:21:34.946059Z","shell.execute_reply.started":"2024-07-02T06:21:34.931276Z","shell.execute_reply":"2024-07-02T06:21:34.944712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.947861Z","iopub.execute_input":"2024-07-02T06:21:34.948367Z","iopub.status.idle":"2024-07-02T06:21:34.956037Z","shell.execute_reply.started":"2024-07-02T06:21:34.948325Z","shell.execute_reply":"2024-07-02T06:21:34.954904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = df['binds'].to_numpy()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.957468Z","iopub.execute_input":"2024-07-02T06:21:34.957891Z","iopub.status.idle":"2024-07-02T06:21:34.965445Z","shell.execute_reply.started":"2024-07-02T06:21:34.957855Z","shell.execute_reply":"2024-07-02T06:21:34.964200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.966954Z","iopub.execute_input":"2024-07-02T06:21:34.967394Z","iopub.status.idle":"2024-07-02T06:21:34.976734Z","shell.execute_reply.started":"2024-07-02T06:21:34.967357Z","shell.execute_reply":"2024-07-02T06:21:34.975696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.977997Z","iopub.execute_input":"2024-07-02T06:21:34.978381Z","iopub.status.idle":"2024-07-02T06:21:34.987602Z","shell.execute_reply.started":"2024-07-02T06:21:34.978349Z","shell.execute_reply":"2024-07-02T06:21:34.986365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Building (Using ARTIFICIAL NEURAL NETWORK AND KERAS-TUNER) ","metadata":{}},{"cell_type":"code","source":"# Split the data into train and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:34.988885Z","iopub.execute_input":"2024-07-02T06:21:34.989250Z","iopub.status.idle":"2024-07-02T06:21:35.003393Z","shell.execute_reply.started":"2024-07-02T06:21:34.989217Z","shell.execute_reply":"2024-07-02T06:21:35.001995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense\n\n# Verify GPU is available\nif len(tf.config.experimental.list_physical_devices('GPU')) > 0:\n    print(\"GPU is available and will be used for training.\")\nelse:\n    print(\"GPU is not available, using CPU.\")","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:35.004744Z","iopub.execute_input":"2024-07-02T06:21:35.005139Z","iopub.status.idle":"2024-07-02T06:21:35.012744Z","shell.execute_reply.started":"2024-07-02T06:21:35.005083Z","shell.execute_reply":"2024-07-02T06:21:35.011442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install catboost","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:35.014163Z","iopub.execute_input":"2024-07-02T06:21:35.014510Z","iopub.status.idle":"2024-07-02T06:21:47.563556Z","shell.execute_reply.started":"2024-07-02T06:21:35.014481Z","shell.execute_reply":"2024-07-02T06:21:47.562078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostClassifier, CatBoostRegressor, Pool\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, mean_squared_error\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:47.566025Z","iopub.execute_input":"2024-07-02T06:21:47.566446Z","iopub.status.idle":"2024-07-02T06:21:47.572321Z","shell.execute_reply.started":"2024-07-02T06:21:47.566411Z","shell.execute_reply":"2024-07-02T06:21:47.570940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = CatBoostRegressor(iterations=1000, learning_rate=0.1, depth=6, verbose=0)\n\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:47.573844Z","iopub.execute_input":"2024-07-02T06:21:47.574294Z","iopub.status.idle":"2024-07-02T06:21:50.662836Z","shell.execute_reply.started":"2024-07-02T06:21:47.574264Z","shell.execute_reply":"2024-07-02T06:21:50.661708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate R-squared for training data\nr2_train = model.score(X_train, y_train)\nprint(f'R-squared for training data: {r2_train}')\n\n# Calculate R-squared for testing data\nr2_test = model.score(X_test, y_test)\nprint(f'R-squared for testing data: {r2_test}')","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:50.664180Z","iopub.execute_input":"2024-07-02T06:21:50.664508Z","iopub.status.idle":"2024-07-02T06:21:51.145452Z","shell.execute_reply.started":"2024-07-02T06:21:50.664480Z","shell.execute_reply":"2024-07-02T06:21:51.144066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['binds'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:51.147136Z","iopub.execute_input":"2024-07-02T06:21:51.147584Z","iopub.status.idle":"2024-07-02T06:21:51.158263Z","shell.execute_reply.started":"2024-07-02T06:21:51.147544Z","shell.execute_reply":"2024-07-02T06:21:51.156871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.predict(X_train)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:51.159731Z","iopub.execute_input":"2024-07-02T06:21:51.160178Z","iopub.status.idle":"2024-07-02T06:21:51.168282Z","shell.execute_reply.started":"2024-07-02T06:21:51.160128Z","shell.execute_reply":"2024-07-02T06:21:51.166937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test Prediction","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom sklearn.preprocessing import OneHotEncoder\nfrom rdkit import Chem, DataStructs\n\ntest_file = '/kaggle/input/leash-BELKA/test.csv'\noutput_file = 'submission.csv'\n\nfor i, df_test in enumerate(pd.read_csv(test_file, chunksize=100)):\n    print(f\"\\n{'='*10} Processing Chunk {i + 1} {'='*10}\\n\")\n    \n    # Suppress chained assignment warning\n    pd.options.mode.chained_assignment = None\n    \n    # Convert SMILES to RDkit molecules\n#     print(f\"Step 1: Converting SMILES to RDkit molecules for Chunk {i + 1}...\")\n    df_test['molecule'] = df_test['molecule_smiles'].apply(Chem.MolFromSmiles)\n#     print(\"Converted SMILES to RDkit molecules.\")\n    \n#     print(df_test[\"molecule\"].sample(3), '\\n')\n    \n    # Generate ECFPs\n#     print(f\"Step 2: Generating ECFPs for Chunk {i + 1}...\")\n    df_test['ecfp'] = df_test['molecule'].apply(generate_ecfp)\n#     print(\"ECFPs generated.\")\n    \n#     print(df_test[\"ecfp\"].sample(3), '\\n')\n    \n    # One-hot encode the protein_name\n#     print(f\"Step 3: One-hot encoding protein names for Chunk {i + 1}...\")\n    onehot_encoder = OneHotEncoder(sparse_output=False)\n    protein_numpy_arr = onehot_encoder.fit_transform(df_test['protein_name'].values.reshape(-1, 1))\n#     print(\"Protein names one-hot encoded.\")\n    \n#     print(protein_numpy_arr[:3], '\\n')\n    \n    # Convert lists of integers to numpy array\n#     print(f\"Step 4: Converting ECFP lists to numpy array for Chunk {i + 1}...\")\n    ecfp_lists = [DataStructs.cDataStructs.BitVectToText(bv) for bv in df_test[\"ecfp\"]]\n    ecfp_array = np.array([list(map(int, fp)) for fp in ecfp_lists])\n#     print(\"ECFP lists converted to numpy array.\")\n    \n#     print(ecfp_array[:3], '\\n')\n    \n    # Combine ECFPs and one-hot encoded protein names\n#     print(f\"Step 5: Combining ECFPs and one-hot encoded protein names for Chunk {i + 1}...\")\n    X_test = np.concatenate((ecfp_array, protein_numpy_arr), axis=1)\n#     print(\"Combined ECFPs and one-hot encoded protein names.\")\n    \n    # Predict the probabilities\n#     print(f\"Step 6: Predicting probabilities for Chunk {i + 1}...\")\n    probabilities = model.predict(X_test)\n#     print(\"Predicted probabilities.\")\n    \n    # Create a DataFrame with 'id' and 'binds' columns\n#     print(f\"Step 7: Creating output DataFrame for Chunk {i + 1}...\")\n    output_df = pd.DataFrame({'id': df_test['id'], 'binds': probabilities})\n#     print(\"Output DataFrame created.\")\n    \n    # Append the results to the output file\n    if i == 0:\n#         print(f\"Step 8: Writing initial Chunk {i + 1} to {output_file}...\")\n        output_df.to_csv(output_file, index=False, mode='w')\n#         print(f\"Initial Chunk {i + 1} written to {output_file}.\")\n    else:\n#         print(f\"Step 8: Appending Chunk {i + 1} to {output_file}...\")\n        output_df.to_csv(output_file, index=False, mode='a', header=False)\n#         print(f\"Chunk {i + 1} appended to {output_file}.\")\n    \n#     print(f\"\\n{'='*10} Completed Chunk {i + 1} {'='*10}\\n\")\n\nprint(\"\\nProcessing complete.\")","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:21:51.169831Z","iopub.execute_input":"2024-07-02T06:21:51.170297Z","iopub.status.idle":"2024-07-02T06:24:26.437947Z","shell.execute_reply.started":"2024-07-02T06:21:51.170263Z","shell.execute_reply":"2024-07-02T06:24:26.436202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probabilities","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:24:26.439281Z","iopub.status.idle":"2024-07-02T06:24:26.440472Z","shell.execute_reply.started":"2024-07-02T06:24:26.440164Z","shell.execute_reply":"2024-07-02T06:24:26.440193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_df = pd.DataFrame({'id': df_test['id'], 'binds': probabilities})","metadata":{"execution":{"iopub.status.busy":"2024-07-02T06:24:26.442488Z","iopub.status.idle":"2024-07-02T06:24:26.443055Z","shell.execute_reply.started":"2024-07-02T06:24:26.442770Z","shell.execute_reply":"2024-07-02T06:24:26.442794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}