{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as mplpp\nimport seaborn as sb\nimport pyarrow.parquet as pq\nimport pyarrow as pa\nfrom collections import Counter\nimport time\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-20T16:50:30.268614Z","iopub.execute_input":"2024-05-20T16:50:30.269360Z","iopub.status.idle":"2024-05-20T16:50:31.197768Z","shell.execute_reply.started":"2024-05-20T16:50:30.269324Z","shell.execute_reply":"2024-05-20T16:50:31.196557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**I have shown in this notebook a very simple yet powerful approach to process the whole training data. Here I have performed some EDA tasks but I believe other may also be performed **","metadata":{}},{"cell_type":"code","source":"#defining various variables\nBRD4_count=0\nHSA_count=0\nsEH_count=0\n# Define the columns to read from the CSV file\ncols_to_read = ['molecule_smiles', 'protein_name', 'binds']\nchunk_size =10000000  # may be adjusted as needed\nchunk_no=1\nchunks_to_read=5 # may be adjusted as needed \nbase_path='/kaggle/input/leash-BELKA/'\nbinds_1=0\nbinds_0=0\n# Initialize a Counter to keep track of protein counts\nprotein_counter = Counter()\n# Initialize total_counts as an empty Series\ntotal_counts = pd.Series()\n# defining dictionary to store counts for each protein where binds==1\nbinds_1_protein_counts = {'BRD4': 0, 'HSA': 0, 'sEH': 0}\nbinds_0_protein_counts = {'BRD4': 0, 'HSA': 0, 'sEH': 0}\n#objects to count proteins\nBRD4_counts=0\nHSA_counts=0\nsEH_counts=0\nsmiles_binds_0=0\nsmiles_binds_1=0","metadata":{"execution":{"iopub.status.busy":"2024-05-20T16:50:31.199591Z","iopub.execute_input":"2024-05-20T16:50:31.200065Z","iopub.status.idle":"2024-05-20T16:50:31.207895Z","shell.execute_reply.started":"2024-05-20T16:50:31.200030Z","shell.execute_reply":"2024-05-20T16:50:31.206814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\ndef count_protein_occurance(name):\n    if name == 'BRD4':\n        BRD4_count+=1\n    elif name == 'HSA':\n        HSA_count+=1\n    elif name == 'sEH':\n        sEH_count+=1   \n'''","metadata":{"execution":{"iopub.status.busy":"2024-05-20T16:50:31.209238Z","iopub.execute_input":"2024-05-20T16:50:31.209612Z","iopub.status.idle":"2024-05-20T16:50:31.224488Z","shell.execute_reply.started":"2024-05-20T16:50:31.209578Z","shell.execute_reply":"2024-05-20T16:50:31.223316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute summary statistics for binds_0 and binds_1\ndef summary_statistics(binds_0, binds_1):\n    mean_0 = np.mean(binds_0)\n    median_0 = np.median(binds_0)\n    std_0 = np.std(binds_0)\n    min_0 = np.min(binds_0)\n    max_0 = np.max(binds_0)\n    \n    mean_1 = np.mean(binds_1)\n    median_1 = np.median(binds_1)\n    std_1 = np.std(binds_1)\n    min_1 = np.min(binds_1)\n    max_1 = np.max(binds_1)\n    \n    \n    # Print summary statistics\n    print(\"Summary Statistics for binds_0:\")\n    print(\"Mean:\", mean_0)\n    print(\"Median:\", median_0)\n    print(\"Standard Deviation:\", std_0)\n    print(\"Minimum:\", min_0)\n    print(\"Maximum:\", max_0)\n    print()\n\n    print(\"Summary Statistics for binds_1:\")\n    print(\"Mean:\", mean_1)\n    print(\"Median:\", median_1)\n    print(\"Standard Deviation:\", std_1)\n    print(\"Minimum:\", min_1)\n    print(\"Maximum:\", max_1)","metadata":{"execution":{"iopub.status.busy":"2024-05-20T16:50:31.227300Z","iopub.execute_input":"2024-05-20T16:50:31.227712Z","iopub.status.idle":"2024-05-20T16:50:31.236374Z","shell.execute_reply.started":"2024-05-20T16:50:31.227678Z","shell.execute_reply":"2024-05-20T16:50:31.235363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# correlation analysis between binds_0 and binds_1\ndef coorelation_analysis(binds_0, binds_1):\n    correlation_coefficient = np.corrcoef(binds_0, binds_1)[0, 1]\n\n    print(\"Correlation Coefficient between binds_0 and binds_1:\", correlation_coefficient)","metadata":{"execution":{"iopub.status.busy":"2024-05-20T16:50:31.240046Z","iopub.execute_input":"2024-05-20T16:50:31.240760Z","iopub.status.idle":"2024-05-20T16:50:31.254740Z","shell.execute_reply.started":"2024-05-20T16:50:31.240716Z","shell.execute_reply":"2024-05-20T16:50:31.253330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nParquetFile doesn't load the data into memory. Instead, it creates a handle \nor representation of the Parquet file, allowing you to perform various \noperations on it without necessarily loading the entire dataset into memory \nat once.\n\n'''\npfile = pq.ParquetFile(base_path + 'train.parquet')\n#These columns will be deleted to save memory\ncols_to_delete = ['id', 'buildingblock1_smiles', 'buildingblock2_smiles','buildingblock3_smiles']\n#For loop reads data from the parquet file in the specified chunks. Means it \n#does not load whole data in memory. Instead it will return only a batch or chunk\nfor chunk in pfile.iter_batches(batch_size=chunk_size):\n    df = chunk.to_pandas() #convert chunk in dataframe \n    df.drop(columns=cols_to_delete, inplace=True) #delete unwanted columns\n    gc.collect() #release memory occupied by the deleted columns\n    #print(chunk)\n   \n    binds_1+=df['binds'].sum()\n    binds_0+=(len(df) - binds_1)\n    \n    # counting number of occurance of each protein \n    BRD4_counts+=len(df[df['protein_name']=='BRD4'])\n    HSA_counts+=len(df[df['protein_name']=='HSA'])\n    sEH_counts+=len(df[df['protein_name']=='sEH'])\n   \n    \n    # Identifying proteins where binds == 1 \n    counts_1 = df[df['binds'] == 1]['protein_name'].value_counts()\n    counts_0 = df[df['binds'] == 0]['protein_name'].value_counts()\n    # Update total counts for each protein\n    for protein, count in counts_1.items():\n        binds_1_protein_counts[protein] += count\n    # Update total counts for each protein\n    for protein, count in counts_0.items():\n        binds_0_protein_counts[protein] += count \n        \n    \n    #if chunks_to_read == 0:\n     #   break\n    #chunks_to_read-=1\n#molecules with zero binds\nbinds_0=abs(binds_0)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-20T16:50:31.255991Z","iopub.execute_input":"2024-05-20T16:50:31.256339Z","iopub.status.idle":"2024-05-20T16:56:39.716529Z","shell.execute_reply.started":"2024-05-20T16:50:31.256308Z","shell.execute_reply":"2024-05-20T16:56:39.714972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Visualizing occurance of each protein**","metadata":{}},{"cell_type":"code","source":"print('Occurance of each protein:')\nprint('BRD4: ', BRD4_counts )\nprint('HSA: ', HSA_counts )\nprint('sEH: ', sEH_counts )\n   \nimport matplotlib.pyplot as plt\n\n# Data\nlabels = ['BRD4', 'HSA', 'sEH']\nvalues = [BRD4_counts, HSA_counts, sEH_counts]\n\n# Create a bar graph\nplt.figure(figsize=(8, 6))\nplt.bar(labels, values, color=['blue', 'green', 'red'])\n\n# Adding title and labels\nplt.title('Analyzing occurance of each protein')\nplt.xlabel('Proteins')\nplt.ylabel('Occurance')\n\n# Show the graph\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-20T16:56:39.718317Z","iopub.execute_input":"2024-05-20T16:56:39.718701Z","iopub.status.idle":"2024-05-20T16:56:39.961508Z","shell.execute_reply.started":"2024-05-20T16:56:39.718668Z","shell.execute_reply":"2024-05-20T16:56:39.960272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Analyzing number of molecules with binding affinity **","metadata":{}},{"cell_type":"code","source":"# Number of SMILES molecules with binding affinity and without binding affinity \nprint('binds_1: ',binds_1)\nprint('binds_0: ',binds_0)\ntotal_records=binds_0 + binds_1\nprint('total number of records:', total_records)\n\n# Data\nlabels = ['Molecules with binds 1', 'Molecules with binds 0', 'Total molecules']\nvalues = [binds_1, binds_0, total_records]\n\n# Create a bar graph\nplt.figure(figsize=(8, 6))\nplt.bar(labels, values, color=['Yellow', 'green', 'black'])\n\n# Adding title and labels\nplt.title('Analyzing molecules with binding affinity')\nplt.xlabel('Molecules')\nplt.ylabel('Occurance')\n\n# Show the graph\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-20T16:56:39.962873Z","iopub.execute_input":"2024-05-20T16:56:39.963214Z","iopub.status.idle":"2024-05-20T16:56:40.205653Z","shell.execute_reply.started":"2024-05-20T16:56:39.963183Z","shell.execute_reply":"2024-05-20T16:56:40.204257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extracting keys and values from dictionaries\ncategories = list(binds_0_protein_counts.keys())\nvalues_0 = list(binds_0_protein_counts.values())\nvalues_1 = list(binds_1_protein_counts.values())\n\n# Plotting\nbar_width = 0.35\nindex = range(len(categories))\n\nplt.bar(index, values_0, bar_width, label='molecules with binds 0')\nplt.bar([i + bar_width for i in index], values_1, bar_width, label='molecules with binds 1')\n\nplt.xlabel('Categories')\nplt.ylabel('Counts')\nplt.title('Comparison of Counts between Dict 1 and Dict 2')\nplt.xticks([i + bar_width/2 for i in index], categories)\nplt.legend()\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-20T16:56:40.207072Z","iopub.execute_input":"2024-05-20T16:56:40.207407Z","iopub.status.idle":"2024-05-20T16:56:40.471934Z","shell.execute_reply.started":"2024-05-20T16:56:40.207372Z","shell.execute_reply":"2024-05-20T16:56:40.470650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Summary statistics of SMILES molecules with binding affinity and without binding affinity**","metadata":{}},{"cell_type":"code","source":" \nsummary_statistics(binds_0, binds_1)","metadata":{"execution":{"iopub.status.busy":"2024-05-20T16:56:40.474775Z","iopub.execute_input":"2024-05-20T16:56:40.475160Z","iopub.status.idle":"2024-05-20T16:56:40.481771Z","shell.execute_reply.started":"2024-05-20T16:56:40.475127Z","shell.execute_reply":"2024-05-20T16:56:40.480588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Following is another way to read complete training data in chunks but it is slow as compared PyArrow,\n#so We have commneted this code. You can visit following notebook for analysis of PyArrow, DuckDB and Pandas\n\nhttps://www.kaggle.com/code/tariqcp/pyarrow-vs-duckdb-and-pandas-exploring-pyarrow\n\n'''\nimport pandas as pd\n\n\n# Read the CSV file in chunks\nfor chunk in pd.read_csv(base_path+'train.csv', usecols=cols_to_read, chunksize=chunksize):\n    #print(chunk)\n    binds_1+=chunk['binds'].sum()\n    binds_0+=len(chunk) - binds_1\n    # Replace protein names\n    total_counts = count_proteins(chunk['protein_name'], total_counts)\n    # Identifying proteins where binds == 1 \n    counts = chunk[chunk['binds'] == 1]['protein_name'].value_counts()\n\n    # Update total counts for each protein\n    for protein, count in counts.items():\n        protein_counts_per_binds_1[protein] += count\n    #chunk_no+=1\n    \n    #uncomment this code if you want to process specific number of records. \n    #if chunks_to_read == 0:\n     #   break\n    #chunks_to_read-=1\n    \n    \n    \nprint('binds_1: ',binds_1)\nprint('binds_0: ',binds_0)\nsummary_statistics(binds_0, binds_1)\ncoorelation_analysis(binds_0, binds_1)\nprint('protein occurances------- ')\nprint('total_counts: ',total_counts)\n# Display the total counts of proteins where binds == 1\nprint(\"Total counts of proteins where binds == 1:\")\nprint(protein_counts_per_binds_1)\n'''","metadata":{"execution":{"iopub.status.busy":"2024-05-20T16:56:40.483149Z","iopub.execute_input":"2024-05-20T16:56:40.483494Z","iopub.status.idle":"2024-05-20T16:56:40.494294Z","shell.execute_reply.started":"2024-05-20T16:56:40.483465Z","shell.execute_reply":"2024-05-20T16:56:40.492701Z"},"trusted":true},"execution_count":null,"outputs":[]}]}