{"metadata":{"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Baseline model - Logistic Regression","metadata":{}},{"cell_type":"markdown","source":"This notebook presents a baseline logistic regression model for predicting the binding affinity of small molecules to specific protein targets, as part of the BELKA competition on Kaggle. We tackle the challenge of large dataset management and model training using Dask for efficient data handling and the stochastic gradient descent (SGD) classifier for incremental learning.","metadata":{}},{"cell_type":"code","source":"import dask.dataframe as dd\nfrom dask.diagnostics import ProgressBar\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nfrom sklearn.linear_model import LogisticRegression, SGDClassifier\nfrom sklearn.metrics import log_loss\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OneHotEncoder ","metadata":{"execution":{"iopub.status.busy":"2024-10-30T10:00:37.757171Z","iopub.execute_input":"2024-10-30T10:00:37.758302Z","iopub.status.idle":"2024-10-30T10:00:41.326899Z","shell.execute_reply.started":"2024-10-30T10:00:37.758250Z","shell.execute_reply":"2024-10-30T10:00:41.325807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The initial training will be on csv data only","metadata":{}},{"cell_type":"markdown","source":">Note: The training data in csv is to big to load in normal Kaggle environment so:\n```\ntrain_df = pd.read_csv(f\"{input_path}/train.csv\")\n```\nwill not work.<br>\nFor this reason the `reduce_memory_usage()` function can be proposed as follows, although similar can be achieved with dask","metadata":{}},{"cell_type":"code","source":"def reduce_memory_usage(df):\n    \"\"\" Iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-10-30T10:00:41.328625Z","iopub.execute_input":"2024-10-30T10:00:41.329417Z","iopub.status.idle":"2024-10-30T10:00:41.341803Z","shell.execute_reply.started":"2024-10-30T10:00:41.329384Z","shell.execute_reply":"2024-10-30T10:00:41.340633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The data can now be used with dask","metadata":{}},{"cell_type":"code","source":"%%time\ninput_path = \"/kaggle/input/leash-BELKA\"\n\ntrain_df = dd.read_csv(f\"{input_path}/train.csv\", assume_missing=True)\ntest_df = dd.read_csv(f\"{input_path}/test.csv\", assume_missing=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-30T10:00:41.343359Z","iopub.execute_input":"2024-10-30T10:00:41.343727Z","iopub.status.idle":"2024-10-30T10:00:41.481325Z","shell.execute_reply.started":"2024-10-30T10:00:41.343700Z","shell.execute_reply":"2024-10-30T10:00:41.480155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EDA (Exploratory Data Analysis)\n\nAs the data set is to big to load as `pandas.DataFrame` the EDA is performed with `dask`","metadata":{}},{"cell_type":"code","source":"%%time\n# Function to get basic statistics for numeric columns\ndef compute_summary_statistics(df):\n    # Computes summary statistics for numerical columns\n    with ProgressBar():\n        numerical_stats = df.describe().compute()\n    return numerical_stats\n\n# Example usage\ntrain_summary_stats = compute_summary_statistics(train_df.select_dtypes(include=[np.number]))\nprint(\"Train Summary Statistics:\\n\", train_summary_stats)\n\n# Function to get value counts for categorical columns\ndef compute_value_counts(df, column):\n    # Computes value counts for a specified column\n    with ProgressBar():\n        value_counts = df[column].value_counts().compute()\n    return value_counts\n\n# Example usage for categorical column 'molecule_smiles'\nsmiles_counts = compute_value_counts(train_df, 'molecule_smiles')\nprint(\"Value Counts for 'molecule_smiles':\\n\", smiles_counts)\n\n# Function to compute correlation matrix\ndef compute_correlations(df):\n    # Compute correlation matrix for all numerical columns\n    with ProgressBar():\n        correlation_matrix = df.corr().compute()\n    return correlation_matrix\n\n# Example usage\ncorrelations = compute_correlations(train_df.select_dtypes(include=[np.number]))\nprint(\"Correlation Matrix:\\n\", correlations)\n\n# Function to check missing values\ndef compute_missing_values_table(df):\n    with ProgressBar():\n        result = df.isnull().sum().compute()\n    result = pd.DataFrame(result, columns=['Missing Values'])\n    result['% of Total Values'] = (result['Missing Values'] / len(df) * 100)\n    return result\n\nmissing_values = compute_missing_values_table(train_df)\nprint(\"Missing Value Counts:\\n\", missing_values)\n\ndef plot_histogram(df, column):\n    # Plot histogram for a specified column\n    with ProgressBar():\n        data = df[column].compute()  # Convert to pandas DataFrame for plotting\n    plt.figure(figsize=(10, 6))\n    data.hist(bins=30)\n    plt.title(f'Histogram of {column}')\n    plt.xlabel(column)\n    plt.ylabel('Frequency')\n    plt.grid(False)\n    plt.show()\n\ncolumns = train_df.columns\nprint(\"Columns:\", columns)\n\n\nfor column in columns:\n    try:\n        if column != 'id' and np.issubdtype(dtypes[column], np.number):\n            plot_histogram(train_df, column)\n    except Exception as e:\n        print(f\"EXCEPTION: {e}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-30T10:00:41.483570Z","iopub.execute_input":"2024-10-30T10:00:41.483962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train simple model in chunks\n\nFor the reason mentioned, the model can be trained in chunks","metadata":{}},{"cell_type":"code","source":"%%time\nencoder = OneHotEncoder(handle_unknown='ignore')\n\nchunksize = 100000  # can be adjusted based on memory capacity\nsample_size = 500000  # This should be large enough to cover most of the variability\nsample_data = pd.concat([chunk for chunk in pd.read_csv(f\"{input_path}/train.csv\", chunksize=chunksize, nrows=sample_size)])\nsample_data = sample_data.dropna(subset=['molecule_smiles'])\n\n# Fit the encoder on the sampled data\nencoder.fit(sample_data[['molecule_smiles']])\n\nmodel = SGDClassifier(loss='log_loss', max_iter=1000)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Althouhgh the model is simply, size consideration is required","metadata":{}},{"cell_type":"code","source":"%%time\nchunk_sizes = []\ndropped_na_counts = []\nfor chunk in pd.read_csv(f\"{input_path}/train.csv\", chunksize=chunksize):\n    # There are Nan in the dataset so they need to be dropped\n    original_size = len(chunk)\n    chunk = chunk.dropna(subset=['binds', 'molecule_smiles'])\n    cleaned_size = len(chunk)\n    dropped_na_counts.append(original_size - cleaned_size)\n    chunk_sizes.append(cleaned_size)\n    \n    X_train = encoder.transform(chunk[['molecule_smiles']])\n    y_train = chunk['binds'].values\n    # Ensure classes are explicitly,might not encounter all classes in initial chunks\n    model.partial_fit(X_train, y_train, classes=[0, 1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nplt.figure(figsize=(14, 8))\n\n# Plotting the sizes of chunks processed\nplt.subplot(2, 1, 1)\nplt.bar(range(len(chunk_sizes)), chunk_sizes, color='blue')\nplt.title('Processed Data Chunks')\nplt.xlabel('Chunk Number')\nplt.ylabel('Number of Samples After Dropping NaN')\n\n# Plotting the number of dropped NaN values per chunk\nplt.subplot(2, 1, 2)\nplt.bar(range(len(dropped_na_counts)), dropped_na_counts, color='red')\nplt.title('NaN Values Dropped Per Chunk')\nplt.xlabel('Chunk Number')\nplt.ylabel('Number of Dropped NaN Samples')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npredictions = []\nids = []\n\n# Read the test data in chunks and process each chunk\nfor chunk in pd.read_csv(f\"{input_path}/test.csv\", chunksize=chunksize):\n    # Transform the SMILES data into features\n    X_test = encoder.transform(chunk[['molecule_smiles']])\n    # Predict probabilities for the current chunk\n    chunk_predictions = model.predict_proba(X_test)[:, 1]\n    # Append predictions to the list\n    predictions.extend(chunk_predictions)\n    # Collect IDs from the chunk\n    ids.extend(chunk['id'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The submission in accepted format","metadata":{}},{"cell_type":"code","source":"%%time\nsubmission_df = pd.DataFrame({\n    'id': ids,\n    'binds': predictions\n})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv('sample_submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}